diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index 82cd6af..6fb81bb 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -2,13 +2,14 @@ name: CI on: push: - branches: [main] + branches: [main, dev] pull_request: + workflow_dispatch: jobs: lint: runs-on: ubuntu-latest - container: gitea.tap-tap.win/taptap/fastsync-ci:v9 + container: gitea.tap-tap.win/taptap/fastsync-ci:v10 steps: - name: Checkout uses: actions/checkout@v4 @@ -19,9 +20,13 @@ jobs: - name: cppcheck run: cppcheck --enable=warning,style,performance,portability --suppress=missingIncludeSystem --error-exitcode=1 --inline-suppr src/ tests/ + # Fast PR gate: build + unit tests + a representative subset of integration + # tests (marked `ci`), parallelized with pytest-xdist. Only the full coverage + # jobs below (sanitizers/fuzz/coverage/valgrind and the FULL integration + # suite) run on merge to dev/main, so PR CI stays well under ~3 minutes. build-and-test: runs-on: ubuntu-latest - container: gitea.tap-tap.win/taptap/fastsync-ci:v9 + container: gitea.tap-tap.win/taptap/fastsync-ci:v10 needs: lint steps: - name: Checkout @@ -36,13 +41,19 @@ jobs: - name: Unit Tests run: ctest --test-dir build --output-on-failure -j$(nproc) - - name: Integration Tests - run: python3 -m pytest tests/integration/ -v --tb=short + - name: Integration Tests (PR smoke subset) + if: github.event_name == 'pull_request' + run: python3 -m pytest tests/integration/ -n 4 --dist=load -m ci --durations=25 --tb=short -q + + - name: Integration Tests (full suite) + if: github.event_name == 'push' + run: python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" --durations=25 --tb=short -q sanitizers: runs-on: ubuntu-latest - container: gitea.tap-tap.win/taptap/fastsync-ci:v9 + container: gitea.tap-tap.win/taptap/fastsync-ci:v10 needs: lint + if: github.event_name == 'push' strategy: matrix: sanitizer: [address, undefined] @@ -61,8 +72,9 @@ jobs: fuzz-build: runs-on: ubuntu-latest - container: gitea.tap-tap.win/taptap/fastsync-ci:v9 + container: gitea.tap-tap.win/taptap/fastsync-ci:v10 needs: lint + if: github.event_name == 'push' steps: - name: Checkout uses: actions/checkout@v4 @@ -73,10 +85,18 @@ jobs: - name: Build fuzz targets run: cmake --build build-fuzz -j$(nproc) + - name: Smoke fuzz targets + run: | + for target in build-fuzz/fuzz_*; do + [ -x "$target" ] || continue + timeout 10s "$target" -runs=100 -max_total_time=5 + done + coverage: runs-on: ubuntu-latest - container: gitea.tap-tap.win/taptap/fastsync-ci:v9 + container: gitea.tap-tap.win/taptap/fastsync-ci:v10 needs: lint + if: github.event_name == 'push' steps: - name: Checkout uses: actions/checkout@v4 @@ -98,8 +118,9 @@ jobs: valgrind: runs-on: ubuntu-latest - container: gitea.tap-tap.win/taptap/fastsync-ci:v9 + container: gitea.tap-tap.win/taptap/fastsync-ci:v10 needs: lint + if: github.event_name == 'push' steps: - name: Checkout uses: actions/checkout@v4 diff --git a/AGENTS.md b/AGENTS.md index fd9884a..b2293ae 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -4,28 +4,28 @@ FastSync is a high-performance file synchronization system written in C11. It su ## Dependency installation -**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. The image is built from the repo-root `Dockerfile` and is the same image CI uses: `gitea.tap-tap.win/taptap/fastsync-ci:v7`. It contains the full toolchain: gcc/g++, CMake, libzstd-dev, libssl-dev, make, git, cppcheck, clang-format, python3 + pytest, openssh-client, and Node.js. +**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. The image is built from the repo-root `Dockerfile` and is the same image CI uses: `gitea.tap-tap.win/taptap/fastsync-ci:v10`. It contains the full toolchain: gcc/g++, CMake, libzstd-dev, libssl-dev, make, git, cppcheck, clang-format, python3 + pytest + pytest-xdist, openssh-client, and Node.js. **Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. The Docker image can also be used locally for CI parity. ```bash # Use the prebuilt CI image directly (faster, guaranteed CI parity) -docker pull gitea.tap-tap.win/taptap/fastsync-ci:v7 -docker tag gitea.tap-tap.win/taptap/fastsync-ci:v7 fastsync-ci:local +docker pull gitea.tap-tap.win/taptap/fastsync-ci:v10 +docker tag gitea.tap-tap.win/taptap/fastsync-ci:v10 fastsync-ci:local # Or build the image from the repo-root Dockerfile -# (Note: the prebuilt :v7 image reflects the previous Dockerfile state; +# (Note: the prebuilt :v10 image reflects the previous Dockerfile state; # rebuild from source to pick up any newly added packages like lcov/valgrind.) docker build -t fastsync-ci:local . # Build, run unit tests, and run integration tests inside the container docker run --rm -v "$PWD:/workspace" -w /workspace fastsync-ci:local \ - sh -c 'cmake -B build -S . && cmake --build build -j$(nproc) && ./build/tests && python3 -m pytest tests/' + sh -c 'cmake -B build -S . && cmake --build build -j$(nproc) && ./build/tests && python3 -m pytest tests/integration/ -n 4 --dist=load' # Avoid root-owned build/ artifacts by matching your host UID/GID docker run --rm --user "$(id -u):$(id -g)" -v "$PWD:/workspace" \ -w /workspace fastsync-ci:local \ - sh -c 'cmake -B build -S . && cmake --build build -j$(nproc) && ./build/tests && python3 -m pytest tests/' + sh -c 'cmake -B build -S . && cmake --build build -j$(nproc) && ./build/tests && python3 -m pytest tests/integration/ -n 4 --dist=load' ``` > **Note:** The first `cmake configure` (`cmake -B build -S .`) fetches xxHash from GitHub via `FetchContent` — network access is required. Subsequent reconfigures reuse the cached source. @@ -41,7 +41,7 @@ cmake -B build -S . -DSANITIZER=address # AddressSanitizer (ASan) cmake -B build -S . -DSANITIZER=thread # ThreadSanitizer (TSan) ``` -The CI workflow (`.gitea/workflows/ci.yaml`) runs lint (clang-format, cppcheck), build + test (unit + integration), and sanitizer (currently only `address`) jobs sequentially. +The CI workflow (`.gitea/workflows/ci.yaml`) runs lint (clang-format, cppcheck), then a **fast PR gate** — build + unit + a representative subset of integration tests marked `@pytest.mark.ci`, parallelized with pytest-xdist (`-n 4 --dist=load`). The full coverage jobs (full integration suite as `-m "not setpriv"`, sanitizer, fuzz, coverage, valgrind) run **only on push to `dev`/`main`**; pull requests skip them to keep PR CI under ~3 minutes. The two `setpriv` privilege tests are excluded from CI via a marker because their result depends on the runner/container uid and host mount permissions. ## Build @@ -53,68 +53,27 @@ cmake -B build -S . && cmake --build build -j$(nproc) ```bash ./build/tests # unit tests -python3 -m pytest tests/ # integration tests +python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" # full integration suite (CI excludes env-dependent privilege tests) +python3 -m pytest tests/integration/ -n 4 --dist=load -m ci # PR-gate subset only ``` ## CI Workflow — Waiting for Results When running the CI workflow via `tea` (the task execution agent), always set a sufficient timeout (e.g., 600000ms) to allow CI to finish. After CI completes, check the results yourself — do not assume success. Use `gh run watch` or similar to monitor CI status, then inspect logs on failure. -## Branch Strategy - -Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch before making changes: -```bash -git checkout -b -``` -After committing changes, push the branch and create a PR: -```bash -git push -u origin -gh pr create --fill -``` -Wait for CI to pass on the PR before merging. - -## Batch PR Workflow - -When handling multiple issues split across several PRs that target the same files: - -1. **Group issues by logical category** into separate PR branches (e.g., memory-safety, refactoring, test-coverage). -2. **Fix and push** each branch independently. Let CI run on each PR. -3. **Run all 3 reviewer types** on each PR and post results to Gitea via `tea pr approve/reject` or the Gitea API: - - `reviewer` — general code correctness - - `code-quality-guardian` — code quality, duplication, complexity - - `security-auditor` — vulnerability assessment -4. **Iterate**: if any reviewer requests changes, fix, push, re-review. Repeat until all 3 approve. -5. **Merge approved PRs** one at a time into `main`. -6. **Create a combined merge branch** for the remaining PRs that conflict with the new `main`: - ```bash - git checkout -b merge-all origin/main - for branch in branch1 branch2 branch3; do - git merge origin/$branch --no-edit || true - # Resolve conflicts, build, test - done - ``` -7. **Run review again** on the combined branch. Fix issues, push, re-review until approved. -8. **Merge** the combined PR, **close** the redundant individual PRs, and **close all resolved issues** via the Gitea API: - ```bash - curl -s -X PATCH -H "Authorization: token $TOKEN" \ - -H "Content-Type: application/json" \ - -d '{"state":"closed"}' \ - "https://gitea.tap-tap.win/api/v1/repos/owner/repo/issues/" - ``` - ## CI Troubleshooting ### If lint (clang-format) fails Run clang-format in the CI Docker image to match the exact CI version: ```bash -docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v9 \ +docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v10 \ sh -c 'find src/ tests/ -name "*.c" -o -name "*.h" | xargs clang-format -i' ``` ### If cppcheck fails Fix reported issues locally, then verify with: ```bash -docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v9 \ +docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v10 \ sh -c 'cppcheck --enable=warning,style,performance,portability --suppress=missingIncludeSystem --error-exitcode=1 --inline-suppr src/ tests/' ``` @@ -124,6 +83,99 @@ Run locally before pushing: python3 -m pytest tests/ -v --tb=short ``` +## Branch Strategy + +Two main branches: `dev` (integration) and `main` (stable releases). + +### Rules +- **All PRs target `dev`** — never target `main` directly +- **`dev` is the default branch** in Gitea repo settings +- **`main` is protected** — only merged from `dev` via PR with 2 approvals + full CI pass +- **Feature/bug branches** branch from `dev`, PR back to `dev` +- **`dev` → `main` merges** happen on-demand or weekly, requiring full CI + review + +```bash +# Start a new feature +git checkout dev && git pull +git checkout -b feat/my-feature +# ... work, commit, push +git push -u origin feat/my-feature +# Create PR targeting dev +``` + +### Creating the `dev` branch (one-time setup) +```bash +git checkout main && git pull +git checkout -b dev +git push origin dev +# Then in Gitea: Settings → Repository → Default Branch → dev +``` + +### Branch protection (Gitea repo settings) +**For `dev`:** +- ✅ Require PR for merging +- ✅ Require 1 approval +- ✅ Require status checks (all CI jobs must pass) +- ✅ Delete branch after merge + +**For `main`:** +- ✅ Require PR from `dev` only +- ✅ Require CI +- ✅ Require 2 approvals +- ✅ No direct pushes + +## Automated Agent Workflows + +All agents run locally via the opencode CLI. There is no CI-based agent automation — agents are invoked on-demand by the developer or by this assistant. + +### One-command batch workflow + +For fixing a set of issues and creating one integration PR: + +```bash +# 1. Run each subagent on its category +opencode run --agent security-auditor "Fix all open security issues" +opencode run --agent debugger "Fix all open bugs" +opencode run --agent test-writer "Add missing test coverage" + +# 2. The assistant handles: merging branches, fixing CI failures, +# pushing, creating the integration PR, waiting for CI, iterating. +# The developer only reviews the final PR. +``` + +### Issue triage loop +When you want to fix a batch of issues autonomously: + +1. Tell the assistant: *"Fix all open issues and create one big PR"* +2. The assistant delegates to subagents in parallel +3. Merges their branches, handles CI failures iteratively +4. Pushes and opens the final PR +5. You review the PR once CI passes — no intermediate check-ins + +### Scheduling +For periodic maintenance (security audits, code quality scans), run: + +```bash +opencode run --agent security-auditor "Audit the codebase for vulnerabilities" +opencode run --agent code-quality-guardian "Scan for code quality issues" +``` + +This can be cron'd locally if desired (e.g., `crontab -e` with `opencode run`). + +## Is opencode a good option? + +**Yes, for FastSync's needs.** The hybrid model works well: +- opencode's 17 specialized agents handle deep code analysis, fixes, tests, and reviews +- The assistant orchestrates subagents, merges branches, and iterates on CI +- You only review the final output + +The key limitation: opencode is session-based, not a persistent daemon. But for the "fix all issues, one PR" workflow, this is fine — the assistant runs the full pipeline in one shot. Persistent webhook-driven automation isn't available for Gitea, but the one-shot batch approach is simpler and gives you full control over what gets merged. + +### Recommendations for this project +- **Do** use the batch pattern: delegate to subagents, let the assistant merge + iterate CI, review once +- **Don't** try to run opencode in Gitea Actions — the CI container doesn't have your LLM keys or the interactive context agents need +- **If** you want fully hands-off periodic scans, set up a local cron job or systemd timer that runs `opencode run` and posts results to Gitea via API + ## Gitea API & tea CLI ### Check CI status via API @@ -155,7 +207,7 @@ tea pr close --repo TapTap/FastSync ## Common pitfalls -- **`__thread` on shared SSL context**: io_ssl must NOT be thread-local — worker threads inherit the SSL context from the main thread. Use regular `static SSL* io_ssl`. +- **Per-thread SSL context**: `io_ssl` is stored per-thread (`static __thread SSL* io_ssl`). Each thread that performs protocol I/O must call `io_set_ssl()` to install its own SSL object before using `send_*` / `receive_*` primitives. The main thread's SSL context is not automatically inherited by worker threads. - **SSL WANT_READ/WANT_WRITE retry**: Always retry on `SSL_ERROR_WANT_READ` and `SSL_ERROR_WANT_WRITE` in `send_n_data`/`receive_n_data`. Removing these breaks TLS multithreaded transfers. - **clang-format version**: The CI image uses clang-format 18. Always format inside the CI Docker container for exact match. - **Merge order matters**: Merge the most comprehensive branch first, then smaller ones, to minimize conflicts when creating a combined branch. diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..bcdf3a4 --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,65 @@ +# Changelog + +All notable changes to FastSync are documented here. Versions match +`PROTOCOL_VERSION` (printed by `fastsync --version`); the client and server must +run the same version because the handshake is strict. + +## [2.19.0] - 2026-09-12 + +### Security + +- **Daemon authentication rewritten as SCRAM-SHA-256 challenge/response** + (`STATUS_AUTH_CHALLENGE` → `STATUS_AUTH_RESPONSE` → `STATUS_AUTH_OK`/`STATUS_AUTH_FAILED`), + replacing the old replayable static `SHA-256(password)` bearer credential. + Each proof is bound to a fresh per-connection server nonce plus a client + nonce, so a captured response can never be reused. +- **Salted verifier store.** `--password-file`/`--early-input` now hold + `user:$fastsync$1$pbkdf2-sha256$$$$` + (PBKDF2-HMAC-SHA256, default 600000 iterations, range 100000–10000000). The + legacy `user:SHA256HEX` form is hard-rejected; there is no auto-upgrade. + Generate stores offline with `fastsync-server --hash-credentials FILE + [--iterations N]`. +- **Username-enumeration hardening.** Unknown/off-list users are answered with a + dummy verifier whose salt is a deterministic per-username value + (`HMAC-SHA256(dummy_key, username)`), using the store-wide uniform iteration + count and a constant-time full-length membership scan. The dummy key is + persisted in an owner-only `.dummykey` sidecar (atomic publish, exact + mode 0600) so challenges are stable across restarts. +- **Verified transport for auth-required modules.** A module with `auth users` + accepts credentials only over verified TLS whose client certificate matches + `--client-cn`, or — when `--allow-unauthenticated` is explicitly set — + plaintext from a loopback peer. Remote plaintext is refused before any + challenge. Clients must use `--tls` to send `--password-file` credentials to a + non-loopback daemon; `--client-cn` is mandatory with `--tls`. +- **Secret hygiene.** The plaintext password, derived keys, nonces/proofs and + the dummy key are wiped from memory on every path and never logged. +- Carried-over hardening: `-K` TOCTOU-safe directory walk + (`openat(O_NOFOLLOW)` per component), always shell-quoted SSH remote path, + TLS compression/renegotiation disabled, race-free (open-then-`fstat`) + `--password-file`/`--early-input` checks, log-injection escaping, and lazy + protocol debug escaping. + +### Added + +- `fastsync-server --hash-credentials FILE [--iterations N]` offline tool. +- `.dummykey` sidecar (auto-created, owner-only, 0600). +- Integration tests for auth replay rejection, malformed frames, legacy-store + refusal, and the loopback/TLS transport policy; fuzz targets for config + receive and daemon-auth parsing. + +### Changed + +- **Protocol version 2.18.0 → 2.19.0 (breaking).** The config-frame auth block + is now `[present][username]` (digest removed) and the auth challenge/response + frames are interleaved between the config frame and its `STATUS_OK`. A 2.19.0 + client and a 2.18.0 server (or vice versa) fail cleanly at the handshake. +- Daemon modules declaring `auth users` require a configured credential store at + startup (fail closed); operators regenerate stores from plaintext with + `--hash-credentials`. + +### Notes + +- First tagged release. FastSync implements rsync-compatible file + synchronization over TCP and SSH with TLS (OpenSSL), streaming zstd + compression, multithreaded transfers, and incremental sync. See + [RSYNC_COMPAT.md](RSYNC_COMPAT.md) for the flag-parity matrix. diff --git a/CMakeLists.txt b/CMakeLists.txt index db6a15f..3ec3b24 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1,6 +1,6 @@ cmake_minimum_required(VERSION 3.22) -project(FastFileTransfer) +project(FastFileTransfer VERSION 2.19.0) set(CMAKE_EXPORT_COMPILE_COMMANDS ON) set(CMAKE_C_STANDARD 11) @@ -58,15 +58,18 @@ endif() find_package(OpenSSL REQUIRED) file(GLOB SHARED_SRCS "src/shared/*.c") +set(FILE_STORE_SRCS "${CMAKE_CURRENT_SOURCE_DIR}/src/shared/file_store.c") +list(REMOVE_ITEM SHARED_SRCS ${FILE_STORE_SRCS}) file(GLOB SERVER_SRCS "src/server/*.c") +set(SERVER_RECEIVER_SRCS src/server/receiver.c) file(GLOB CLIENT_SRCS "src/client/*.c") # --- Main executables --- -add_executable(server ${SERVER_SRCS} ${SHARED_SRCS}) +add_executable(server ${SERVER_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS}) target_include_directories(server PRIVATE src/shared src/server src/client) target_link_libraries(server PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash) -add_executable(client ${CLIENT_SRCS} ${SHARED_SRCS}) +add_executable(client ${CLIENT_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS}) target_include_directories(client PRIVATE src/shared src/server src/client) target_link_libraries(client PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash) @@ -79,8 +82,9 @@ set(TEST_INCLUDES tests src/shared src/server src/client) # Monolithic test binary (backward compatible) file(GLOB TEST_SRCS "tests/test_*.c" "tests/runner.c") -add_executable(tests ${TEST_SRCS} ${SHARED_SRCS} src/client/scanner.c) +add_executable(tests ${TEST_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS} src/client/scanner.c src/client/change_list.c src/client/client_cli.c src/client/client_validation.c src/client/usage.c src/server/server_cli.c) target_include_directories(tests PRIVATE ${TEST_INCLUDES}) +target_compile_definitions(tests PRIVATE FASTSYNC_TEST_BUILD) target_link_libraries(tests PRIVATE ${TEST_LIBS}) add_test(NAME unit_all COMMAND tests) @@ -93,7 +97,7 @@ if(ENABLE_FUZZ) file(GLOB FUZZ_SRCS "tests/fuzz/*.c") foreach(FUZZ_SRC ${FUZZ_SRCS}) get_filename_component(FUZZ_NAME ${FUZZ_SRC} NAME_WE) - add_executable(${FUZZ_NAME} ${FUZZ_SRC} ${SHARED_SRCS}) + add_executable(${FUZZ_NAME} ${FUZZ_SRC} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS}) target_include_directories(${FUZZ_NAME} PRIVATE ${TEST_INCLUDES}) target_compile_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined -fno-omit-frame-pointer) target_link_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined) diff --git a/Dockerfile b/Dockerfile index c94d26f..965384e 100644 --- a/Dockerfile +++ b/Dockerfile @@ -3,7 +3,7 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ gcc g++ make libc6-dev cmake libzstd-dev libssl-dev git ca-certificates curl cppcheck clang-format \ python3 python3-pip python3-venv openssl openssh-client \ lcov valgrind clang libclang-rt-18-dev && \ - pip3 install --break-system-packages pytest && \ + pip3 install --break-system-packages pytest pytest-xdist && \ curl -fsSL https://deb.nodesource.com/setup_20.x | bash - && \ apt-get install -y --no-install-recommends nodejs && \ rm -rf /var/lib/apt/lists/* diff --git a/README.md b/README.md index 353133c..204a19d 100644 --- a/README.md +++ b/README.md @@ -1,129 +1,135 @@ -# FastSync +#FastSync -A high-performance file synchronization system with SSH and TCP transport, TLS encryption, streaming zstd compression, multithreaded transfer, incremental sync, metadata preservation, and rsync-compatible CLI flags. +FastSync is a high-performance file synchronization tool designed to become a +drop-in replacement for common `rsync` workflows. It keeps the familiar +source/destination model and rsync-style options while adding optional +multithreading, streaming zstd compression, chunking, zero-copy TCP transfers, +and native TCP/TLS transports. -## Technical Overview +The release version is FastSync's client/server protocol version (printed by +`fastsync --version`); client and server must match. See +[CHANGELOG.md](CHANGELOG.md) for the history. -1. **Dual transport**: custom TCP client-server or SSH subprocess (rsync-style `user@host:/path`) -2. **TLS encryption**: OpenSSL-based TLS 1.2+ for encrypted TCP connections with optional CA verification -3. **Chunked file transfer**: files grouped into configurable-size chunks (default ~10 MB) -4. **Streaming zstd compression** (levels 1–22) using `ZSTD_compressStream2` -5. **Multithreading**: producer-consumer pipeline with thread-safe queues (scanner → loader → sender) -6. **Incremental sync**: skip files unchanged since last transfer (compares size + mtime) -7. **Batch incremental**: send incremental checks in batched groups for reduced round-trips -8. **Metadata preservation**: `mode`, `uid`, `gid`, `mtime` restored on disk when enabled -9. **`sendfile()` zero-copy** on TCP (~2× faster on loopback) -10. **SSH ControlMaster** for connection reuse across repeated invocations -11. **Bandwidth limiting**: token-bucket throttling (`--bwlimit`) -12. **`--delete`**: receiver removes files not present in sender manifest -13. **`--exclude` / `--include`**: glob-pattern filename filtering -14. **Path traversal protection**: `..` sequences in file paths are rejected automatically -15. **Connection limits**: server enforces maximum concurrent connections (default 100) -16. **Keep-alive**: periodic `STATUS_KEEPALIVE` messages detect stalled connections -17. **Abort handling**: `SIGINT` sends `STATUS_ABORT` for clean server-side teardown -18. **Atomic writes**: received files are written to a temporary name then atomically renamed -19. **Backup mode**: `--backup` preserves overwritten files with optional `--backup-dir` -20. **Log file**: `--log-file` redirects log output to a file instead of stderr -21. **Transfer statistics**: `--stats` prints summary of transferred bytes, files, and timing +The compatibility target is straightforward: -## System Architecture +- Existing rsync commands should keep the same meaning. +- FastSync-only performance options should be additive and optional. +- A normal compatibility-mode transfer should prioritize rsync filesystem + semantics over maximum throughput. -### Client -- Recursively scans source directories (BFS), supports exclude and include patterns -- Groups files into chunks (configurable size) -- Streaming zstd compression with configurable level -- Chunk serialization (compact binary format) or per-file transfer -- Incremental transfer: sends file metadata to server, skips unchanged files -- Batch incremental: groups incremental checks to minimize round-trips -- Manifests all sent paths when `--delete` is active -- Sends via TCP `sendfile()` or SSH pipe -- Optional progress display with throughput -- Bandwidth limiting via token-bucket algorithm -- Configurable I/O and connection timeouts (`--timeout`, `--contimeout`) -- Quiet mode (`-q`/`--quiet`) suppresses all non-error output -- Backup overwritten files (`--backup`) with optional directory (`--backup-dir`) -- Transfer statistics summary (`--stats`) -- Maximum directory depth control (`--max-depth`) -- Log file output (`--log-file`) -- Configurable multithreaded queue size (`--queue-size`) -- Exclude patterns from file (`--exclude-from`) +FastSync currently speaks its own protocol to `fastsync-server`. SSH mode +starts that server remotely; it does not yet interoperate with an unmodified +rsync client or rsync daemon. See [Compatibility Status](#compatibility-status) +for the current boundary. -### Server -- TCP mode: listens on configurable port (default 8080); SSH mode: runs via `--stdio` -- TLS mode: wraps TCP connections with OpenSSL with optional CA verification -- Receives and reassembles files -- Decompresses (streaming zstd), deserializes, restores metadata -- Handles incremental checks: compares size + mtime against destination files -- Handles batch incremental checks for reduced round-trips -- Processes `STATUS_MANIFEST` for `--delete`: walks destination tree, removes extras -- Per-connection concurrency via `fork()` with configurable connection limit (default 100) -- Thread pool for parallel processing -- Atomic writes: files written to `.tmp` path then atomically renamed on success -- Abort handling: cleanly shuts down on `STATUS_ABORT` from client -- Path traversal protection: rejects file paths containing `..` +## Why FastSync -## Protocol Details +FastSync uses a producer-consumer transfer pipeline and can combine several +optimizations for large or high-latency transfers: -### Status Codes -| Code | Meaning | -|------|---------| -| `STATUS_OK` | Operation successful | -| `STATUS_ERROR` | Error occurred | -| `STATUS_FINISHED` | Transfer complete | -| `STATUS_NEXT` | Ready for next file (per-file mode) | -| `STATUS_CHUNK` | Following data is a serialized chunk | -| `STATUS_MANIFEST` | Following data is a file manifest (for `--delete`) | -| `STATUS_CHECK` | Incremental check: client sends file path + size + mtime, server responds with OK (skip) or NEXT (send) | -| `STATUS_CHECK_BATCH` | Batch incremental check: multiple file checks sent in one message | -| `STATUS_KEEPALIVE` | Keep-alive heartbeat to detect stalled connections | -| `STATUS_ABORT` | Abort signal: client interrupts, server cleans up and exits | -| `STATUS_DELTA_SIGNATURE` | Delta sync: following data is a file signature (rsync-style rolling hash) | -| `STATUS_DELTA_DATA` | Delta sync: following data is a delta patch for a file | +- Multithreaded scanning, loading, and sending. +- Streaming zstd compression with levels 1 through 22. +- Configurable file chunking and compact chunk serialization. +- `sendfile()` zero-copy transfers over TCP. +- Batched incremental checks to reduce round trips. +- Optional block-level delta transfer for FastSync peers. +- Bandwidth limiting, progress reporting, statistics, and backups. +- TCP, SSH, and TLS transports. +- Atomic temporary-file writes by default. -### Wire Format — Metadata +These optimizations are disabled or selected independently. Users can start +with rsync-style commands and add FastSync options when they are useful. -When `use_metadata` is enabled (`-M`), each file entry carries a 4-byte `present` flag followed by five fields (`mode`, `uid`, `gid`, `mtime_sec`, `mtime_nsec`). When disabled globally, no metadata bytes are sent — zero wire overhead. +## Compatibility Status -### Transfer Flow -``` -Config → (STATUS_NEXT | STATUS_CHUNK | STATUS_CHECK | STATUS_CHECK_BATCH)* → [STATUS_MANIFEST] → STATUS_FINISHED → STATUS_OK -``` +FastSync is currently an rsync-compatible CLI in progress, not a complete +replacement for every rsync feature or protocol mode. -Keep-alive (`STATUS_KEEPALIVE`) may be sent at any point during the transfer. The receiver resets its inactivity timer on receipt. If no data arrives within the receive timeout, the connection is aborted. +### Working today -Abort (`STATUS_ABORT`) may be sent at any point. On receipt the server cleans up temporary files and exits the child process. +- Recursive directory scanning. +- Rsync-style source and destination arguments. +- SSH transport using `user@host:destination` paths below the remote authorized root. +- TCP client/server transfers. +- Dry runs, excludes, includes, size filters, backups, statistics, and + bandwidth limiting. +- Incremental size/mtime checks and optional xxHash64 content checks. +- FastSync-native delta transfer for changed files. +- Optional mode and timestamp preservation. +- Delete manifests with server-side delete authorization. +- Temporary-file writes with atomic rename by default. +- Path traversal checks and destination-root confinement. -### Protocol Version +### Not yet equivalent to rsync -`1.3.0` — server and client must match. Mismatch results in `STATUS_ERROR`. +- The FastSync wire protocol is not the rsync wire protocol. +- SSH mode requires `fastsync-server` on the remote host. +- Archive mode does not yet provide all of rsync's `-rlptgoD` behavior. +- Symlink transfer is incomplete; link targets are not yet recreated in all + modes. +- Owner/group, ACL, xattr, and hard-link handling is incomplete or + unavailable. +- Device and special-file preservation is implemented with documented + divergences: recreated device nodes require `CAP_MKNOD` on the receiver (a + non-root receiver skips the entry), and sockets cannot be recreated (FIFOs + are). +- Sparse-file hole preservation (`-S`, `--sparse`) is implemented receiver-side: + long all-zero runs are written as holes (no wire change; the full file image + is already in memory). +- `--partial`, `--partial-dir`, `-P`, `--append`, and `--append-verify` keep + the write atomic (temp + rename). With `--partial`, a failed/interrupted write + now retains the already-written temp at the destination path (best-effort) so + a later `--append`/`--append-verify` run can resume it. +- `--dirs` is not implemented. Its compatibility aliases `--old-dirs` and + `--old-d` are recognized but rejected explicitly rather than silently using + FastSync's recursive directory behavior. +- Short-option names are now rsync-parity (Phase 7 Wave A): FastSync's former + collisions were renamed (`-j`/`--threads`, `--preserve`, `--sendfile`, + `--chunk-serialization`, `--timeout`, `--ssh-port`), so `-m`, `-M`, `-f`, + `-s`, `-T`, `-p`, `-c`, `-a`, and `-z` follow rsync. See `RSYNC_COMPAT.md`. -## Command-Line Arguments +The detailed flag matrix is maintained in +[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md). It distinguishes implemented, +partial, alternate, and planned behavior. + +## Quick Start + +### Build ### Client | Argument | Description | |----------|-------------| | Positional | ` ` — automatic SSH detection if dest contains `:` | -| `-c [level]` | Compression with optional level (1–22, default 5) | -| `-z [level]` | Alias for `-c` | -| `-a, --archive` | Archive mode: enables `-c -m -M` (no `-s`) | -| `-m` | Multithreading mode | -| `-s` | Chunk serialization (batch all files per chunk) | -| `-f, --sendfile` | Sendfile zero-copy. Incompatible with `-c` / `-s`. TCP only. | -| `-M, --preserve` | Preserve file metadata (mode, uid, gid, mtime) | +| `-c, --checksum` | Verify content by checksum instead of size+mtime | +| `-z, --compress [level]` | Enable streaming zstd compression (level 1–22, default 5) | +| `-a, --archive` | rsync archive mode (`-rlptgoD`): links, metadata, devices and specials (not compression/multithreading) | +| `-j, --threads` | Multithreading mode | +| `-m` | rsync `--prune-empty-dirs` (short form now rsync-parity) | +| `--chunk-serialization` | Chunk serialization (batch all files per chunk; long form only) | +| `-s` | rsync `--secluded-args` compatibility no-op (remote SSH argv is already injection-safe) | +| `--sendfile` | Sendfile zero-copy. Incompatible with compression / chunk serialization. TCP only. Long form only. | +| `--preserve` | Preserve supported file metadata (mode and mtime; ownership and atime are unsupported) | | `-n, --dry-run` | Scan and print what would be transferred | -| `-p ` | SSH port (default: 22) | +| `-p, --perms` | Preserve permission bits (part of the metadata bundle) | +| `--ssh-port ` | SSH port (default: 22) | | `-v, --verbose` | Enable debug logging | -| `-q, --quiet` | Suppress all non-error output | -| `--silent` | Alias for `--quiet` | +| `-q, --quiet` | Suppress non-error output | | `--progress` | Show real-time transfer speed | -| `--delete` | Delete files on receiver not present in source | +| `-P` | Enables partial-transfer mode + progress output; interrupted writes retain the already-written temp for resumption | +| `--delete` | Delete files on receiver not present in source (default timing: delete-after, i.e. only after the whole transfer succeeded) | +| `--delete-before` | Delete extras before the transfer starts (implies `--delete`) | +| `--delete-during`, `--del` | Delete extras once the keep-set is known, before data is applied (implies `--delete`) | +| `--delete-delay` | Delete extras only after a successful transfer (implies `--delete`) | +| `--delete-after` | Explicit delete-after timing (implies `--delete`) | | `--exclude ` | Exclude files matching glob pattern (repeatable) | | `--exclude-from ` | Read exclude patterns from a file (one per line) | | `--include ` | Only transfer files matching glob pattern (repeatable, whitelist) | | `--max-size ` | Skip files larger than n bytes | | `--min-size ` | Skip files smaller than n bytes | -| `--incremental` | Skip files unchanged since last transfer (size + mtime). Auto-enables `--preserve`. Incompatible with `-s`. | +| `--max-alloc ` | Maximum single allocation (binary units: B, K, M, G, T, P, E; default 1G) | +| `--incremental` | Skip files unchanged since last transfer (size + mtime). Auto-enables `--preserve`. Incompatible with `--chunk-serialization`. | +| `--existing` | Skip files not already present at the destination; update existing files normally. | | `--bwlimit ` | Bandwidth limit in kilobytes per second | | `--chunk-size ` | Chunk size in bytes (default: 10485760) | | `--timeout ` | I/O timeout in seconds (default: 30) | @@ -131,9 +137,9 @@ Abort (`STATUS_ABORT`) may be sent at any point. On receipt the server cleans up | `--backup` | Backup existing destination files before overwriting | | `--backup-dir ` | Target directory for backups (requires `--backup`) | | `--stats` | Print transfer statistics at end (bytes, files, timing) | +| `-h, --human-readable` | Format transfer byte sizes with binary units | | `--max-depth ` | Maximum directory depth to recurse (0 = unlimited, default: 0) | | `--log-file ` | Write log messages to file instead of stderr | -| `--queue-size ` | Queue capacity for multithreaded mode (default: 100) | | `--source-dir ` | Source directory (overrides `FASTSYNC_SOURCE_DIR`) | | `--dest-dir ` | Server destination directory (overrides `FASTSYNC_DEST_DIR`) | | `--save-to-disk` | Write received files to disk | @@ -143,6 +149,7 @@ Abort (`STATUS_ABORT`) may be sent at any point. On receipt the server cleans up | `--cert ` | TLS certificate file (PEM) | | `--key ` | TLS private key file (PEM) | | `--ca ` | TLS CA certificate file for verification (PEM) | +| `--client-cn ` | TLS client certificate common name; mandatory with `--tls` (a TLS connection always verifies the client CN) | ### Server @@ -154,6 +161,9 @@ Abort (`STATUS_ABORT`) may be sent at any point. On receipt the server cleans up | `--cert ` | TLS certificate file (PEM) | | `--key ` | TLS private key file (PEM) | | `--ca ` | TLS CA certificate file for verification (PEM) | +| `--destination-root ` | Authorized destination root (default: `.`) | +| `--allow-delete` | Permit manifest deletion | +| `--allow-unauthenticated` | Permit plaintext TCP clients. For an `auth users` module this opts in **loopback plaintext only**; remote auth still requires verified TLS, so the flag never permits remote plaintext auth. | | `-v, --verbose` | Enable debug logging | | `--help` | Show help | @@ -176,26 +186,46 @@ Abort (`STATUS_ABORT`) may be sent at any point. On receipt the server cleans up ### Data Structures 1. **Chunk** — collection of files (~10 MB total by default) 2. **File** — path, content (`Data`), optional `FileMetadata` pointer -3. **FileMetadata** — `mode`, `uid`, `gid`, `mtime_sec`, `mtime_nsec` +3. **FileMetadata** — `mode`, `uid`, `gid`, `mtime_sec`, `mtime_nsec`; +uid / gid are advisory wire fields and are never applied by the receiver; +atime is unsupported 4. **Config** — runtime parameters (transported over wire, TLS settings excluded). Includes `timeout`, `contimeout`, `quiet`, `backup`, `backup_dir`, `stats`, `max_depth`, `log_file`, `queue_size`. 5. **Queue** — thread-safe bounded queue with condition variables 6. **DirectoryScanner** — recursive BFS traversal with exclude and include pattern support, max-depth enforcement ### Key Algorithms -1. **File scanning** — BFS directory traversal; entries matched against exclude and include patterns, max-depth enforced -2. **Chunking** — files accumulated until `chunk_size` threshold, then flushed -3. **Compression** — streaming zstd via `ZSTD_compressStream2` / `ZSTD_decompressStream` -4. **Network protocol** — status-code-driven exchange with metadata packing, keep-alive, and abort support -5. **Incremental check** — client sends `STATUS_CHECK` + path + size + mtime; server compares against destination. Can be batched via `STATUS_CHECK_BATCH` for reduced round-trips. +1. **File scanning** — BFS directory traversal; +entries matched against exclude and include patterns, + max - depth enforced 2. * *Chunking ** — files accumulated until `chunk_size` threshold, + then flushed 3. * + *Compression ** — streaming zstd + via `ZSTD_compressStream2` / `ZSTD_decompressStream` 4. * + *Network protocol ** — status - + code - driven exchange with metadata packing, + keep - alive, + and abort support 5. * *Incremental check ** — client sends `STATUS_CHECK` + path + size + + mtime and, + with `--checksum`, XXH64 content checksum; server compares against destination. Can be batched via `STATUS_CHECK_BATCH` for reduced round-trips. 6. **Bandwidth limiting** — token-bucket algorithm with `nanosleep` throttling on 64 KB write chunks 7. **Metadata restoration** — `chmod()`, `chown()`, `utimensat()` on the receiving side -8. **`--delete`** — sender tracks all sent paths; receiver walks destination tree and removes unlisted files/directories -9. **SSH transport** — `socketpair()` + `fork()` + `execvp("ssh", ...)` with `ControlMaster` and port support -10. **TLS transport** — OpenSSL `SSL_CTX` with TLS 1.2 minimum, optional CA verification, transparent `SSL_read`/`SSL_write` via `io_set_ssl()` -11. **Path traversal protection** — `has_path_traversal()` rejects any file path containing `..` components, preventing directory escape attacks -12. **Connection limiting** — server tracks active connections and rejects new ones beyond `max_connections` (default 100) -13. **Keep-alive** — idle connections receive periodic `STATUS_KEEPALIVE` to detect half-open TCP connections -14. **Abort handling** — `SIGINT` sets an abort flag; the next protocol operation sends `STATUS_ABORT` for clean server cleanup +8. **`--delete`** — sender tracks all sent paths; +receiver walks destination tree and removes unlisted files / directories 9. * + *SSH transport * + * — `socketpair()` + `fork()` + `execvp("ssh", + ...)` with `ControlMaster` and port support + 10. * + *TLS transport ** — OpenSSL `SSL_CTX` with TLS + 1.2 minimum, + mutual CA verification, + transparent `SSL_read`/`SSL_write` via `io_set_ssl()` 11. * + *Path traversal protection ** — `has_path_traversal()` rejects any file path + containing `..` components, + preventing directory escape attacks 12. * + *Connection limiting ** — server tracks active connections and rejects + new ones beyond `max_connections` (default 100)13. * + *Keep + - alive ** — idle connections receive periodic `STATUS_KEEPALIVE` to detect half + - open TCP connections 14. * *Abort handling ** — `SIGINT` sets an abort flag; the next protocol operation sends `STATUS_ABORT` for clean server cleanup 15. **Atomic writes** — files are written to a `.tmp` suffix then atomically renamed via `rename()`, preventing partial files 16. **Backup** — before overwriting, existing files are moved to `--backup-dir` (or same directory with `~` suffix) preserving the original @@ -205,7 +235,7 @@ Abort (`STATUS_ABORT`) may be sent at any point. On receipt the server cleans up All received file paths are validated by `has_path_traversal()` before any disk operation. Any path containing `..` components is rejected with `STATUS_ERROR`, preventing directory escape attacks. ### TLS Certificate Verification -When `--ca` is provided, the server performs mutual TLS verification (`SSL_VERIFY_PEER` with depth 4). Without `--ca`, TLS is still encrypted but peer certificates are not verified. +TLS requires `--ca` and performs mutual TLS verification (`SSL_VERIFY_PEER` with depth 4). Connections without certificate verification are rejected. ### Connection Limits The server enforces a maximum of 100 concurrent connections (configurable via `max_connections` in `Server`). When the limit is reached, new connections are immediately rejected and closed. @@ -240,130 +270,368 @@ nix-shell # provides zstd, openssl, cmake, gcc ## Building ```bash -cmake -B build -S . && cmake --build build -j$(nproc) +cmake -B build -S . +cmake --build build -j$(nproc) ``` -## Running +With Nix: -### Server (TCP mode) ```bash -./build/server +nix-shell +cmake -B build -S . +cmake --build build -j$(nproc) ``` -### Server with TLS +### SSH transfer + +The remote host must have `fastsync-server` available in `PATH`, or use +`--fastsync-server-path`. SSH starts `fastsync-server --stdio` in its remote +working directory, so use a destination below that directory unless the +remote server is otherwise configured with a matching authorized root. + ```bash -./build/server --tls --cert server.pem --key server-key.pem +ssh user@host 'mkdir -p destination' +./build/client /path/to/source user@host:destination ``` -### Server via SSH -Place the `fastsync-server` binary in the remote `$PATH`. The client runs `ssh user@host fastsync-server --stdio` automatically when an SSH-style destination is given. +### TCP transfer + +Start the FastSync server: -### Client — SSH (rsync-style) ```bash -./build/client /path/to/send user@host:/path/to/receive +./build/server --destination-root /path/to -p 8080 ``` -### Client — TCP +Then run the client: + ```bash -./build/client --source-dir /path/to/send --dest-dir /path/to/receive --save-to-disk +./build/client --server-host 127.0.0.1 --server-port 8080 \ + --source-dir /path/to/source --dest-dir /path/to/destination \ + --save-to-disk ``` -### Client — TCP with TLS +Plain TCP requires the explicit `--allow-unauthenticated` server option. Use TLS for +authenticated network connections. + +### TLS transfer ```bash +./build/server --destination-root /path/to --tls --cert server.pem --key server-key.pem -p 8443 ./build/client --tls --cert client.pem --key client-key.pem --ca ca.pem \ - --source-dir /path/to/send --dest-dir /path/to/receive --save-to-disk + --server-host example.com --server-port 8443 \ + --source-dir /path/to/source --dest-dir /path/to/destination \ + --save-to-disk ``` -### Common Options +## Common Workflows + +These examples show the intended rsync-style workflow. Options marked as +FastSync-native are optional performance or transport extensions. + ```bash -# Archive mode (compression + multithreading + metadata) -./build/client -a /path/to/send user@host:/path +#Basic synchronization +./build/client /source/ /destination/ -# Dry run -./build/client -n /path/to/send /path/to/receive +#Archive - style synchronization(current FastSync archive behavior) +./build/client -a /source/ user@host:destination/ -# With progress and custom chunk size -./build/client --progress --chunk-size 2097152 /src user@host:/dst +#Preview a transfer without changing the destination +./build/client -n /source/ /destination/ -# Exclude temporary files + delete extras on receiver -./build/client --exclude "*.tmp" --exclude "*.o" --delete /src user@host:/dst +#Exclude temporary and object files +./build/client --exclude '*.tmp' --exclude '*.o' \ + /source/ user@host:destination/ -# Incremental sync (skip unchanged files) -./build/client --incremental /src user@host:/dst +#Remove destination entries not present in the source +./build/client --delete /source/ user@host:destination/ -# Bandwidth limit to 1 MB/s -./build/client --bwlimit 1024 /src user@host:/dst +#Skip unchanged files using size and modification time +./build/client --incremental /source/ user@host:destination/ -# With timeouts, quiet mode, and stats -./build/client --timeout 60 --contimeout 15 --quiet --stats /src user@host:/dst +#Verify content when size and time are not sufficient +./build/client --incremental --checksum /source/ user@host:destination/ -# Backup overwritten files to a directory -./build/client --backup --backup-dir /backups /src user@host:/dst +#Preserve supported mode and timestamp metadata +./build/client -M /source/ user@host:destination/ -# Exclude patterns from file, limit depth -./build/client --exclude-from ignore.txt --max-depth 3 /src user@host:/dst - -# Custom queue size for multithreading -./build/client -m --queue-size 200 /src user@host:/dst - -# Log to file -./build/client --log-file /tmp/fastsync.log /src user@host:/dst - -# All features -./build/client -a --progress --chunk-size 5242880 --exclude "*.log" --delete /src /dst +#Keep backups of overwritten destination files +./build/client --backup --backup-dir backups \ + /source/ user@host:destination/ ``` +## FastSync Extensions + +FastSync-native options are intended to add performance or operational +features without changing the meaning of ordinary compatibility options. + +| Option | Purpose | +|---|---| +| `-j`, `--threads` | Enable the multithreaded scanner/loader/sender pipeline. | +| `-z [level]`, `--compress [level]` | Enable streaming zstd compression, levels 1-22. | +| `--compress-level ` | Set the zstd compression level. | +| `--zc ` | Alias for `--compress-choice`. FastSync supports `zstd` and `none`. | +| `--zl ` | Alias for `--compress-level`. | +| `--skip-compress ` | Skip compression for comma-separated suffixes; incompatible with `--chunk-serialization`. | +| `--compress-threads ` | Use `n` zstd compression workers. Requires compression and a zstd build with threaded support; the setting affects sender CPU work only. | +| `--chunk-size ` | Set the transfer chunk size. | +| `--chunk-serialization` | Enable FastSync chunk serialization (long form only; `-s` is rsync's `--secluded-args`). | +| `--sendfile` | Use TCP `sendfile()` zero-copy transfer. Incompatible with compression and chunk serialization. Long form only. | +| `--delta` | Use FastSync-native block delta transfer. Requires `--incremental`. | +| `--delta-block ` | Set the FastSync delta block size (`--block-size` is an alias). | +| `--delta-max ` | Limit files eligible for FastSync delta transfer. | +| `--server-host ` | Select the TCP server host. | +| `--server-port ` | Select the TCP server port. | +| `--tls` | Enable TLS for TCP transport. | +| `--bwlimit ` | Apply token-bucket bandwidth limiting. | +| `--progress` | Show transfer progress and throughput. | +| `--stats` | Print transfer statistics. | +| `--timeout ` | Set I/O timeout. | +| `--contimeout ` | Set connection timeout. | + +Short-option conflicts with rsync have been resolved for the CLI namespace +(Phase 7): `-c` is now rsync's `--checksum`, `-m` is `--prune-empty-dirs`, `-M` +is `--remote-option`, `-f` is `--filter`, `-s` is `--secluded-args`, `-p` is +`--perms`, and `-T` is `--temp-dir`. FastSync's own flags were renamed to +long-form-only or new shorts: multithreading is `-j`/`--threads`, metadata +is `--preserve`, sendfile is `--sendfile`, chunk serialization is +`--chunk-serialization`, timeout is `--timeout`, and SSH port is `--ssh-port`. +`-a`/`--archive` is now real rsync archive (`-rlptgoD`). + +`--secluded-args` (and its short form `-s`) is accepted as a compatibility +no-op. It does not change FastSync's transport or protocol behavior, because +remote SSH argv is already built injection-safe. + +## Client Options + +### Selection and transfer + +| Option | Description | +|---|---| +| `-a`, `--archive` | rsync archive mode (`-rlptgoD`): links, metadata, devices and specials. | +| `-n`, `--dry-run` | Scan and report without writing files. | +| `--delete` | Request removal of destination entries absent from the source. The server must allow deletion. Default timing is delete-after: extras are removed only after the whole transfer succeeded. | +| `--delete-before` | Delete extras before the transfer starts (implies `--delete`). | +| `--delete-during`, `--del` | Delete extras once the keep-set manifest is known, before data is applied (implies `--delete`; early mode, same engine behaviour as `--delete-before`). | +| `--delete-delay` | Delete extras only after a successful transfer (implies `--delete`; commit mode, same behaviour as `--delete-after`). | +| `--delete-after` | Explicit delete-after timing: delete only after the transfer succeeded (implies `--delete`). | +| `--exclude ` | Exclude matching paths. Repeatable. | +| `--include ` | Include matching paths. Repeatable. | +| `--exclude-from ` | Read exclude patterns from a file. | +| `--include-from ` | Read include patterns from a file. | +| `--max-size ` | Skip files larger than the limit. | +| `--min-size ` | Skip files smaller than the limit. | +| `--max-depth ` | Limit recursive scanning depth; +zero means unlimited.| | `--incremental` | Skip files matching destination size and mtime.| + | `--checksum` | Include xxHash64 content checks in incremental comparisons.| | `--backup` | + Back up overwritten files.| | `--backup - dir` | Store backups under a separate directory.| +| `--suffix` | Set the backup filename suffix.| | `--partial` | + Select partial - transfer handling. On failed/interrupted writes the + already-written temp file is retained (best-effort) for resumption.| + With `--partial --partial-dir `, completed files are written under the + partial directory and installed atomically. | | `--partial - dir` | + Set a relative partial - transfer directory below the server destination root. + Use with `--partial`. | +| `--inplace` | Write directly to the destination instead of using a temporary file. | + +### Metadata and links + +| Option | Description | +|---|---| +| `--preserve` | Preserve supported file metadata, currently mode and modification time (long form only). | +| `-l`, `--links` | Request symlink preservation; +link-target transfer remains incomplete. | +| `--copy-links` | Copy symlink referents. | +| `--safe-links` | Skip symlinks that point outside the transfer tree. | +| `--copy-unsafe-links` | Copy unsafe symlink referents. | +| `-S`, `--sparse` | Sparse-file handling: receiver preserves holes (zero runs are written as holes; no wire change). | + +### Output and logging + +| Option | Description | +|---|---| +| `-v`, `--verbose` | Enable debug logging. | +| `--progress` | Show live transfer progress. | +| `--stats` | Print transfer statistics. | +| `--log-file ` | Write log output to a file. | +| `-V`, `--version` | Print the FastSync protocol version. | +| `--help` | Print command usage. | + +### Paths and transport + +| Option | Description | +|---|---| +| `--ssh-port ` | SSH port for the SSH transport (default: 22). Note the short `-p` is now rsync's `--perms`. | +| `--fastsync-server-path ` | Remote FastSync server path for SSH mode. | +| `--source-dir ` | Set the source directory explicitly. | +| `--dest-dir ` | Set the destination directory explicitly. | +| `--save-to-disk` | Enable server-side disk persistence. | +| `--server-host ` | TCP server address. | +| `--server-port ` | TCP server port. | +| `--tls` | Enable TLS. Requires `--cert` and `--key`. | +| `--cert ` | TLS certificate file. | +| `--key ` | TLS private key file. | +| `--ca ` | CA file for peer verification. | + +## Server Options + +| Option | Description | +|---|---| +| `--stdio` | Serve one SSH connection over standard input/output. | +| `-p ` | TCP listen port. | +| `--tls` | Enable TLS. | +| `--cert ` | TLS certificate file. | +| `--key ` | TLS private key file. | +| `--ca ` | CA file for peer verification. | +| `--destination-root ` | Confine received files to this server-side root; +defaults to the current directory. | +| `--allow-delete` | Permit client delete manifests. Deletion is refused by default. | +| `-v`, `--verbose` | Enable debug logging. | +| `--help` | Print server usage. | + +## Architecture + +### Client + +- Recursively scans the source tree with include, exclude, size, and depth + filters. +- Sends individual files or serialized chunks. +- Performs incremental checks and optional content checksums. +- Uses a multithreaded producer-consumer pipeline when requested. +- Sends over TCP, TLS-wrapped TCP, or an SSH subprocess. +- Supports progress, statistics, backups, timeouts, and bandwidth limiting. + +### Server + +- Runs as a TCP listener or one-shot SSH `--stdio` server. +- Receives and reassembles files and decompresses streaming zstd data. +- Applies supported metadata and writes files through a confined destination + root. +- Uses temporary files and atomic rename by default. +- Handles delete manifests only when explicitly authorized. +- Enforces connection, message-size, and path-safety limits. + +## Protocol and Security + +FastSync protocol version `2.19.0` is shared by the client and server. The +current protocol is sender-driven and includes configuration negotiation, +including the maximum allocation limit, incremental checks, checksums, +manifests, keep-alives, abort handling, per-file remove-source results, and +FastSync-native delta messages. +Client and server versions must currently match exactly. + +Daemon modules that declare `auth users` authenticate with a SCRAM-SHA-256-style +challenge/response against a salted PBKDF2 verifier store: no password and no +replayable bearer credential crosses the wire or is stored on the daemon. All +store entries share one iteration count, and an unknown user is answered with a +deterministic per-username dummy challenge, so probing the daemon cannot +enumerate users. Store lines are generated with +`fastsync-server --hash-credentials ` (see `RSYNC_COMPAT.md`); +redirect that output to an owner-only (mode 0600) file, and note that legacy +`user:SHA256HEX` stores are rejected. FastSync also maintains an owner-only +(mode 0600) `.dummykey` sidecar next to the store: it holds the store-wide +dummy key, is auto-created on first load, and must be preserved across daemon +restarts so the dummy challenge for an unknown user stays stable (the key is +never regenerated while the sidecar exists). The sidecar is secret material and +must be protected like the credential store: keep it owner-only (mode 0600) and +include it with the store in backups and credential rotation. If the sidecar +cannot be created (a process-substitution/FIFO store path such as `/dev/fd/N`, a +read-only filesystem, a missing directory, or a create, write, fsync, link, or +fchmod failure), the daemon logs a warning and uses a transient key, so the +cross-restart guarantee does not hold for those deployments. One residual is +accepted: the store +iteration count is observable pre-auth by design, since the miss path must match +a hit. + +An `auth users` module accepts credentials only when one of two conditions +holds: (a) the connection is an encrypted, verified TLS connection whose client +certificate matches the server's `--client-cn`, or (b) the connection is +plaintext from a loopback peer **and** the operator explicitly passed +`--allow-unauthenticated`. A remote plaintext peer is refused before any +challenge is sent, and `--allow-unauthenticated` never permits remote plaintext +auth: remote peers still require verified TLS regardless of the flag. Clients +sending daemon credentials with `--password-file` to a non-loopback daemon must +therefore use `--tls`; the client rejects a non-local plaintext credential +destination before any network I/O. Daemon modules are a `--daemon`-only +feature: the SSH `--stdio` path never loads a daemon config and is not an auth +transport for them. + +Because the loopback allowance trusts whichever peer the kernel reports as +`127.0.0.1`, it assumes nothing relays remote connections to the daemon. A local +TCP forwarder or a TLS-terminating proxy in front of an auth-module listener +makes remote clients appear as loopback and bypasses the mutual-TLS identity +check, so do not front an auth-module listener with such a relay. `--tls` always +mandates `--client-cn`, so a TLS connection to an auth-required module always +has its client CN verified (`--client-cn` matches the certificate's CN only, not +a subjectAltName, which is acceptable for a private CA). + +TLS provides encrypted TCP transport. Supplying `--ca` enables certificate +verification; without it, traffic is encrypted but peer identity is not +verified. Use certificate verification for deployments where authentication +matters. The default TCP transport is not encrypted. + +The receiver protects its destination root with path validation, `openat()` +directory traversal, `O_NOFOLLOW`, temporary files, and atomic renames. Delete +operations require the server's explicit `--allow-delete` policy. + +## Compatibility Roadmap + +The project will reach the drop-in replacement goal in stages: + +1. Correct rsync option meanings, including short options, combined options, + and `--option=value` syntax. +2. Add differential tests that compare FastSync and rsync contents, metadata, + links, deletes, filters, dry runs, and exit codes. +3. Make `-a` implement the expected recursive, links, permissions, times, + owner/group, and supported special-file behavior. +4. Complete symlink, sparse-file, metadata, delete-policy, and resumable-write + semantics. +5. Add rsync remote-shell and daemon protocol interoperability. +6. Keep FastSync performance options as negotiated, optional extensions. + +The exhaustive implementation matrix and compatibility notes are in +[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md). + ## Testing -```bash -# Unit tests (18 suites — array_list, chunk, compression, config, data, delta, file, glob, -# metadata, property, protocol, queue, robustness, scanner, -# shared_utils, stress, transport_tcp, transport_ssh, transport_tls) -./build/tests +Run the unit test binary: -# Integration + benchmark suite -python3 test.py +```bash +./build/tests ``` -The benchmark prints throughput metrics, best configuration, and speedup vs rsync. +Run the Python integration suite: -## Performance Considerations +```bash +python3 -m pytest tests/ +``` -1. Chunk size (~10 MB default) balances memory and transfer efficiency -2. Compression level trades CPU for bandwidth -3. `sendfile()` bypasses userspace — ~2× faster on localhost for large files -4. Multithreading scales with core count; `--queue-size` controls pipeline buffering -5. Metadata transfer adds negligible overhead (~24 bytes per file when enabled) -6. SSH socketpair buffer set to 1 MB for improved pipe throughput -7. SSH ControlMaster reuses connections across repeated invocations -8. Incremental sync eliminates redundant transfers entirely -9. Batch incremental reduces round-trips by grouping multiple checks into one message -10. Bandwidth limiting uses token-bucket with nanosleep for accurate throttling -11. Atomic writes add a single `rename()` per file — negligible overhead -12. Path traversal check is O(n) in path length with negligible cost +For stricter local validation: -## Benchmark Results +```bash +cmake -B build-strict -S . -DSTRICT_WARNINGS=ON +cmake --build build-strict -j$(nproc) +cmake -B build-asan -S . -DSANITIZER=address +cmake --build build-asan -j$(nproc) +``` -25 MB of mixed file sizes over `localhost` with disk I/O throttled (reads ≤ 15 MB/s, writes ≤ 10 MB/s) and network emulation via `tc netem`. Each test was run 3×; the median is reported below. +The benchmark tool compares FastSync configurations with rsync under +controlled local and network conditions: -### LAN (1000 Mbit, 20 ms ±1 ms, 0.1% loss) +```bash +python3 benchmark/bench.py --help +``` -| Configuration | Time | vs rsync (archive) | vs rsync (compress) | -|---|---|---|---| -| **Best: `-m -c`** | **0.20 s** | **11.2× faster** | **3.6× faster** | -| Compression (`-c`) | 0.31 s | 7.3× faster | 2.3× faster | -| Standard | 1.27 s | 1.8× faster | — | -| rsync (archive) | 2.27 s | — | — | -| rsync (archive + compress) | 0.72 s | — | — | +Benchmark results measure transfer performance only. They do not establish +rsync protocol or filesystem-semantic compatibility. -### WAN (100 Mbit, 50 ms ±10 ms, 1% loss) +## Performance Guidance -| Configuration | Time | vs rsync (archive) | vs rsync (compress) | -|---|---|---|---| -| **Best: `-m -c`** | **0.39 s** | **44.8× faster** | **3.8× faster** | -| Compression (`-c`) | 0.64 s | 27.3× faster | 2.3× faster | -| Standard | 7.12 s | 2.4× faster | — | -| rsync (archive) | 17.44 s | — | — | -| rsync (archive + compress) | 1.47 s | — | — | +- Use `-m` for workloads with many files or enough CPU parallelism. +- Use `-c` or `-z` when network bandwidth is more constrained than CPU. +- Tune `--chunk-size` for file sizes, memory limits, and network latency. +- Use `-f` for large uncompressed TCP transfers where zero-copy I/O helps. +- Use `--incremental` to avoid retransmitting unchanged files. +- Use `--delta` for changed files when both endpoints are FastSync peers. +- Use `--bwlimit` when sharing a link with other traffic. -Compression reduces the data on the wire enough that the transfer becomes latency-bound rather than bandwidth-bound. On WAN, the best configuration runs 10.8× faster than the theoretical limit for uncompressed data, since zstd shrinks the 25 MB payload to a fraction of its original size over the wire. +Always validate the compatibility behavior required by a deployment before +replacing an existing rsync job. diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md new file mode 100644 index 0000000..ef5ead2 --- /dev/null +++ b/RSYNC_COMPAT.md @@ -0,0 +1,886 @@ +# Rsync Feature Compatibility + +This document maps rsync's full feature set to FastSync's current implementation status. + +## Summary + +| Status | Count | Description | +|--------|-------|-------------| +| ✅ Implemented | 143 | Feature works end-to-end | +| 🔀 Alt Arg | 0 | Functionality exists but under different flag/semantics | +| ⛔ Impossible/Divergence | 4 | Flag is a documented divergence or cannot be implemented on any portable filesystem call | +| ⚠️ Partial | 0 | Flag parsed/stored but behavior incomplete | +| 🔄 Compatibility No-op | 0 | Flag is accepted for CLI compatibility but has no effect | +| ❌ Not Implemented | 0 | Flag not recognized or no behavior | +| **Total** | **147** | | + +--- + +## 1. General Options + +| Flag | Rsync Description | FastSync Status | Notes | +|------|-------------------|-----------------|-------| +| `-a`, `--archive` | Archive mode is -rlptgoD | ✅ Implemented | Phase 7 Wave A: real rsync archive. `-a`/`--archive` now implies `--links` + metadata (perms/times/group/owner as FastSync's broad bundle) + `--devices` + `--specials`. FastSync is always recursive, so no `-r` is needed. It no longer implies compression or multithreading (those moved to `-z`/`-j`). The short-option namespace is now rsync-parity (see the Phase 7 note) | +| `-v`, `--verbose` | Increase verbosity | ✅ Implemented | Sets `log_level=DEBUG` | +| `-q`, `--quiet` | Suppress non-error messages | ✅ Implemented | Suppresses client output while preserving errors | +| `--help` | Show help | ✅ Implemented | Prints usage and exits; `-h` is not accepted | +| `-V`, `--version` | Print version | ✅ Implemented | | +| `--info=FLAGS` | Fine-grained info verbosity | ✅ Implemented | Supports `copy`, `misc`, `skip`, `stats`, `all`, and `none`; explicit flags override `--verbose`, and `none` suppresses info output; unsupported names are rejected | +| `--debug=FLAGS` | Fine-grained debug verbosity | ✅ Implemented | `io`, `proto`, `pack`, and `util` are supported; `--debug=help` lists flags; other rsync categories are rejected | +| `--stderr=MODE` | Change stderr output mode | ⛔ Impossible/Divergence | `errors` (default) and `all` are supported; `client` is rejected with a clear error (`--stderr=client is not supported`) because FastSync has no rsync client-message channel — the rejection itself is the documented behavior (Phase 7 Wave B decision). The modes that exist work; the missing rsync channel cannot be emulated without a wire change | +| `--no-motd` | Suppress daemon MOTD | ✅ Implemented | Client-only display switch (Wave C): the daemon still sends the configured `motd file` on a `host::module/path` connection; the client reads and discards the frame without showing it. Without the flag the MOTD is printed to stdout after the config/auth handshake and escaped so control bytes cannot inject terminal sequences | +| `--exclude=PATTERN` | Exclude files matching pattern | ✅ Implemented | Glob matching in scanner | +| `--include=PATTERN` | Include files matching pattern | ✅ Implemented | Glob matching in scanner | +| `-C`, `--cvs-exclude` | Auto-ignore CVS files | ✅ Implemented | Applies the well-known rsync default exclude set as exclude rules during scanning (RCS SCCS CVS CVS.adm RCSLOG cvslog.* tags TAGS .make.state .nse_depinfo *~ #* .#* ,* _$* *$ *.old *.bak *.BAK *.orig *.rej .del-* *.a *.olb *.o *.obj *.so *.exe *.Z *.elc *.ln core .svn/ .git/ .hg/ .bzr/); `.git/`-style repo dirs are pruned without descending | + +## 2. Modifying Output + +| Flag | Rsync Description | FastSync Status | Notes | +|------|-------------------|-----------------|-------| +| `--stats` | Give transfer stats | ✅ Implemented | Prints file/byte counts | +| `-h`, `--human-readable` | Human-readable numbers | ✅ Implemented | Formats transfer byte sizes using binary units | +| `-i`, `--itemize-changes` | Per-file change summary | ✅ Implemented | Prints rsync-style `>f+++++++++` lines to stdout only for files actually sent (also under `-j`/`--threads`); unchanged files print nothing, matching single-`-i` behavior | +| `--progress` | Show progress | ✅ Implemented | Progress callback in sender | +| `-P` | Same as --partial --progress | ✅ Implemented | Phase 7 Wave B: `-P` parses to `--partial` + `--progress`. On a failed/interrupted write the receiver now retains the already-written temp file at the destination path (best-effort rename instead of unlink when configured), so a later `--append`/`--append-verify` run can resume it; `--partial-dir` still stages completed files under the confined partial dir and installs them atomically. The retention never runs when `--partial` is off, when no data was actually written, or under `--ignore-existing`/`--existing` (the destination is not ours to overwrite), and it only ever renames the already-written temp (never a corrupt blend; a failed rename falls back to the normal unlink). See the `-S`/`--sparse` interplay note (a retained sparse temp has full logical size) | +| `--out-format=FORMAT` | Custom output format | ✅ Implemented | Per-transfer template on stdout; tokens `%f` `%n` `%l` `%b` `%M` `%%` (`%b` is the source length, always `== %l`; post-compression/delta wire bytes are not counted); unknown escapes preserved | +| `--log-file=FILE` | Log to file | ✅ Implemented | `log_file` config field | +| `--log-file-format=FMT` | Log format | ✅ Implemented | Requires `--log-file`; writes one template line per transferred file using the same token set as `--out-format` (including `%b` `==` source length) | +| `--8-bit-output`, `-8` | Leave high-bit chars unescaped | ✅ Implemented | Applies to displayed paths and protocol debug output | +| `--list-only` | List files instead of copying | ✅ Implemented | `ls -l`-style listing of files that would be transferred; scans the source only, contacts no server, writes nothing; also works with `-n` | + +## 3. File Selection + +| Flag | Rsync Description | FastSync Status | Notes | +|------|-------------------|-----------------|-------| +| `--exclude-from=FILE` | Read exclude patterns from file | ✅ Implemented | Reads patterns from file | +| `--include-from=FILE` | Read include patterns from file | ✅ Implemented | Reads patterns from file | +| `--filter=RULE` | Add file-filtering rule | ✅ Implemented | Long option only: rsync's short `-f` conflicts with FastSync sendfile (see FastSync-specific list), so `-f` is not reassigned. Supported subset: `+`/`-` include/exclude, implicit-exclude patterns, `include`/`exclude` word forms, a leading `/` anchor (to the transfer root, or to a `.rsync-filter` file's directory), and a trailing `/` for dir-only rules; first match wins with a default of include inside the filter layer. Filters are an independent layer from `--exclude`/`--include` (an entry must pass both). Rejected with a clear error (no silent no-ops): `merge`/`dir-merge`/`hide`/`show`/`protect`/`risk`/`clear` words, rules that begin with `:`/`.`/`!` (merge/dir-merge/list-clear shorthands), and include/exclude modifiers other than `/` (`! C s r p x`) | +| `--files-from=FILE` | Read source file list from file | ✅ Implemented | Entries are paths relative to the source root (leading `./` stripped, `..`/absolute entries rejected at parse time, blank lines ignored; NUL-delimited with `-0`). A listed regular file is transferred; a listed directory transfers its whole subtree (FastSync recursion is always on, unlike rsync's non-recursive default). Non-listed paths and their subtrees are pruned by the scanner. A listed entry that does not exist under the source (and an empty list) is a hard error reported before any transfer, unless `--ignore-missing-args` / `--delete-missing-args` is given (see the Safety & Security rows): those flags downgrade the listed-but-missing case to a skip and, for `--delete-missing-args`, a destination deletion; an empty list stays a hard error in every mode. Listing `.` (whole tree) and empty listed directories are fine. Scalability note: `file_list_affects` is O(list size) per scanned entry, so a very large `--files-from` list against a huge tree is quadratic; lists are typically small enough that this is acceptable, but it is the documented bound. The delete manifest still derives from what was actually sent, so `--delete` stays consistent with the subset | +| `-0`, `--from0` | Delimit *-from files with NULs | ✅ Implemented | `--files-from` entries become NUL-delimited; the flag may appear before or after `--files-from` on the command line. NUL mode preserves entry bytes exactly (trailing CR/LF are part of the name; only newline mode trims them) | +| `--max-size=SIZE` | Skip files larger than SIZE | ✅ Implemented | `max_size` in scanner | +| `--min-size=SIZE` | Skip files smaller than SIZE | ✅ Implemented | `min_size` in scanner | +| `-I`, `--ignore-times` | Don't skip files matching size+time | ✅ Implemented | `ignore_times` config field (crosses the wire). Disables the size+mtime quick-check in the `--incremental` per-file handshake and the basis-dir quick-match, forcing the file to be transferred rather than skipped as unchanged. Receiver-side policy: `match_by_metadata` (file_receive.c) is bypassed, so the receiver never replies `STATUS_OK` for a matching size+mtime. Requires `--incremental` to have the handshake to act on (rsync does its quick check by default; FastSync's `-I`/`--size-only`/`--modify-window` only take effect under `--incremental`, exactly like they take effect through the basis check) | +| `--size-only` | Skip based on size only | ✅ Implemented | With `--incremental`, ignores mtime | +| `-@`, `--modify-window=NUM` | Mod-time comparison accuracy | ✅ Implemented | Whole-second tolerance with nanosecond-aware comparisons | +| `--existing` | Skip creating new files on receiver | ✅ Implemented | Existing destination files continue through normal update handling | +| `--ignore-existing` | Skip updating existing files | ✅ Implemented | `ignore_existing` config field (crosses the wire; receiver-side policy). For a destination entry that already exists, the receiver skips the write: in the regular-file path, existing/delay-updates-staged, hardlink-sibling, and special/device handlers all return `FILE_SAVE_SKIPPED` without overwriting (passed as `no_replace` to the write engine), and `--backup` is disabled for skipped files. Note: it is applied at write time, so an existing dest whose size+mtime differ still has its data (or delta) transmitted before the write is discarded — functionally correct, bandwidth-suboptimal vs rsync, which short-circuits earlier. Like rsync, it does not apply to directories/symlinks (those return before the block). Combines with `-j`/`--threads` and `--delay-updates`. See Phase-4/— notes below | +| `--remove-source-files` | Sender removes regular files after confirmed transfer | ✅ Implemented | | +| `-x`, `--one-file-system` | Do not cross filesystem boundaries | ✅ Implemented | Sender scanner captures the root device and skips descending into mount-point crossings (`st_dev` differs); cross-filesystem mount-point subdirectories are dropped entirely, matching rsync | +| `-F` | Add the default `.rsync-filter` rules | ✅ Implemented | Reads one filter rule per line from each directory's `.rsync-filter` file during traversal and applies it to that directory's subtree; the current directory's rules are evaluated before its ancestors', so deeper files override shallower ones and per-directory files override the command-line `--filter`/`-C` base by default (matching rsync's first-match-wins precedence); `.rsync-filter` files are never transferred. The rsync `-FF` behavior (also `.cvsignore`) is out of scope; unsupported rule types inside the file abort with a clear error | + +## 4. Directory Options + +| Flag | Rsync Description | FastSync Status | Notes | +|------|-------------------|-----------------|-------| +| `-r`, `--recursive` | Recurse into directories | ✅ Implemented | Default behavior | +| `-R`, `--relative` | Use relative path names | ✅ Implemented | Meaningful together with `--files-from` (FastSync's default full-tree scan always mirrors the full source argument path below the destination root, so -R does not change it). With `-R` + `--files-from` each listed entry is transmitted under its bare relative destination path: an entry `sub/x.txt` lands at `/sub/x.txt` (its leading components preserved) instead of under the `/` mirror. Only the path sent on the wire changes; the client still reads the absolute source path, and the delete manifest derives from the sent (relative) paths so `--delete` and `--remove-source-files` stay consistent in both layouts. Works single-threaded and under `-j`/`--threads` (including chunk serialization) | +| `--no-implied-dirs` | Don't send implied dirs with -R | ✅ Implemented | Client-side, meaningful only with `-R` + `--files-from`. rsync would normally create the ancestor directories implied by a listed file so it can be written; with `--no-implied-dirs` a listed file whose parent directory is not itself (or via an ancestor) explicitly listed cannot be placed, and FastSync fails the whole run up front with a clear error (`--no-implied-dirs: cannot place file '...': parent directory '...' is not explicitly listed`). Listing the directory (or an ancestor of it, or the whole tree `.`) permits the file. In every other mode the option has no effect. FastSync has no per-entry skip channel, so the rsync "omit the file" case is surfaced as a hard pre-transfer error | +| `-d`, `--dirs`, `--old-dirs`, `--old-d` | Transfer dirs without recursing | ✅ Implemented | `-d ` transmits an explicit directory entry for the source-root directory, so the destination mirror is created empty and nothing is descended into. With `--files-from` exactly the listed items are transferred: a listed directory is created empty (no descent) and a listed file is transferred with its content; the dest layout follows the same -R rules as plain files. A new wire frame (`STATUS_MKDIR`) carries each directory entry — the path and, when `--preserve`/`-a` (metadata mode) is negotiated, the directory's metadata; the receiver creates it with the same confined mkdir-parent semantics as regular writes, in single-threaded and `-j`/`--threads` receivers (chunk serialization carries a per-entry type marker). Directory entries appear in the delete manifest so `--delete` prunes correctly. Directory TIMES are transmitted (the `STATUS_DIR_TIMES` frame carries every traversed source directory's captured times, including `--dirs` entries) and applied by the receiver at the END of the transfer, after all children and the delete/publication phases, so a later child write cannot clobber a directory's mtime (`-O`/`--omit-dir-times` skips this application). FastSync divergences: directory modes/ownership are still not applied (only times are), and empty directories are still never created (a `STATUS_DIR_TIMES` entry is record-only), filter/`--exclude` rules are not re-applied to the listed dirs mode (there is no descent during which they would apply), and `-d` never creates the intermediate directories between the destination root and a listed file beyond the usual on-demand parent creation. Under `--delay-updates` only regular files are staged: directory entries are created immediately, so a delayed run that fails part way can leave the already-created empty directories behind (matching rsync, which also creates directories as it processes the file list and only delays regular-file data) | +| `--mkpath` | Create missing path components | ✅ Implemented | Wire option (client → server). At connection start the server creates the client's destination root directory (and any missing leading components below its own authorized root) when `--mkpath` is set, failing the connection cleanly if it cannot. Without `--mkpath` a destination root that does not exist yet is rejected up front (rsync semantics), so the flag is the only way to transfer into a not-yet-created destination directory. Creation is confined by the same secure mkdir walk as file writes (`O_NOFOLLOW`, no `..`) | + +## 5. Transfer Modifications + +| Flag | Rsync Description | FastSync Status | Notes | +|------|-------------------|-----------------|-------| +| `-u`, `--update` | Skip files newer on receiver | ✅ Implemented | `update` config field (crosses the wire; receiver-side policy, implies `-M` metadata). Before writing a regular file, the receiver checks `file_destination_is_newer_secure()` (via `stat_is_newer`, second-then-nanosecond strict `>` on the existing destination) and skips the write when the destination is newer than the source (`FILE_SAVE_SKIPPED`); equal-or-older destination (or a newer source) is transferred normally. Applied at write time on the regular-file, delay-updates-staged, hardlink-sibling, and special/device paths. Only regular destinations can be guarded (the newer-check requires `S_ISREG`), and like the other write-time policies it does not short-circuit the data transfer for a differing-size dest. `--remove-source-files` correctly respects the receiver's skip outcome so a skipped source is not removed | +| `--inplace` | Update files in-place | ✅ Implemented | Direct write mode | +| `--append` | Append data to shorter files | ✅ Implemented | Tail-only resume. When an existing destination file is SHORTER than the source, the receiver negotiates a resume offset with the sender and only the tail is transferred; the receiver rebuilds the full file (retained prefix + tail) and installs it through the normal atomic store path, so the result is byte-identical to the source whenever the retained prefix matches. Plain `--append` does NOT content-verify that prefix (rsync parity): a destination whose prefix differs from the source is resumed anyway, so the result (wrong prefix + correct tail) is NOT byte-identical and the file is effectively left corrupt — the documented rsync-parity risk (use `--append-verify` when the prefix cannot be trusted). Non-content attributes (permissions/ownership/mtime, via `-M`) are still applied. Requires the per-file `STATUS_CHECK` handshake, so it implies `--incremental`; it takes precedence over block delta for a growing file and falls back to delta/full when the destination is not shorter. Incompatible with `-s` (chunk serialization) and `--whole-file` (both rejected up front so the mode never silently degrades to a full transfer). Combines with `--inplace`, `--partial`/`--partial-dir`, and `--delay-updates` (the reconstructed full file flows through those paths unchanged). Divergence: rsync appends in place; FastSync reconstructs and atomically installs, so an interrupted or failed resume never leaves a half-written file at the destination (no corruption window), and `--append` is thus safe to use with the normal atomic path — not only with in-place writes | +| `--append-verify` | Append with old-data checksum | ✅ Implemented | Like `--append`, but the retained prefix IS verified before resuming: the sender transmits the source prefix checksum and the receiver compares it to the xxHash64 of the retained destination prefix; on a match only the tail is transferred, on a MISMATCH the run falls back to a clean full transfer so the result is always a byte-identical source copy (never a corrupt prefix+tail blend). Wire/protocol: the append handshake adds `STATUS_APPEND` / `STATUS_APPEND_SIG` / `STATUS_APPEND_OK` / `STATUS_APPEND_DATA` frames and `PROTOCOL_VERSION` was bumped **2.9.0 → 2.10.0** (peers must match, and both must be 2.10.0 or the run fails the version check). Same implications/incompatibilities as `--append`; when both spellings are given `--append-verify` wins (the safer semantics). See the Phase-3 append notes below | +| `-W`, `--whole-file` | Copy whole file (no delta) | ✅ Implemented | `whole_file` config field. Forces a full (whole-file) copy, disabling the block-level delta machinery: the sender only sends `STATUS_NEXT` + full data (client_send.c) and the receiver never requests a delta signature/reconstruction — the receiver's `try_delta = use_delta && !whole_file && ...` short-circuits. `whole_file` crosses the wire folded into `use_delta` (the wire carries `use_delta && !whole_file`), so no separate field/bump is needed. Delta is opt-in (`--delta` needs `--incremental`); `-W` additionally makes `--fuzzy` inert (no similar-file delta basis). `--append`/`--append-verify` are incompatible with `-W` and rejected up front (both sides). See the delta/append notes below | +| `--block-size=SIZE` | Force checksum block-size | ✅ Implemented | Phase 7 Wave B: `--block-size` is an alias for `--delta-block`; both set `config->delta_block_size` (default `DELTA_BLOCK_SIZE_DEFAULT`, bounds `DELTA_BLOCK_SIZE_MIN..MAX`, out-of-range values are rejected with the default kept). The value is genuinely honored by the delta engine end-to-end: `delta_signature_create_seeded(old, size, config->delta_block_size, seed)` on the sender and receiver, `delta_apply(old, ...)` with the same size, so a non-default block size changes the block count of every signature the harnesses exchange (verified by unit + integration tests) | + +## 6. Destination Handling + +| Flag | Rsync Description | FastSync Status | Notes | +|------|-------------------|-----------------|-------| +| `-n`, `--dry-run` | Trial run with no changes | ✅ Implemented | `dry_run` config field | +| `-b`, `--backup` | Make backups of overwritten files | ✅ Implemented | Backup before overwrite | +| `--backup-dir=DIR` | Backup directory hierarchy | ✅ Implemented | `backup_dir` config field | +| `--suffix=SUFFIX` | Backup suffix (default ~) | ✅ Implemented | `suffix` config field | +| `--delay-updates` | Put updated files in place at end | ✅ Implemented | Successfully received files are staged under a private 0700 `.fastsync-stage` dir inside the receive root and atomically renamed into their final destinations only after the whole transfer (manifest/delete handling included) succeeds, just before the success/outcome frame is sent. The delete walker deliberately skips the staging dir at the receive root, so `--delete` removes genuine extras but never the staged files (deletion runs before publication; rsync's delete-after ordering is not implemented). `--existing`/`--ignore-existing`/`--update` decide against the final destination path at stage time; `--backup` moves the old file aside at publication. Incompatible with `--inplace` and with `--backup-dir=.fastsync-stage` (the internal staging name is reserved; both are rejected). The staging dir name is fixed, so two simultaneous delayed transfers to the same destination root are serialized with an exclusive advisory lock held for the whole transfer: the second session fails cleanly instead of corrupting the first. Aborting or failing before publication installs nothing and removes the staging tree; a crash between stage and publish leaves staged leftovers that the next delayed run wipes at start (process death releases the lock). A stage→publish failure aborts the transfer (best-effort cleanup of the not-yet-published staged files; already-published files are not rolled back). Works in single-threaded and `-j`/`--threads` modes | +| `-T`, `--temp-dir=DIR` | Create temporary files in DIR | ✅ Implemented | `--temp-dir` with the rsync short `-T` (Phase 7 Wave A; the timeout alias moved to long-only `--timeout`). Scratch dir is resolved under the receive root; temp copies use a unique name there and are atomically renamed into place. If the scratch dir and destination are on different filesystems the atomic rename fails with EXDEV and the file save fails, which aborts the whole transfer (FastSync has no per-file skip/resume on a save error; rsync's non-atomic copy fallback is deliberately not used). `--inplace` and `--partial-dir` writes bypass the scratch dir | + +## 7. Deletion + +| Flag | Rsync Description | FastSync Status | Notes | +|------|-------------------|-----------------|-------| +| `--delete` | Delete extraneous files from dest | ✅ Implemented | `use_delete` config field. Deletion is always derived from the transmitted keep-set manifest of the paths the sender sent/keeps (never from unchecked input), runs through the symlink-safe walker bounded by `MAX_SERVER_DELETE_COUNT`, and skips the `.fastsync-stage` staging dir under `--delay-updates`. FastSync's default timing when no timing flag is given is **delete-after** (extras are removed only once the whole transfer succeeded) — intentionally NOT rsync's `--del`/delete-during default, to preserve FastSync's commit-style safety. By default the destination mirror of a path the source scan pruned (filter/exclude/size rules) is **protected** from deletion — matching rsync, which does not delete excluded files under `--delete`; `--delete-excluded` opts back into deleting them (see below). The bounded deletion is **all-or-nothing**: if the destination holds more extras than the effective bound (a client `--max-delete=NUM` or the 100000-entry server bound) nothing is deleted and the run fails with a distinct error instead of silently truncating | +| `--delete-before` | Delete before transfer | ✅ Implemented | Implies `--delete`. The sender runs a full source pre-scan (paths only) and transmits the keep-set manifest BEFORE any file data; the receiver validates it, removes every destination entry not listed (all-or-nothing bounded walk, staging-dir skip, protected prefixes honored), then acks `STATUS_OK`. The sender only starts streaming after the deletion committed, or aborts if the receiver reported a deletion error. By definition the deletions already happened when a later transfer phase fails — rsync's delete-before is destructive the same way; a subsequent failure does not restore the removed files. Divergence: the keep-set is the pre-scan snapshot, so a file that appears on the source between the pre-scan and the data pass is still transferred but was not protected from deletion | +| `--del`, `--delete-during` | Delete during transfer | ✅ Implemented | Both spellings accepted; imply `--delete`. FastSync streams the source in a single directory scan and has no per-directory generator pass, so deletions cannot be interleaved per-directory the way rsync's delete-during does. `--delete-during` therefore selects the same early engine mode as `--delete-before` (manifest transmitted before any data, extras removed and acknowledged before data is applied); observable success/failure behaviour equals `--delete-before`. That is the documented divergence from rsync, where `--del` is the default meaning of `--delete` | +| `--delete-delay` | Find deletions during, delete after | ✅ Implemented | Implies `--delete`. Commit-mode timing: extras are removed only after the whole transfer succeeded. rsync's delete-delay records the deletion list during its scan and applies it at the end; FastSync never snapshots the destination while data flows (the keep-set is the transmitted manifest and the destination is listed only at deletion time), so `--delete-delay` is implemented as the same end-of-transfer commit as `--delete-after` with identical safety. That is the documented divergence | +| `--delete-after` | Delete after transfer | ✅ Implemented | Implies `--delete`. The delete-after timing is also what plain `--delete` does: the keep-set manifest closes the data stream and the receiver commits the bounded deletion only after the terminal `STATUS_FINISHED` proves the whole transfer (every data frame received and stored) succeeded. A failed or aborted transfer removes nothing | +| `--delete-excluded` | Also delete excluded files | ✅ Implemented | `delete_excluded` config field. Under `--delete` FastSync now protects (rsync's default) the destination mirror of paths the sender's source scan pruned by user-selection rules — the `--filter`/`-F`/`-C` layer, the legacy `--exclude`/`--include` layer, and `--max-size`/`--min-size`. The sender transmits those concrete pruned paths as **protected prefixes** in the delete-manifest frame (see the Phase-3 notes below); the walker never descends into or removes them. `--delete-excluded` opts back in: the sender sends an empty protected list, so the excluded destination mirrors become ordinary extras and are removed. Divergences (documented): protection is derived only from what the source scan actually pruned — a stray destination-only file that happens to match an exclude rule is not protected (FastSync never re-applies rules to the destination, keeping deletion sender-derived), and `--files-from` subset pruning stays keep-set-only (an unlisted source path is treated as absent and its mirror is deletable, matching the `--files-from` delete note below). The two are orthogonal: `--delete-excluded` removes filter-excluded mirrors; it does not make `--files-from` prune things | +| `--max-delete=NUM` | Max files to delete | ✅ Implemented | `max_delete` config field (default -1 = no client limit; 0 = delete nothing). NUM bounds a `--delete` run with rsync's all-or-nothing semantics: the receiver rehearses the deletion first and, if the destination holds more than NUM extras, deletes NOTHING and fails the transfer with a distinct `--max-delete` error. A run at or below NUM deletes exactly the extras. NUM only applies together with `--delete` (it is inert otherwise, matching rsync). The hard server bound `MAX_SERVER_DELETE_COUNT` (100000) still caps the walk; a NUM above it never raises that cap, and exceeding the server bound is its own all-or-nothing error. Directories count toward the limit (each removed empty directory is one deletion), like rsync | +| `--ignore-errors` | Delete even with I/O errors | ✅ Implemented | Sender-side, client-only config field. rsync suppresses `--delete` when the transfer had I/O errors; FastSync's equivalent is a source-scan I/O error (an unreadable directory, e.g. EACCES): by default the scan aborts the run so no deletion happens. With `--ignore-errors` the scan continues past the unreadable directory, the readable tree is transferred and the deletion still runs (the mirror of the unreadable directory is treated as an extra). The run still exits non-zero (the error is reported, matching rsync's error status). Divergence: without the flag FastSync aborts the whole run on the scan error, whereas rsync transfers the rest of the tree and merely skips the deletion; both leave the deletion undone | +| `--force` | Force deletion of non-empty dirs | ✅ Implemented | `force_delete` receiver config field (crosses the wire). rsync's `--force` lets an incoming non-directory replace a destination directory; FastSync implements exactly that: when a regular file is written to a path that is currently a (possibly non-empty) destination directory, `--force` removes that directory tree first — confined to the receive root and symlink-safe (O_NOFOLLOW fd walk, symlinks removed by name, never followed) — so the atomic install can place the file. Without `--force` such a write fails and the run aborts. Divergence: `--force` acts on the immediate-install path only; under `--delay-updates` a blocking directory is not cleared (publication renames over regular files) | +| `-m`, `--prune-empty-dirs` | Prune empty dir chains | ✅ Implemented | `-m`/`--prune-empty-dirs` (Phase 7 Wave A freed the rsync short `-m`; FastSync multithreading is now `-j`/`--threads`). FastSync's recursive transfer records directory times but never CREATES an empty directory (a `STATUS_DIR_TIMES` entry is record-only, and `--dirs` empty entries are pruned by this flag), so empty directories are inherently never transferred (which is rsync's `-m` behavior) and truly-empty destination directory chains are removed by `--delete` regardless of this flag. The flag's additional real effect is on the `--dirs` explicit directory-entry generator: a plain `-d ` run omits the empty source directory's entry, so nothing is created at the destination (no `STATUS_MKDIR`, no `-i`/`--out-format` change line, and an existing empty mirror becomes an extra that `--delete` prunes). Explicitly `--files-from`-listed directories always pass through (documented `--files-from` behavior). A directory that still holds an excluded-but-protected file survives, matching the `--delete-excluded` default | + +**Deletion-timing implementation notes (Phase 3):** the delete flags above are +real. Two new config booleans (`delete_during`, `delete_delay`) join the already +serialized `delete_before`/`delete_after`, so the on-the-wire config layout +changed and `PROTOCOL_VERSION` was bumped **2.7.0 → 2.8.0** (peers must match). +The `STATUS_MANIFEST` frame is count-delimited and position-independent: the +receiver commits the deletion either when the manifest arrives (early modes: +`--delete-before`/`--delete-during`, which additionally acknowledge with +`STATUS_OK` before data flows) or after the terminal `STATUS_FINISHED` proves +the whole transfer succeeded (commit modes: plain `--delete`/`--delete-after`/ +`--delete-delay`). Timing is chosen purely from the config, so server policy +(`--allow-delete` off) still disables deletion without deadlocking the early +manifest ack. `--delete-delay` and `--delete-during` are each implemented as +the closest safe approximation their engine mode allows; the divergences are +noted in the rows above. + +**Deletion-policy notes (Phase 3, delete-policy wave):** this wave made the +deletion family real — `--delete-excluded`, `--max-delete`, `--ignore-errors`, +`--force`, `--prune-empty-dirs` — and, to support them, the `STATUS_MANIFEST` +frame now carries **two sections**: the keep-set paths followed by a list of +**protected prefixes** (destination-relative paths the source scan pruned by +user-selection rules, which the walker must never delete unless +`--delete-excluded` opted out). Two config booleans were added for the wave: +`force_delete` (crosses the wire; the receiver clears a directory that blocks an +incoming file) and `ignore_errors` (client-only; the sender's scan continues +past an unreadable directory). `max_delete`'s default became -1 ("no client +limit"). These wire/layout changes bumped `PROTOCOL_VERSION` **2.8.0 → 2.9.0** +(peers must match). All four wire additions — `force_delete`, +`delete_excluded`, `prune_empty_dirs`, `max_delete` — round-trip unchanged and +are validated on receive. + +**Missing-args note (Phase 3, missing-args wave):** `--ignore-missing-args` and +`--delete-missing-args` are implemented as described in the Safety & Security +rows. Wire impact: the `STATUS_MANIFEST` frame now carries a **third section** — +a list of destination-relative **exact-delete paths** (the missing entries' +mirrors) — and the config frame gained a `delete_missing_args` boolean +(`ignore_missing_args` stays client-only, exactly like `ignore_errors`). These +wire/layout changes bumped `PROTOCOL_VERSION` **2.9.0 → 2.10.0** (peers must +match). The receiver validates the third section identically to the keep-set +(non-empty, relative, traversal-free; `MAX_MANIFEST_ENTRIES` per section, a +single `MAX_MANIFEST_BYTES` budget shared across all three). On commit the +receiver runs the exact-path deletions FIRST (`manifest_delete_missing_args`: +confined per-path unlink/rmdir, deep removal only under `--force`/`--delete`, +staging/basis protected, never blocked by the protected-prefix list) and then +the ordinary extras walk when `--delete` is active (`manifest_delete_all`). A +client may request the exact-path deletions without `--delete`; the server's +`--allow-delete` policy gates them exactly like `--delete`, so an unauthorized +server ignores the request while the missing entries are still skipped. + +The deletion walker is now **all-or-nothing**: before any unlink it rehearses +the deletion (an fd-relative walk identical to the delete pass, counting every +regular file it would unlink and every directory it would remove) and refuses to +start when the extras exceed the effective bound — a client `--max-delete=NUM` +below the hard bound, or the hard `MAX_SERVER_DELETE_COUNT` (100000) bound +itself. Previously the walker removed up to `MAX_SERVER_DELETE_COUNT` extras and +then reported an error (a truncated deletion); it now removes nothing and fails +with an error naming the bound. Directories count toward the bound. A directory +that still holds entries the walker leaves in place (a protected excluded file, +a kept manifest entry, a symlink) is left behind rather than failing the run — +matching rsync's "cannot delete non-empty directory" behaviour. The +all-or-nothing guarantee holds only while the destination is not concurrently +modified: rehearsal and delete are two separate walks, so a concurrent change +between them (another process adding or removing destination entries) can make +the actual deletion diverge from the counted set. + +Manifest size: the sender's keep-set and protected-prefix collections (streaming +or early pre-scan) are unbounded, but the receiver rejects a manifest beyond +`MAX_MANIFEST_ENTRIES` (1 048 576 entries, applied to EACH section — a frame can +therefore total up to 2 097 152 entries) / `MAX_MANIFEST_BYTES` (16 MB of paths, +counted across BOTH sections) as a hard protocol error. A heavily filtered +source whose exclusion list grows large thus fails the run cleanly on the +receiver (STATUS_ERROR) instead of being silently truncated. In the commit +modes this only means the deletion is refused after the data already arrived; in +the early modes (`--delete-before`/`--delete-during`) the manifest is the first +frame, so an oversized keep-set or protected list aborts the whole transfer +BEFORE any data is sent. Keep the source tree small enough for the receiver's +manifest caps when using the early timing. + +Early-delete ACK wait: after committing a large deletion (up to +`MAX_SERVER_DELETE_COUNT` removals) the receiver's `STATUS_OK`/`STATUS_ERROR` +reply can legitimately take much longer than a normal round trip, so the sender +waits for that single ACK with an extended explicit deadline (1 hour) instead +of the default 60 s per-message receive window. A receiver that is genuinely +gone still aborts the wait via connection close/error; the extended bound only +protects against aborting after the deletion already committed on the receiver. + +Flag-conflict policy: unlike rsync's last-one-wins behaviour, every deletion +timing flag implies `--delete`, and combining a timing flag with `--no-delete` +(in either argument order) — or more than one timing flag — is rejected as a +configuration error rather than silently resolved. Note the check is +order-independent because it runs over the fully parsed config. The deletion +POLICY flags (`--delete-excluded`, `--max-delete`, `--ignore-errors`, `--force`) +do NOT imply `--delete`; without `--delete` they are inert (matching rsync). + +**Append-resume notes (Phase 3, append wave):** `--append` and `--append-verify` +are real. Both are negotiated when an existing destination file is found to be +**shorter** than the source during the per-file `STATUS_CHECK`; the receiver +replies with a new `STATUS_APPEND` frame carrying the resume offset (the prefix +length it already holds) instead of `STATUS_NEXT`/`STATUS_DELTA_SIGNATURE`. +The sender transmits ONLY the tail. For `--append-verify` it first sends the +source's prefix xxHash64 in a `STATUS_APPEND_SIG` frame; the receiver compares +it to the retained prefix and answers `STATUS_APPEND_OK` (transfer the tail) or +`STATUS_NEXT` (prefix mismatch → the sender falls back to a byte-exact full +transfer). The tail arrives in a `STATUS_APPEND_DATA` frame (compression and +metadata still apply). The receiver then rebuilds the full file in memory +(prefix + tail) and routes it through the existing atomic store engine, so all +of `--inplace`, `--partial`/`--partial-dir`, `--delay-updates`, `--backup`, +`--existing`/`--ignore-existing`/`--update` and delete-manifest behaviour is +unchanged and the result is a byte-identical source copy (given a matching +prefix). These new frames changed the wire, so `PROTOCOL_VERSION` was bumped +**2.9.0 → 2.10.0** (peers must match; the pre-existing `append`/`append_verify` +config booleans already crossed the wire). CLI: both flags imply `--incremental` +(the handshake needs it); they are incompatible with `-s` (chunk serialization) +and `--whole-file` (both rejected up front, never a silent full transfer); when +both spellings are given `--append-verify` wins. The FastSync divergence from +rsync is intentional and safer: rsync appends in place, whereas FastSync +reconstructs the whole file and atomically installs it, so an interrupted or +failed resume never leaves a partial/corrupt file at the destination — this is +why plain `--append` works on the normal atomic path, not only with `--inplace`. + +## 8. Metadata Preservation + +| Flag | Rsync Description | FastSync Status | Notes | +|------|-------------------|-----------------|-------| +| `-M`, `--preserve` | Preserve file metadata | ✅ Implemented | Mode, uid, gid, mtime | +| `-p`, `--perms` | Preserve permissions | ✅ Implemented | Phase 7 Wave A: `-p`/`--perms` now preserve permission bits, folded into FastSync's broad metadata bundle (`--preserve`); the SSH port moved to `--ssh-port`. rsync-parity short form | +| `-o`, `--owner` | Preserve owner | ✅ Implemented | Part of -M | +| `-g`, `--group` | Preserve group | ✅ Implemented | Part of -M | +| `-t`, `--times` | Preserve modification times | ✅ Implemented | Part of -M | +| `-E`, `--executability` | Preserve executability | ✅ Implemented | Preserves executable permission bits (implies metadata preservation) | +| `--chmod=CHMOD` | Affect file permissions | ✅ Implemented | Supports numeric and symbolic `ugo` `rwx` changes; retains receiver safety masking | +| `-A`, `--acls` | Preserve ACLs | ✅ Implemented | Implemented on Linux via the POSIX-ACL xattr representation: the sender captures the `system.posix_acl_access` / `system.posix_acl_default` xattrs into the same bounded whitelisted set as `-X`, transmits them per-file, and the receiver re-applies them fd-relative. Setting an ACL the receiver is not permitted to set (non-root on a file it does not own, unsupported filesystem) is logged and skipped, never fatal. libacl is **not** required. Only the `system.posix_acl_*` namespaces plus `user.*` are ever applied; privileged namespaces are never applied (see the Phase-4 xattr/ACL notes below). Implies metadata transmission | +| `-X`, `--xattrs` | Preserve extended attributes | ✅ Implemented | Preserves unprivileged `user.*` extended attributes (Linux `listxattr`/`getxattr` on capture, `fsetxattr` on the written destination fd). Both capture (sender) and application (receiver) are restricted to the `user.*` namespace and the two POSIX ACL xattrs, so a client can **never** force a `security.*`/`trusted.*`/privileged attribute onto the destination; the receiver independently re-validates every incoming name against this whitelist and rejects anything else. Payloads are bounded (per-name ≤255B, per-value ≤1MiB, per-file count ≤256 total bytes ≤4MiB) on both ends, and an oversized/malformed frame is a clean protocol rejection (no OOM). Applied fd-relative to the exact written file. Implies metadata transmission. Incompatible with `-s` (chunk serialization), rejected up front (see the notes); a `--link-dest`/`-H` hard-link copy fallback re-applies the attributes so they are not dropped when a link is refused | +| `-H`, `--hard-links` | Preserve hard links | ✅ Implemented | Files on the source that share an inode (`st_dev`+`st_ino`, e.g. a `cp -al` tree) are re-created as hard links to one another on the destination, so duplicate links stay deduplicated and only the first member's data is sent (later members are transmitted as payload-less `STATUS_HARDLINK` frames). The receiver links each sibling to the first member's installed file with an atomic link + rename; on `link()` failure it falls back to a byte-identical local copy of the first member, never a partial/corrupt file. Requires the sequential scan for ordering (the first member is always emitted and installed before any sibling is linked). Works single-threaded and under `-j`/`--threads`, `--inplace`, `--delay-updates` (links staged and published by rename) and `--partial`. Crosses the wire (`preserve_hard_links` bool; `PROTOCOL_VERSION` bumped **2.11.0 → 2.12.0**, peers must match). Incompatible with `-s` (chunk serialization) and `--append`/`--append-verify`, rejected up front with a distinct error. See the Phase-4 hard-links notes below | +| `-D` | Same as --devices --specials | ✅ Implemented | Implies `--devices --specials`. `-D` was unassigned in FastSync (verified: no collision), so it is free to imply both device-node and special-file preservation. See the `--devices`/`--specials` rows and the Phase-4 devices notes below | +| `--devices` | Preserve device files | ✅ Implemented | Recreates char/block device nodes on the destination via `mknod` instead of transferring content. Type + rdev are validated strictly (S_IFMT from the transmitted mode; major/minor range-checked, non-negative), and creation is **privilege-gated**: `mknod` needs `CAP_MKNOD`, so a non-root receiver (CI runs via setpriv as non-root) logs a warning and **skips the device entry safely** — the whole transfer never aborts just because the node could not be made. The node is created fd-relative below the receive root (`mknodat` on the confined secure parent), so it can never be placed outside the authorized root, never follows a symlink, and never replaces an existing directory. Only a char/block mode is honored. Crosses the wire (a new `STATUS_SPECIAL` frame carries the path + metadata mode + rdev; `PROTOCOL_VERSION` bumped **2.12.0 → 2.13.0**). Divergence: per-entry skip (not a hard error) when the receiver lacks `CAP_MKNOD`, documented in the Phase-4 devices notes | +| `--specials` | Preserve special files | ⛔ Impossible/Divergence | **FIFO recreation works**: FIFOs are recreated on the destination via `mkfifo` (unprivileged, so this is a real, assertable behavior under CI). **Only socket recreation is impossible**: a socket entry can be created only by `bind(2)` on a live socket, not by any filesystem call, so a source socket is skipped with an explicit note. That one unsupported node kind is why the flag is classified Impossible/Divergence even though FIFO recreation itself works; its normal path is otherwise complete. FIFO creation is privileged-gated only in the sense of graceful skip on any permission failure. Node creation is confined below the receive root (`mkfifoat` on the secure fd-relative parent; no `..`, no symlink follow). Crosses the wire like `--devices` (the `STATUS_SPECIAL` frame; `PROTOCOL_VERSION` bumped **2.12.0 → 2.13.0**). See the Phase-4 devices notes | +| `--copy-devices` | Copy device contents as file | ✅ Implemented | Copy a device's CONTENT into an ordinary regular file on the destination instead of recreating the node — non-privileged and safe. FastSync scans a device/FIFO as a regular file: its reported size (`st_size`, typically 0 for char devices and FIFOs) is copied, so a FIFO or a non-readable device becomes an empty (or size-bounded) regular file. The default data path is size-bounded and never blocks (it sends exactly `st_size` bytes, never an unbounded pseudo-device stream); with `--sendfile`, a non-regular source (FIFO/device) is detected from its `stat` mode and falls back to that same buffered read, so `--copy-devices --sendfile` cannot hang either. The run always succeeds and never crashes on such input. **Deliberate, safe divergence from rsync's dd-like unbounded device read.** See the Phase-4 devices notes | +| `--write-devices` | Write to devices as files | ✅ Implemented | Write the received data directly into an **existing** device node on the destination instead of creating a regular file. Restricted and best-effort: the destination must already exist and be a char/block device (opened only under the confined receive root, with `O_NOFOLLOW` + `O_NONBLOCK`); a missing, symlinked, FIFO-with-no-reader (`ENXIO`), non-device destination, or any write failure is **skipped with a warning** rather than allowed, so a run can never clobber the system, never blocks on a special-file target, and never aborts on an unusable target. See the Phase-4 devices notes | +| `-U`, `--atimes` | Preserve access times | ✅ Implemented | Captures the source access time (from the scanner's pre-read stat, so it is not clobbered by reading the file for transfer) and transmits it over the wire; the receiver restores it together with the mtime via `futimens`/`utimensat`. Implies metadata transmission (the times travel inside the `-M` metadata payload), but does not enable ownership application (that stays opt-in via the identity flags). Wire: new `atime` fields on the metadata frame + a `preserve_atimes` config boolean; `PROTOCOL_VERSION` bumped **2.11.0 → 2.12.0** | +| `-N`, `--crtimes` | Preserve create times | ⛔ Impossible/Divergence | Birth-times cannot be set by any portable filesystem call (`utimensat`/`futimens` only set atime/mtime), so this row is an explicit **Impossible/Divergence** (Phase 7 Wave B). Capture + transmit stays: `statx(STATX_BTIME)` on Linux records the source birth time as a wire field; the receiver logs a debug note that it cannot be applied and continues — never failing the transfer and never pretending it worked. On platforms without `statx` it parses as a documented no-op (flag accepted; nothing is captured). Implies metadata transmission. Wire: new `crtime` fields + a `preserve_crtimes` config boolean; `PROTOCOL_VERSION` bumped **2.11.0 → 2.12.0** (see the Phase-4 metadata-time notes) | +| `-O`, `--omit-dir-times` | Omit dirs from --times | ✅ Implemented | Real modifier now that FastSync preserves directory times. With metadata on, the scanner captures every traversed source directory's mtime (and atime under `-U`) and the sender transmits them in trailing `STATUS_DIR_TIMES` frame(s) **after all file data and the optional delete manifest** (chunked at the receiver's `MAX_MANIFEST_ENTRIES` per-frame cap); a dir-time entry only RECORDS metadata and never creates the directory, so empty source directories stay untransferred. The receiver defers applying them until its delete / `--delay-updates` publication phases have committed, so writing or removing a child never clobbers a parent directory's mtime (rsync applies directory times at the end for exactly this reason). When `-O` is set (the boolean crosses the wire) the receiver does not apply any of them; without `-O` an `-a`/`--preserve` transfer now restores directory times (reversing the old "never preserves dir times" divergence). Wire change: the terminal `STATUS_DIR_TIMES` frame; `PROTOCOL_VERSION` bumped **2.16.0 → 2.17.0** | +| `-J`, `--omit-link-times` | Omit symlinks from --times | ✅ Implemented | Real modifier now that FastSync preserves symlink times. Symlink entries already carried their metadata on `STATUS_SYMLINK`; the receiver now applies it with **no-follow primitives only** (`utimensat(..., AT_SYMLINK_NOFOLLOW)`, plus best-effort `fchmodat(..., AT_SYMLINK_NOFOLLOW)` and policy-gated `fchownat(..., AT_SYMLINK_NOFOLLOW)`), so the link itself is stamped without ever dereferencing it, confined fd-relative below the authorized receive root. A symlink has no children, so the times are applied immediately at creation. When `-J` is set (the boolean crosses the wire) the receiver skips the timestamps (mode/ownership are unaffected); without `-J` an `-a`/`-l` transfer restores symlink mtimes. Wire change alongside `-O`: the shared `STATUS_DIR_TIMES` frame; `PROTOCOL_VERSION` bumped **2.16.0 → 2.17.0** | +| `--super` | Receiver attempts super-user activities | ✅ Implemented | Phase 7 Wave E: receiver-side **safe-subset + clear-refusal** privilege model, tri-state `super_mode` (auto/on/off). `--super` **permits** the receiver to attempt super-user activities — ownership application and char/block device-node creation — that are already confined fd-relative below the authorized receive root; `--no-super` **forbids** them even when the receiver is root; the default (`auto`) preserves the pre-existing **best-effort** behavior of *attempting* them (not only when already root: an unprivileged attempt is refused by the kernel and skipped per entry, matching FastSync's history). The server additionally accepts an operator-level `--no-super` veto that forces `OFF` for every connection it accepts (so it also refuses any client `--copy-as`/`--super`); the `--fake-super` owner replay and the `--write-devices` write path are gated by the same policy. **FastSync never elevates**: no `setuid`/`seteuid`/`setgid` is ever called, and `--super` never bypasses the confinement floor (`file_open_secure_parent`, `O_NOFOLLOW`, root checks) — it only permits an attempt that is already confined. `--super` does **not** imply `--numeric-ids` and never enables client-chosen ownership on its own: ownership is applied only when an explicit identity policy (`--usermap`/`--groupmap`/`--chown`/`--numeric-ids`/`--copy-as`) is also given. A non-root receiver given `--super` logs exactly one warning at activation and each confined attempt is then refused by the kernel and skipped per entry (never aborts); `--no-super` suppresses ownership, char/block `mknod`, `--write-devices` and the fake-super owner replay, while unprivileged FIFO creation is unaffected. Wire: one trailing `super_mode` int on the config frame (validated 0..2), sent **before** the `--copy-as` block (fixed order: super int, then copy-as presence int + ids); `PROTOCOL_VERSION` bumped **2.17.0 → 2.18.0**. **Documented divergence from rsync:** rsync's `--super` runs the receiver with elevated privilege; FastSync only permits a confined attempt and never elevates | +| `--fake-super` | Store/recover privileged attrs via xattrs | ✅ Implemented | Phase 7 Wave B: full record **and replay**. The receiver writes the source `uid:gid:mode:mtime_sec:mtime_nsec` into a reserved `user.fastsync.stat` xattr on each written file (best-effort, fd-relative, format unchanged), then immediately re-applies it via `fake_super_restore_fd`: `fchown` (only where privileged — a non-root EPERM/EACCES is skipped silently, matching FastSync's identity philosophy), `fchmod`, and `futimens`. The OWNER leg is additionally skipped unless an explicit ownership identity policy (`--numeric-ids`/`--usermap`/`--groupmap`/`--chown`/`--copy-as`) is active — `--fake-super` on its own only *records* the source owner and must not act as an un-gated chown primitive — when `--no-super` forbids super-user activities (even for root), or when an active `--copy-as` is authoritative, so the recorded source owner can never override a forced `--copy-as` owner; the xattr record is still stored/replayed for a later privileged restore and mode/mtime still apply, so unprivileged `--fake-super` keeps working. The restored mode goes through the same sanitization as the normal metadata path (group/other write bits are never granted, so a recorded 0666 restores as 0644), so fake-super replay can never grant group/other-write that plain `--preserve` would refuse. Absence or a malformed record is a silent no-op, never fatal. The recording format diverges from rsync's `user.rsync.%stat%`; no cross-tool conversion is attempted. Implies metadata transmission so the source uid/gid/mode/mtime are available. Both it and `-X`/`-A` are incompatible with `-s` (chunk serialization), rejected up front | +| `--open-noatime` | Avoid changing access time when opening files | ✅ Implemented | Sender-side policy: the sender opens source files with `O_NOATIME` (Linux) when reading them for transfer, so the open/read does NOT bump the source's on-disk access time. Degrades safely when `O_NOATIME` is unavailable (not defined) or refused (`EPERM`, since it needs `CAP_FOWNER` or file ownership): the code falls back to a normal open, so the data always transfers — only the atime-bump is skipped. It does not itself capture/preserve atime; it only avoids modifying it. **Client-only, never crosses the wire.** Exposed as `file_open_for_read()` and applied to both the buffered data path and the sendfile path | +| `--numeric-ids` | Do not map uid/gid by name | ✅ Implemented | Ownership is applied through FastSync's opt-in identity path (see the Phase-4 identity notes below). `--numeric-ids` is a mapping-policy modifier: when applying ownership it uses the transmitted numeric uid/gid directly, skipping the name lookup. Without an ownership-affecting option it is inert (FastSync only applies ownership when the user opts in). It does not need `-M` to be parsed, but ownership is only applied when metadata (hence the source uid/gid) is actually transmitted (see the notes) | +| `--usermap=STRING` | Map usernames | ✅ Implemented | Opt-in ownership application. rsync subset implemented: comma-separated `FROM:TO` rules evaluated in order, first match wins; `FROM`/`TO` are group/user names (resolved on the SOURCE machine at parse time), `*` (FROM matches any id / TO = the receiving process's current euid), and an `@N` or bare `N` numeric id. Rules are carried over the wire as resolved numeric id pairs; the receiver applies a matching rule (else falls back to `--chown`, `--numeric-ids`, then a best-effort name lookup) via an fd-relative `fchown`. Malformed/unresolvable specs are rejected with a clear error, never a silent no-op. Implies metadata preservation so the source uid/gid travel. Only effective when the receiver can actually change ownership (root or membership); otherwise it warns and continues | +| `--groupmap=STRING` | Map group names | ✅ Implemented | Same rsync subset and semantics as `--usermap` but for the group (gid) side and the group databases. See the Phase-4 identity notes | +| `--chown=USER:GROUP` | Map owner and group | ✅ Implemented | Opt-in ownership override applied receiver-side. Forms: `USER:GROUP`, `USER` (owner only), `:GROUP` (group only); a `*` for USER/GROUP means the current/root user or group as appropriate; an `@N`/bare `N` numeric id is accepted. A `:` inside a name may be escaped as `\:`. Equivalent to a trailing `*:*` usermap+groupmap rule (so an explicit `--usermap`/`--groupmap` match wins). Malformed or unresolvable specs are clear parse errors. Implies metadata preservation. Only effective when the receiver has permission to chown; otherwise it warns and continues (rsync parity) | +| `--copy-as=USER[:GROUP]` | Perform the copy as another user/group | ✅ Implemented | Safe-subset implementation, an explicit divergence from rsync's **real identity switching**. rsync makes the receiving process actually assume USER/GROUP (setuid/setgid); FastSync's receiver is multithreaded, so a real credential drop would be unsafe and is never attempted — FastSync never calls `setuid`/`seteuid`/`setgid`. Instead the receiver FORCES the ownership of every entry it writes to `copy_as_uid`/`copy_as_gid` through the existing confined, fd-relative identity path (the same `fchown`/`fchownat` mechanism as `--chown`/`--usermap`/`--groupmap`; symlinks use `fchownat(..., AT_SYMLINK_NOFOLLOW)`, and directories — including intermediate parents created implicitly while writing a nested file — and char/block/FIFO nodes are owned no-follow too, so a directory never keeps the receiver's owner while its children get the target owner), with `--copy-as` at the **highest priority** — it beats usermap/groupmap/`--chown`/`--numeric-ids` and the best-effort name lookup. This REQUIRES a privileged (root) receiver: an unprivileged receiver REFUSES the whole transfer up front at the config handshake (`server_module_gate`, running inside `config_receive_with_validate` before the `STATUS_OK` ack) with a clear error and no file data exchanged — never a silent wrong-ownership result. A server running with an operator `--no-super` veto also refuses it, and a **daemon** refuses `--copy-as`, like every other client-chosen-ownership request (`--numeric-ids`/`--chown`/`--usermap`/`--groupmap`/`--fake-super`/explicit `--super`), unless the selected module opts in with `client owner = yes`; without that per-module opt-in a daemon must not honor an arbitrary client-selected owner (the standalone listener and SSH `--stdio` server keep honoring these for their single operator-authorized root). `--fake-super` interaction: `--copy-as` is authoritative, so the recorded source owner is never replayed over the forced target owner. If the ownership apply still fails with EPERM/EACCES (capability-restricted root, root-squash, read-only mount) the failure is logged at ERROR and the **entry is reported as failed** rather than written with the wrong owner, which fails the transfer (fail-fast) so overall success is never reported with the wrong owner. USER is resolved on the client against the user database (a name, an `@N`/bare `N` numeric id, or `*` meaning the client's current euid); when `:GROUP` is present it is resolved against the group database (`*` meaning the client's egid). **Group-default rule:** when the group is omitted FastSync uses the user's primary gid (`getpwuid(uid)->pw_gid`); a numeric id with no local passwd entry has no primary gid to look up, so `gid` falls back to `uid` (documented divergence). Malformed/empty/unresolvable specs are clear parse errors, never a silent no-op. Never elevates privileges and never bypasses the confined receive root. Implies metadata preservation (the source uid/gid must be transmitted). Wire: a new trailing config-frame block **sent after** the `--super` int (presence int, then the two int32 ids, both validated `>= 0` on receive; the ids are also rejected if they do not fit int32 at CLI parse time); `PROTOCOL_VERSION` bumped **2.17.0 → 2.18.0** | + +**Phase-4 metadata-time notes:** `-U/--atimes`, `-N/--crtimes`, +`-O/--omit-dir-times`, `-J/--omit-link-times`, and `--open-noatime` are new. +They change the wire: the per-file metadata frame grows `atime_valid` + +`atime_sec` + `atime_nsec` and `crtime_valid` + `crtime_sec` + `crtime_nsec` +(appended after the existing mode/uid/gid/mtime fields, preserving the exact +positions of every pre-existing field), and the config frame grows four +booleans — `preserve_atimes`, `preserve_crtimes`, `omit_dir_times`, +`omit_link_times` — that CROSS the wire so the receiver knows what to apply / +suppress. `--open-noatime` is **client-only** and is never serialized (it only +governs the sender's source reads). `PROTOCOL_VERSION` was bumped **2.11.0 → +2.12.0** (peers must match, exactly as prior phases did). + +**Client-vs-wire split:** `-U` and `-N` affect both the sender (capture) and the +receiver (apply), so they and their metadata fields cross the wire; +`-O`/`-J` are receiver-side preferences and cross as config booleans; +`--open-noatime` is purely a client/sender open flag and stays off the wire +(mirroring the existing convention where `ignore_errors` is client-only while +`force_delete` crosses the wire). + +**Phase-4 xattr/ACL notes (`-X/--xattrs`, `-A/--acls`, `--fake-super`):** these +are new in protocol 2.13.0 and add a bounded per-file xattr block to the +per-file metadata frame (count + each `name`/`value`, sent only when xattr +transport is enabled, i.e. with zero overhead on unaffected runs). The config +frame carries `preserve_xattrs`, `preserve_acls` (in the existing file-options +block) and a trailing `fake_super` boolean — all CROSS the wire so the receiver +knows the negotiated behavior; the derived `use_xattrs` flag is recomputed on +the receiver. `PROTOCOL_VERSION` was bumped **2.12.0 → 2.13.0** (peers must +match, exactly as prior phases did). + +- **Security model (both `-X` and `-A`):** only `user.*` and the + `system.posix_acl_access` / `system.posix_acl_default` namespaces are ever + captured (sender) or applied (receiver). `security.*` (SELinux, capabilities, + ...), `trusted.*`, and all other `system.*` attributes are never transmitted + or applied, so a client can never compel the receiver to set a privileged + xattr. The receiver re-validates each incoming name against this whitelist + even though the sender already filtered, so a malicious/compromised sender's + `security.capability` payload is rejected outright (a clean protocol error), + never applied. +- **Bounds / memory safety:** per-name length ≤ 255 B, per-value ≤ 1 MiB, + per-file count ≤ 256 names, per-file name+value total ≤ 4 MiB. Both the + sender (during capture) and the receiver (during receive) enforce these; an + oversized or malformed frame is rejected, never a large allocation. +- **Confined application:** xattrs are applied with `fsetxattr` on the exact + just-written destination file fd (before the atomic rename), never on a + caller-controlled path; this is the same confinement as mode/time restore. + The `--link-dest` / `-H` hard-link copy fallback (a byte copy when `link()` + is refused) also re-applies the incoming (or, for `-H`, the first member's) + xattrs and the `--fake-super` stat, so attributes are preserved rather than + silently dropped when the link fails. +- **Reserved fake-super key is receiver-only:** the `user.fastsync.stat` key is + excluded from sender capture AND from receiver application, so it can only be + written by the receiver's own `--fake-super` handling. A source file that + already carries such a record is never forwarded on a plain `-X` run, so it + cannot be spoofed to mislead a later privileged restore. +- **`-A` requires no libacl** — ACLs travel as the `system.posix_acl_*` xattrs. + Applying an ACL is owner-privileged: `fsetxattr` failure (e.g. non-root, + unsupported filesystem) is logged (collapsed to one line per file) and never + fatal. +- **`--fake-super`**: see the row above; the reserved key is `user.fastsync.stat` + with the documented `uid:gid:mode:mtime_sec:mtime_nsec` (mode octal) format. + It is honest but partial — there is no replay, and it does not interoperate + with rsync's `user.rsync.%stat%`. +- **Chunk serialization (`-s`) incompatibility:** the per-file xattr block rides + the streaming per-file frame, which `-s` replaces with a fixed buffer format, + so `-X` / `-A` combined with `-s` is rejected up front on both ends (mirroring + the existing `-H` + `-s` rejection) rather than silently dropping attributes. + +**atime capture does not clobber the source atime:** the sender records the +access time from the **same pre-read stat the scanner already took** (inside +`file_metadata_create`), before any file data is read for transfer. So `-U` +alone captures the correct atime even without `--open-noatime`. `--open-noatime` +is orthogonal: it keeps the source's on-disk atime from being bumped by the read +that actually ships the data (only honoured where `O_NOATIME` works; it degrades +to a normal open otherwise, so the data always transfers). + +**crtime handling:** `-N` captures the source birth time via `statx`/`STATX_BTIME` +(guarded `#ifdef STATX_BTIME` on Linux) and transmits it. On the receiver, **no +portable setter exists** (`utimensat` can only set atime/mtime), so the receiver +deliberately does **not** apply it: it logs a debug note and continues — it never +fails the transfer and never pretends the crtime was applied. This is the +explicit, documented unsupported-attribute handling. On platforms without +`statx` the flag is accepted but nothing is captured (a documented no-op). + +**omit-dir-times / omit-link-times:** `-O` and `-J` are **real modifiers** as of +P7 Wave D (`🔄 → ✅ Implemented`). FastSync now preserves directory mtimes +(captured by the scanner, transmitted in trailing `STATUS_DIR_TIMES` frame(s), +applied only after all children and the delete/publication phases) and symlink +mtime/owner/mode (no-follow `utimensat`/`fchownat`/`fchmodat` at link creation). +`-O` makes the receiver skip the directory-time set; `-J` makes it skip the +symlink timestamps (ownership/mode application is unaffected and stays governed +by the identity opt-in). Both config booleans already crossed the wire. See the +`-O`/`-J` rows and the Wave D note below. + +**-U/-N and -M interaction:** because FastSync carries all metadata (mode, uid, +gid, mtime, and now atime/crtime) in one bounded payload that is only sent when +metadata transmission is on, `-U` and `-N` imply metadata transmission (the +times travel inside that payload). They do **not** enable ownership application, +which remains opt-in strictly through the identity flags (`--numeric-ids` / +`--usermap` / `--groupmap` / `--chown`). + +**Phase-4 identity notes:** `--numeric-ids`, `--usermap`, `--groupmap`, and +`--chown` are real. They introduce a **controlled, opt-in, privilege-gated** +ownership-application path on the receiver: plain `-M`/`--preserve` still does +NOT apply client-supplied ownership (FastSync's deliberate conservative +default, byte-for-byte backward compatible); ownership is only attempted once a +client explicitly requests an ownership-affecting option. Application goes +through an fd-relative `fchown()` in the receiver's metadata-restore path (after +the file is fully written, before timestamps are set), so it is confined and +symlink-safe — never a path-based `chown`. When the receiver lacks permission +(typically non-root, e.g. the CI `nobody` user) `EPERM`/`EACCES` is logged as a +warning and the transfer CONTINUES with exit status success, matching rsync. +A no-op default means existing transfers are unaffected. + +Resolution of the destination uid/gid on the receiver: a matching +`--usermap`/`--groupmap` rule wins; else the matching `--chown` side; else, with +`--numeric-ids`, the transmitted numeric id is used raw (no name lookup); else a +best-effort name lookup on the receiver's own account databases (skipped when +the transmitted id has no name present there). `--chown` enforces the receiver +side and is validated at parse time (malformed specs are clear errors, never a +silent no-op). + +Wire/version: the config frame gained `numeric_ids`, `chown_uid_set`, +`chown_uid`, `chown_gid_set`, `chown_gid`, and the `usermap`/`groupmap` tables +(count-delimited lists of resolved int32 FROM/TO id pairs), so +`PROTOCOL_VERSION` was bumped **2.10.0 → 2.11.0** (peers must match). All new +fields cross `config_send`/`config_receive` with full symmetry and are validated +on receive (bounded map sizes below `MAX_IDENTITY_MAP`, ids `>=` the `-1` +sentinels). + +Documented divergences from rsync: because FastSync transmits only numeric +uid/gid (not names) on the wire, name-based values (`--usermap`/`--groupmap` +names, `--chown` names) are resolved to numbers at CLI parse time against the +**client (sender) machine's** account databases; this reproduces rsync's +semantics on a shared-account source/destination and is documented for a +genuinely different destination. The interesting named-value subset is +supported (`*` FROM wildcard, `*` TO = current user, `@N`/bare-`N` numerics); a +lone-`@` "use the FROM value unchanged" rsync form is not implemented. Also +unlike rsync, plain `-M` never applies ownership and `--usermap`/`--groupmap`/ +`--chown` each imply metadata preservation so the source uid/gid actually travel +(the flags only take effect where ownership is being preserved/applied). + +**Phase-4 hard-links notes:** `-H`/`--hard-links` is real and introduces a +deduplicating wire path for files whose source entries share a filesystem inode. +On the sender, the scanner records each distinct `(st_dev, st_ino)` encounter and +assigns it a stable, run-local link-group id (`HardLinkTable`, mutex-guarded so a +multi-threaded scan could share one instance). The FIRST member of a group is +transferred normally and carries the data; each later (sibling) member is +transmitted as a payload-less `STATUS_HARDLINK` frame carrying its destination +path, the group id, and the first member's destination-relative wire path. +Ordering is guaranteed by forcing the sequential scanner whenever `-H` is on +(even under `-j`/`--threads`), so the first member is always emitted — and, on the receiver's +single write thread, installed — before any of its siblings; the receiver is +therefore always able to link to an already-present first member, including the +"first member already up-to-date/skipped" case (the sibling links to or copies +the existing file). Asymmetric existence policies are handled gracefully: under +`--existing`, if the first member's destination is absent (so it is skipped) but +a sibling's own destination already exists, that existing sibling is left in +place rather than the transfer aborting on the missing first member. The receiver +installs each sibling beneath its confined root +as an atomic hard link (temp link + rename); when `link()` fails (cross-device, +filesystem refuses links) it falls back to a byte-identical local copy of the +first member, never a partial/corrupt file. `--delay-updates` stages each sibling +as a hard link to the first member's STAGED file, so publication's renames +preserve the shared inode; `--inplace` and `--partial` are unaffected (a sibling +is a fresh link/copy). Because a hard link shares an inode, metadata is applied +exactly once on the first member and never re-written through the sibling (whose +members are byte-identical by construction), so all members agree. + +Wire/version: `PROTOCOL_VERSION` was bumped **2.11.0 → 2.12.0** (peers must +match). The config frame already carried the `preserve_hard_links` boolean +(round-trips through `config_send`/`config_receive`); the only new wire element +is the `STATUS_HARDLINK` frame described above. Incompatibilities (rejected up +front with a distinct error on the client, and re-checked on receive): `-H` with +`-s` chunk serialization (the chunk wire has no per-file hard-link info) and `-H` +with `--append`/`--append-verify` (a payload-less sibling cannot be tail-resumed). + +**Phase-4 devices notes:** `--devices`, `--specials`, `-D`, `--copy-devices`, +and `--write-devices` are new. They change the wire: the config frame grows three +booleans — `preserve_specials`, `copy_devices`, `write_devices` — that CROSS the +wire (`preserve_devices` already existed), and a new `STATUS_SPECIAL` frame (used +by `--devices`/`--specials`/`-D`) carries a special/device entry: the destination +path, the metadata frame (whose mode's S_IFMT bits carry the node kind, requiring +the flags to imply metadata transmission), and two int32 `rdev` major/minor +fields. The chunk-serialized wire (`-s`) grows a matching per-file special +marker + rdev so `--devices/--specials` also work under `-s`. `PROTOCOL_VERSION` +was bumped **2.12.0 → 2.13.0** (peers must match, exactly as prior phases did). + +**Privilege gating (the crux):** making a device node requires `CAP_MKNOD` (root). +CI runs the integration suite as a NON-ROOT user (via setpriv), so `mknod` fails +with `EPERM`. The receiver treats this as a graceful, logged *skip of the entry* +returned as a success/skip outcome — the whole transfer NEVER aborts just because +the environment cannot create the node. `mkfifo` (FIFOs) is unprivileged, so +`--specials` FIFO creation is a real, assertable behavior under CI; sockets cannot +be recreated by any standard filesystem call and are skipped with an explicit +note. The "device actually created" integration assertions are guarded to run +only as root. User-facing expectation: point `--devices` at devices and a +non-root receiver will faithfully skip them while transferring everything else. + +**Confinement & validation:** a special/device node is created with +`mknodat`/`mkfifoat` on the parent directory opened fd-relative below the receive +root (`file_open_secure_parent`: `O_NOFOLLOW`, no `..` components, root-checked), +so a node can never be created outside the authorized destination root and never +through a symlinked parent. The transmitted type is derived ONLY from the +validated S_IFMT bits of the metadata mode (char/block/FIFO honored, socket +skipped, regular/dir rejected as an invalid special), and the transmitted rdev is +validated both on the wire (`file_receive_special`, `chunk_deserialize`) and at +the creation site (`file_special_rdev_valid`): a negative, oversize, or +non-device-carrying rdev is rejected outright (receiver aborts the frame), and a +node is never replaced over an existing directory or unrelated entry (a matching +existing node is left in place). `--write-devices` is the deliberately restricted +danger path: it only ever opens an existing char/block node under the confined +root, and every failure mode (missing, non-device, write error, EPERM) is a +warning + skip, never a system-clobbering write or an abort. + +**Documented divergences (honest subset):** +- A device entry the receiver cannot create (missing `CAP_MKNOD`) is *skipped*, + not a transfer failure — rsync under the same conditions would error. +- `--copy-devices` copies the device's *reported size* (typically 0 for char + devices/FIFOs) into a regular file and never reads an unbounded pseudo-device; + this is the safe, non-hanging alternative to rsync's dd-like read. +- `--write-devices` requires the device to already exist at the destination and + never creates it; unsupported/inaccessible targets are skipped, not written. +- Ownership is not applied to recreated nodes (identity `fchown` needs an fd and + would require opening the node); permissions and mtime are applied at + creation / via `utimensat`. + +## 9. Symlink Handling + +| Flag | Rsync Description | FastSync Status | Notes | +|------|-------------------|-----------------|-------| +| `-l`, `--links` | Copy symlinks as symlinks | ✅ Implemented | A symlink is transmitted as a real symlink: its target string crosses the wire (a new `STATUS_SYMLINK` frame / chunk entry type) and the receiver creates it with `symlinkat` beneath the receive root. This makes the previously-`-l`-included-but-targetless symlink handling complete. See the Phase-4 symlink-trust notes | +| `-L`, `--copy-links` | Transform symlink to referent | ✅ Implemented | `copy_links` config field | +| `--copy-unsafe-links` | Transform unsafe symlinks | ✅ Implemented | `copy_unsafe_links` config field | +| `--safe-links` | Ignore symlinks outside tree | ✅ Implemented | `safe_links` config field | +| `--munge-links` | Munge symlinks for safety | ✅ Implemented | Sender rewrites each transmitted symlink target with a `#SYMLINK/` marker; a target that could escape the receive root (absolute or containing `..`) is never transmitted (contained/skipped); the receiver strips the marker to restore the real target. See the Phase-4 symlink-trust notes | +| `-k`, `--copy-dirlinks` | Transform symlink to dir | ✅ Implemented | A symlink whose referent is a directory is dereferenced and recursed as a real directory; a symlink to a regular file stays a symlink. Sender-side only. See the Phase-4 symlink-trust notes | +| `-K`, `--keep-dirlinks` | Treat symlinked dir as dir | ✅ Implemented | On the receiver, an existing destination symlink-to-a-directory is used as that directory (followed) instead of being replaced; it is followed only when it resolves to a directory that stays beneath the receive root. See the Phase-4 symlink-trust notes | + +**Phase-4 symlink-trust notes:** `-l/--links`, `-k/--copy-dirlinks`, +`-K/--keep-dirlinks`, and `--munge-links` form the "symlink trust boundaries" +row. Making all three new flags have an observable, security-sane effect +required transmitting symlink targets, so FastSync's `-l/--links` is now real: +a symlink-type entry carries its target on the wire (a new `STATUS_SYMLINK` +frame for the per-file path, and a new entry type `2` in the `-s` chunk +serializer) and the receiver creates it with `symlinkat` under an `O_NOFOLLOW` +parent walk, never following the target. Wire changes: `STATUS_SYMLINK`, +the chunk entry type `2`, a per-entry symlink-target string, and two new config +booleans that CROSS the wire — `munge_links` and `keep_dirlinks`; `PROTOCOL_VERSION` +was bumped **2.12.0 → 2.13.0** (peers must match, exactly as prior phases did). + +**Per-flag semantics and divergences.** +- **`-l/--links`** copies a symlink as a symlink: the scanner `readlink`s the + target, the sender transmits it, and the receiver `symlinkat`s it. FastSync + `-l` never preserved symlink targets before (the flag was documented partial + and, in fact, tried to read the referent as file data); it now does, matching + rsync. Divergences: because the receiver enforces the symlink containment + predicate unconditionally, a plain `-l` sync **refuses to round-trip a + legitimate absolute symlink target** (it is dropped, never created pointing + outside the root — see the `--munge-links` note for the symmetric trust + boundary); a relative in-root target is copied as-is. As of P7 Wave D FastSync + also applies the symlink's own metadata with no-follow primitives + (`utimensat`/`fchownat`/`fchmodat` with `AT_SYMLINK_NOFOLLOW`), so `-J` is a + real omit switch rather than a no-op. +- **`-k/--copy-dirlinks`** (sender): a symlink whose referent is a directory is + dereferenced and recursed into as a real directory; a symlink to a regular + file (or any non-directory) is kept as a symlink. This is rsync's `-k`. When + `-L/--copy-links` or `--safe-links`/`--copy-unsafe-links` are active, their + (dereference) semantics take precedence, so `-k` is subsumed exactly as in + rsync. +- **`-K/--keep-dirlinks`** (receiver, crosses the wire): when a directory is to + be created (on-demand parent creation for a child write) and the destination + path is already an existing symlink that resolves to a directory *within* the + receive root, that symlinked directory is used (followed) instead of being + replaced by a real directory; new entries are written beneath it. The follow + is confined: it only happens where `realpath` of the symlink resolves to a + still-within-root real directory, so a malicious link pointing outside the + root is never followed. Scope: `-K` acts on the write path (parent/`mkdir` + creation); the delete walker still never follows symlinks (a documented + divergence for `--delete` over an existing symlinked dir). Without `-K` the + destination symlink is not followed (the O_NOFOLLOW walk fails the write), + which is the safe default. +- **`--munge-links`** (sender security rewrite; crosses the wire so the receiver + unmunges): every transmitted symlink target is prefixed with the marker + `#SYMLINK/`; the receiver strips the marker (only when the negotiated + `munge_links` policy is on — a plain `-l` run never strips the prefix, so a + source symlink that genuinely begins with `#SYMLINK/` round-trips verbatim) + and restores the exact real target. The trust boundary is **symmetric and + enforced receiver-side**, independent of the sender: `file_symlink_at_secure` + refuses any target that `file_symlink_target_contained` rejects (absolute + `/...` or relative with a `..` component), and `file_save_to_disk_full` + contains such an entry (skipped) rather than materializing it. A deliberate confinement trade-off: because the receiver + enforces containment unconditionally, a plain `-l` (no `--munge-links`) sync + *refuses to round-trip a legitimate absolute symlink target* — such target is + dropped, never created pointing outside the root. This is a stricter subset of + rsync: rsync stores munged targets on the RECEIVING side and depends on both + ends running `--munge-links`; FastSync additionally enforces the containment + predicate at the receiver regardless of what the sender transmitted. When no + symlink is being transmitted (`-l`/`-k`/`-a` off) `--munge-links` has nothing + to rewrite and is inert. -*K/`--keep-dirlinks` policy is installed per + connection at config-accept (stable for the whole transfer, never racy under + `-j`/`--threads`), and only ever follows an in-root symlink-to-directory.* + +**Compatibility (byte-identical when all three are absent):** `-k`, `-K` and +`--munge-links` are opt-in. Without them the scanner's link handling, the wire +frames, and the receiver's writes are unchanged for every other option set, so a +run that previously worked continues to behave identically. `-l/--links` itself +now transmits targets (the prior behavior was broken/partial); its status moved +`⚠️ Partial → ✅ Implemented`. + +## 10. Sparse & Device + +| Flag | Rsync Description | FastSync Status | Notes | +|------|-------------------|-----------------|-------| +| `-S`, `--sparse` | Sparse block handling | ✅ Implemented | Phase 7 Wave B: real hole preservation with no wire change. The receiver's sparse-aware writer (`write_all_sparse`, next to `write_all` in `src/shared/file.c` and `src/shared/file_store.c`) walks the in-memory file image and emits any all-zero run ≥ 4096 bytes as a hole via `lseek(SEEK_CUR)` (the pre-size `ftruncate` guarantees the offset bookkeeping and logical size), `ftruncate(size)` after the last run pins the final size even with a hole tail. Wired into both the atomic temp+rename store and `--inplace` when `sparse` is set; the non-sparse path is byte-identical to before. **Sparse wins over `--preallocate`** (posix_fallocate is skipped when sparse is set, so the holes are not re-allocated). Interplay note: under `--partial` a retained sparse temp already has the full logical size (trailing content is holes), so `--append`'s "shorter destination" resume does not re-run; the retained file is still valid and a normal re-transfer (or `-W`/delta) repairs it — documented so the combination is never surprising | +| `--preallocate` | Allocate dest files before writing | ✅ Implemented | The receiver preallocates the destination file's full expected space before any data is written, so a transfer that would overflow disk fails fast at allocation time (a clean error, not a half-written file) and the file is laid out contiguously, avoiding fragmentation. Crosses the wire (the config frame carries a `preallocate` boolean; `PROTOCOL_VERSION` bumped **2.10.0 → 2.11.0**, peers must match) so the sender knows the receiver will preallocate and the receiver performs it. **Allocation approach:** `posix_fallocate()` is preferred because it reserves *real* disk blocks (true fail-fast on ENOSPC), falling back to plain `ftruncate()` only when the filesystem reports the allocation is unsupported (`EOPNOTSUPP`/`ENOSYS`); `ftruncate` still extends the logical size so the intent degrades gracefully. **Fallback/error semantics:** `EOPNOTSUPP`/`ENOSYS` → clean fallback to `ftruncate` (best-effort, preallocates the logical size and never fails a transfer on filesystems that lack `posix_fallocate`); a genuine allocation failure (`ENOSPC`/`EDQUOT`/`EFBIG`/…) aborts the file/receive with a distinct `preallocate failed ... transfer aborted` error — it does **not** fall back to a normal non-preallocated write, preserving the fail-fast purpose. **Size-known requirement:** preallocation only runs when the final size is already known up front (the normal regular-file case); unknown-length data is skipped (never failed). **Orthogonality:** applies uniformly across the atomic temp+rename store path, `--inplace`, `--partial`/`--partial-dir`, `--delay-updates` (the staged temp file is preallocated before data flows) and the `--link-dest` copy fallback; it neither implies nor conflicts with `-s`, `--append`, or delta. rsync-divergence: rsync signals that `--preallocate` is ignored with `--sparse`; FastSync gives **sparse precedence** — when both are set, `posix_fallocate` is skipped so the holes the sparse writer creates are not re-allocated (the `ftruncate` presize sizing stays), matching the intent of "sparse wins". See the Phase-4 preallocate notes below | + +**Preallocate notes (Phase 4, preallocate wave):** `--preallocate` is implemented as a real receiver-side allocation of the destination file's space before data is written. It is a plain boolean config flag that crosses the wire (serialized in the config frame's selection-options block, mirroring `--inplace`/`--append`/`--force`), so the run requires matching ends: `PROTOCOL_VERSION` was bumped **2.10.0 → 2.11.0** (peers must match or the version check fails). The allocation is performed on the exact destination fd, immediately after it is opened, before any bytes are streamed; `posix_fallocate` (and the `ftruncate` fallback) leave the fd's file offset untouched, so the subsequent data write at offset 0 is unaffected and complete. Because FastSync writes each file's byte payload in one in-memory batch, the "full expected size" is exactly the known `data_size`, which is what gets preallocated. Unknown-length/streamed payloads are skipped rather than failed. A failed allocation logs a distinct `preallocate failed` error and aborts the file (the atomic temp is unlinked, the inplace target is left untrimmed) so the run fails cleanly and never silently degrades to a non-preallocated write — preserving rsync's fail-fast intent on a full disk. + + +## 11. Checksum & Comparison + +| Flag | Rsync Description | FastSync Status | Notes | +|------|-------------------|-----------------|-------| +| `--checksum` | Skip based on checksum | ✅ Implemented | With `--incremental`, compares per-file whole-file content digests to skip unchanged files. The digest algorithm is `xxh64` with seed 0 by default and is selectable via `--checksum-choice`/`--cc` (xxh64/xxhash or md5) and `--checksum-seed=NUM` (see those rows); `-c` remains compression | +| `--checksum-choice=STR`, `--cc=STR` | Choose checksum algorithm | ✅ Implemented | Real algorithm selection for the per-file whole-file digest used by the `--incremental`/`--checksum` handshake and by the basis-dir content verification. FastSync genuinely supports `xxh64` (the default, exact xxHash64, seeded by `--checksum-seed`) and `md5` (via OpenSSL EVP); `xxhash` is accepted as rsync's spelling of xxHash64. Any other name (md4/sha1/sha256/crc32/none/…) is rejected with a clear error at parse time — never a silent no-op. `--cc` is the alias (`--cc=ALG` and space forms both parse). The algorithm id and seed cross the wire with the config frame, so the receiver hashes its on-disk old file with the SAME algorithm+seed the sender used and both agree on a match; the sender's digest and the receiver's comparison live in the per-file `STATUS_CHECK` handshake, which now carries a length-prefixed, bounded (1..16 byte) digest instead of a fixed 64-bit value, and the receiver pins the received length to the negotiated algorithm's digest length (defense-in-depth: a mismatched/malicious length only forces a safe re-transfer). Note: `md5` is a FIPS-non-approved algorithm, so under an OpenSSL build with FIPS mode enabled `--checksum-choice=md5` fails loudly rather than silently falling back. Protocol/layout: `PROTOCOL_VERSION` bumped **2.9.0 → 2.10.0** (peers must match). Defaults preserve the pre-existing behavior byte-for-byte (xxh64, seed 0). Like rsync, the choice only takes effect where a whole-file digest is actually computed (`--checksum` on, or a basis-dir flag); it does not itself enable `--checksum`. Closely-related divergence: the delta BLOCK strong checksum (§11 delta) stays xxHash32 — `--checksum-choice` selects only the whole-file digest, matching rsync where the per-block checksum is independent of the whole-file checksum choice | +| `--compare-dest=DIR` | Compare dest files relative to DIR | ✅ Implemented | DIR is a receiver-side basis relative to the destination root (confined below it; absolute/`..`/`.` rejected, `//` collapsed and trailing `/` dropped). On the receiver's per-file check (implies `--incremental`) an exact match = same size + mtime (unless `--size-only`; `-I` disables matching) **and** equal xxHash64 of the sender's file; a match suppresses the data transfer. compare-dest never copies: it only skips a file the destination does **not** already hold (sparse destination, rsync parity), and is consulted before the normal delta/full paths. Repeatable; searched in command-line order, first match wins. Divergences: when the destination already holds a *different* version rsync deletes it but FastSync instead transfers the data (keeps the mirror complete; never deletes without `--delete`); attribute-only differences on a match are not re-applied (data is skipped so the sender never sends metadata); content is verified by xxHash64, stricter than rsync's default quick check. Sizing: FastSync's whole-file payload limit is 256 MiB on **every** transfer path (not basis-specific); rsync applies basis dirs to arbitrary sizes, so FastSync refuses a basis run whose source contains a larger file up front with a clear error before any transfer. Wire: a basis-count field is always present on the config frame (protocol 2.9.0, so clients and servers must both be 2.9.0) | +| `--copy-dest=DIR` | Include copies of unchanged files | ✅ Implemented | Same basis rules as `--compare-dest`, but an exact match materializes a **local copy** of the DIR file into the destination (via the normal atomic temp+rename store path, so `--existing`/`--ignore-existing`/`--update`/`--backup`/`--delay-updates` all still apply) instead of transferring data. Repeatable; command-line order = priority. Content is xxHash64-verified before the copy. Divergences: a basis-hit destination keeps the basis file's own mode/uid/gid and mtime (the sender sends no metadata on a skip), so with `--size-only` its mtime can differ from the source and attribute-only differences are copied with the basis attributes rather than rsync's "copy + fix attributes". Requires `--incremental` (implied); incompatible with `-s`. Wire: protocol 2.9.0 | +| `--link-dest=DIR` | Hardlink to files when unchanged | ✅ Implemented | Same basis rules as `--copy-dest`, but an exact match installs an atomic **hard link** to the DIR file (temp hard link + rename) so no data or disk space is used; where the link is impossible (basis on another filesystem, filesystem refuses links) it falls back cleanly to a byte-identical local copy, never a corrupt/partial file. `--delay-updates` stages the link and publishes by rename, so the final entry stays a real hard link. Repeatable (searched in command-line order, first match wins). Content is xxHash64-verified before linking. Divergences and caveats: an already up-to-date destination file is not re-linked to a basis file (only files that would otherwise be written are linked); a link keeps the basis inode's own mode/uid/gid and mtime — metadata is never written through the shared inode (that would mutate the basis file), so a later `--inplace` run that rewrites such a destination path **will mutate the basis snapshot** through the shared inode (use `--copy-dest` when the destination must stay independently writable); with `--size-only` the linked mtime can differ from the source; a `--remove-source-files` source satisfied by a basis dir is treated as skipped and therefore **retained** (never removed); basis dirs are excluded from `--delete`. Requires `--incremental` (implied); incompatible with `-s`. Wire: protocol 2.9.0 | +| `-y`, `--fuzzy`, `--no-fuzzy` | Find similar file for basis | ✅ Implemented | `-y/--fuzzy` is a pure bandwidth optimization on the existing receiver-driven delta path: when a file must be transferred and the destination holds no usable content at the exact path (file absent, or the destination file is outside the delta engine's size bounds), the receiver searches the SAME destination directory for an existing regular file whose basename is similar to the incoming name and uses it as the delta basis, so the sender transmits only the differences instead of the whole file. The output is always byte-exact regardless of which (or whether any) basis is chosen. Decision location: the receiver performs the candidate search inside `receive_incremental_check` and sends the normal `STATUS_DELTA_SIGNATURE`; the sender never learns the basis was a different file, so no new frame type or sender logic was needed — only the config frame grew a `fuzzy` boolean, so `PROTOCOL_VERSION` was bumped **2.8.0 → 2.9.0** (peers must match). Similarity heuristic (deterministic, simpler than rsync's deliberately-fuzzy matching, and documented precisely): candidates are the target's sibling entries in its destination directory, opened `O_NOFOLLOW`/`AT_SYMLINK_NOFOLLOW` under the confined root (symlinks never followed; nothing outside the destination root is ever read or hashed); dotfiles, directories, the target's own name, and the `.fastsync-stage`/temp scratch names are excluded; like the ordinary delta path, the block signature the receiver transmits is derived from on-disk content it may not otherwise send, so a negotiated `--fuzzy` run exposes the destination's sibling files (at block granularity) to the sender as a known-plaintext oracle — the same information class as the normal delta handshake over the file being replaced; the size gate is the delta engine's own bounds (both files ≥ 16 KiB, ≤ `--delta-max`, ratio ≤ 10×) rather than rsync's ~1.5× size window; the name gate is a Levenshtein edit distance between the basenames accepted only when ≤ half the length of the longer basename; the single best candidate (smallest distance, tie-break size closest to the incoming file then lexicographically smaller basename) is read; the directory scan is capped at 4096 entries so a pathological directory cannot stall a transfer. When fuzzy applies: only to files the receiver would otherwise send whole — the destination's own file is always preferred as the delta basis when it exists and fits the delta size bounds, so fuzzy does NOT replace an existing-but-different destination basis; FastSync's 10× delta size-ratio bound means an existing destination file that is too far away in size still lets the fuzzy search run. When no similar candidate exists the transfer falls back to the normal whole-file transfer. rsync-divergence note: rsync's own matching uses a fuzzy name/size rule set; FastSync implements the closest safe deterministic approximation above. Because FastSync's delta machinery is off by default (rsync's is on), `--fuzzy` implies `--incremental` + `--delta` (unless `--whole-file`/`-W` or an explicit `--no-delta` switched delta off, in which case fuzzy is inert — matching rsync where `--whole-file` makes fuzzy irrelevant). Unlike the basis-dir options, `--fuzzy` honors an explicit `--no-incremental` (it does not force the handshake back on); an explicit `--no-incremental` also suppresses the delta implication so no invalid `--delta requires --incremental` config results. `--no-fuzzy` negates it. All surrounding semantics are untouched: a fuzzy-reconstructed file is stored as a normal file, so `--remove-source-files`, itemize/`-i`, `--stats`, `--backup`, `--delay-updates`, `--existing`/`--ignore-existing`/`--update` behave exactly as for a whole-file transfer (the fuzzy delta does not skip the file) | + +## 12. Compression + +| Flag | Rsync Description | FastSync Status | Notes | +|------|-------------------|-----------------|-------| +| `-z`, `--compress` | Compress file data | ✅ Implemented | Always uses zstd (rsync supports multiple algorithms — a documented divergence, selectable via `--compress-choice`). Phase 7 Wave A: `-z` is now the compression short form; `-c` is rsync's `--checksum` | +| `--compress-choice=STR`, `--zc=STR` | Choose compression algorithm | ✅ Implemented | FastSync supports `zstd` and `none` | +| `--compress-level=NUM`, `--zl=NUM` | Set compression level | ✅ Implemented | 1-22, default 5 | +| `--compress-threads=NUM` | Set compression threads | ✅ Implemented | `compression_threads` config field (client-only; does not cross the wire). Sets the number of worker threads used by the zstd compression pool to NUM (1..64; 0/garbage/oversized rejected up front). Accepted in both `--compress-threads=NUM` and two-argument `--compress-threads NUM` forms. Composes with `-z`/compression; under the `-j`/`--threads` multithreaded pipeline it parallelizes compressed chunk encoding. See test_tcp.py `-z --compress-threads=2` and test_client_cli.c | +| `--skip-compress=LIST` | Skip compress for suffixes | ✅ Implemented | Comma-separated, case-insensitive suffix list; empty list skips none; incompatible with FastSync chunk serialization (`-s`) | + +## 13. Connectivity + +| Flag | Rsync Description | FastSync Status | Notes | +|------|-------------------|-----------------|-------| +| `-e`, `--rsh=COMMAND` | Remote shell to use | ✅ Implemented | `-e`/`--rsh` (and `--rsh=COMMAND`) select the remote-shell program used to build the SSH child argv, overriding the default `ssh`. The command is whitespace-split into the leading argv words so rsync's `-e "ssh -p 2222"` works; the standard `-o` family, an optional `-p` port, `user@host` and the quoted remote command (`fastsync-server --stdio`) follow. Stored in the `rsh_command` config field. **Client-only, never crosses the wire** (it is a launch concern, not a handshake property) | +| `--rsync-path=PROGRAM` | rsync binary on remote | ✅ Implemented | Alias for `--fastsync-server-path`: both write the `fastsync_server_path` config field used as the remote-side server program (always quoted as one remote-shell word), which CROSSES the wire as before. Kept separate from `--rsh`, which names the local connecting program | +| `--port=PORT` | Alternate daemon port | ✅ Implemented | rsync's daemon-port flag maps to the client-side `server_port` config field: a client connects to a TCP/TLS server (incl. `host::module/path` daemon destinations) with `--server-port`, and the `fastsync-server --daemon` listener's port is taken from its config's `port` key (default 873) or overridden by `--dparam port=` / `-p` | +| `--sockopts=OPTIONS` | Custom TCP options | ✅ Implemented | Comma-separated allowlist of `OPT=VAL` applied via `setsockopt` after `socket()` before `connect()`/`bind()`. Only `TCP_NODELAY`, `SO_KEEPALIVE`, `SO_REUSEADDR` (0/1) and `SO_RCVBUF`/`SO_SNDBUF` (byte count) are accepted; an unknown option name or a bad value is rejected up front, never silently ignored. A value is required for every option (`OPT=VAL`; a bare name is an error). Applied to the outgoing TCP and TLS client socket; absent by default. `SockOptEntry`/`sockopts` config fields. Local socket concern: never crosses the wire | +| `--blocking-io` | Use blocking I/O for remote shell | ✅ Implemented | With `--blocking-io` the SSH-transport socketpair socket is left without `SO_RCVTIMEO`/`SO_SNDTIMEO`, so the transfer blocks naturally; by default it gets the same read/write timeout as the TCP transport (see `--timeout`). `blocking_io` config bool. **Client-only, never crosses the wire** | +| `--outbuf=N\|L\|B` | Set output buffering | ✅ Implemented | `N` (none/unbuffered) → `_IONBF`, `L` (line) → `_IOLBF`, `B` (block, the default) → `_IOFBF` via `setvbuf` on stdout and stderr. Garbage values are rejected. `outbuf` config field (`OutbufMode`). **Client-only, never crosses the wire** | +| `--address=ADDRESS` | Bind address for outgoing socket | ✅ Implemented | Binds the outgoing client socket to a local source address before `connect()` (resolved with the same `-4`/`-6` family hints as the destination). Local socket concern: never crosses the wire | +| `-4`, `--ipv4` | Prefer IPv4 | ✅ Implemented | Forces `AF_INET` in the `getaddrinfo` hints for client destination/source resolution and the server bind (see the Phase 5, Wave B note). Mutually exclusive with `-6` | +| `-6`, `--ipv6` | Prefer IPv6 | ✅ Implemented | Forces `AF_INET6` in the `getaddrinfo` hints for client destination/source resolution and the server bind. Mutually exclusive with `-4` | +| `--remote-option=OPT`, `-M` | Send an option only to the remote side | ✅ Implemented | Each value is appended to the remote server invocation over SSH as an individually single-quote-escaped shell word in `ssh_build_remote_command()`. Values are validated (non-empty, no control characters) and shell metacharacters cannot break out of the quoting (`;`, `&`, `|`, `, `$`, `(`, `)`, quotes are neutralized), so a value cannot inject an arbitrary remote command and a subsequent `--` on the client line cannot be turned into one. The options never cross the binary config frame. Phase 7 Wave A: the short `-M` form is now available (as `-M OPT` and `-M=OPT`), matching rsync; metadata mode moved to long-only `--preserve` | + +## 14. Daemon Mode + +| Flag | Rsync Description | FastSync Status | Notes | +|------|-------------------|-----------------|-------| +| `--daemon` | Run as rsync daemon | ✅ Implemented | Wave A: a real persistent listener. `fastsync-server --daemon --config FILE` (plus `--no-detach` to stay foreground; without it the listener detaches to the background after binding) reads a FastSync-native module config file and serves each connection confined to the requested module's `path` root (never a client-chosen root; every client-chosen-ownership/super-user request (`--numeric-ids`/`--chown`/`--usermap`/`--groupmap`/`--fake-super`/`--copy-as`/explicit `--super`) is refused unless the module opts in with `client owner = yes`, and the operator `--no-super` veto is honored). TCP/TLS via the existing `--tls` stack; plaintext still requires `--allow-unauthenticated` (same secure default as the standalone server). Client destinations use rsync's `host::module/path` form. Wire/protocol: the config frame gained a trailing daemon-module string and `PROTOCOL_VERSION` was bumped **2.14.0 → 2.15.0** (see the Daemon Mode notes below). Daemon mode is built in FastSync's own protocol/config grammar, not rsync's SMB/daemon option encoding | +| `--config=FILE` | Alternate rsyncd.conf file | ✅ Implemented | Wave A: selects the daemon config file. Default when omitted (in `--daemon` mode): `~/.config/fastsync/fastsyncd.conf` if it exists, else `/etc/fastsyncd.conf`. The grammar is FastSync-native (documented in the Daemon Mode notes below) and strictly rejects unknown keys so a typo can never silently change what a module serves; requires `--daemon` | +| `--dparam=OVERRIDE` | Override global daemon config | ✅ Implemented | Wave A: overrides one global scalar from the command line (`--dparam port=8734` and `--dparam=KEY=VALUE` both work). Limited to the global scalar keys the grammar defines (`port`, `motd file`, `address`); keys are case-insensitive and unknown keys/invalid values are rejected. Requires `--daemon` | +| `--no-detach` | Don't detach from parent | ✅ Implemented | Wave A: with `--daemon`, keeps the listener in the foreground (what integration tests use). Without it the daemonizes (fork/setsid, stdio redirected to /dev/null) after the listening socket is bound. Requires `--daemon` | +| `--password-file=FILE` | Read daemon password from file | ✅ Implemented | A7 daemon auth. Client: `--password-file` supplies `user:password` for a `host::module/path` destination (the username is taken from this file, so `user@host::module` stays rejected); the literal password is held client-side only for the SCRAM handshake and wiped at teardown. Server (`fastsync-server --daemon --password-file FILE`): the salted-PBKDF2 verifier store that modules with `auth users` are verified against. **Neither the password nor any replayable bearer value crosses the wire or is stored server-side** — the store holds a per-user salt plus derived keys, and the daemon proves the secret with a per-connection nonce challenge. The file must be private to its owner: both the client and server verify the exact inode they read (open-then-`fstat`, so the check cannot be raced) and refuse a `--password-file`/`--early-input` that is not owned by the current user or grants any group/other permission bit (mode 0600), mirroring the TLS private-key check. A process-substitution pipe (`--early-input <(vault ...)`) is still accepted when it satisfies those checks. See the Daemon Mode notes below for the file formats and the plaintext/TLS caveat | +| `--early-input=FILE` | Use FILE for daemon early exec | ✅ Implemented | Server-only (requires `--daemon`): a second credential-store file, same new-format grammar as `--password-file`, read before the listener accepts connections (a secrets-manager / process-substitution source). Its entries layer over `--password-file`: byte-identical verifiers dedupe, a conflicting verifier for the same user is a startup error. A daemon whose modules declare `auth users` must be given at least one of the two, or it refuses to start (fail closed) | +| `--hash-credentials=FILE`, `--iterations N` | Hash a plaintext credential file | ✅ Implemented | Server-only offline tool (A7): reads the `user:password` lines of FILE (same owner-only 0600 check) and prints one new-format store line per entry to stdout, then exits. `--iterations` sets the PBKDF2 work factor (default 600000, range 100000–10000000). Dependency-free and does not run a listener. Use its output as `--password-file` for `--daemon`. There is no auto-upgrade: a legacy store line is hard-rejected by the loader and must be regenerated | + +**Daemon Mode notes (Wave A protocol 2.15.0; A7 auth protocol 2.19.0; MOTD no bump):** FastSync daemon mode is supported in FastSync's own protocol/config grammar, not rsync's SMB/daemon option encoding. + +- **Config grammar** (`fastsyncd.conf`): line-based; an implicit global section first, then `[module]` sections. Keys are case-insensitive, values are trimmed and may be wrapped in one layer of double quotes (`path = "/srv/my dir"`). `#` and `;` at the start of a line (after leading whitespace) are full-line comments; inline comments and `\` continuations are not supported. Lines are bounded (4096 chars). Global keys: `port` (default 873), `motd file` (the daemon sends its bounded, escaped content to a client after the module gate/auth accepts, unless the client passes `--no-motd`), `address` (optional bind address). Module keys: `path` (required; the daemon-side authorized root for that module), `read only` (yes/no/true/false/1/0, default no), `client owner` (yes/no/true/false/1/0, default no; opts the module into client-chosen ownership — see below), `auth users` (comma list). **Unknown keys and malformed lines are parse-and-reject errors** (never silently ignored), so a typo cannot change what a module serves. +- **Module selection & confinement:** the client requests a module with an rsync-style `host::module[/path]` destination. The module name crosses the wire as a trailing string on the config frame (bumping `PROTOCOL_VERSION` 2.14.0 → 2.15.0; the bump is required because the config-frame layout changed and the strict same-version handshake is what prevents a peer from desynchronizing on the new trailing field). The daemon looks the module up in ITS OWN config and uses the module's `path` as the authorized root through the exact same `configure_authorization` confinement the standalone server applies to `--destination-root` (`file_open_secure_parent`, `has_path_traversal`, `path_is_within`); the client never supplies the root, every client-chosen-ownership/super-user request is refused unless the module declares `client owner = yes` (the daemon's per-module opt-in, see below), and the operator `--no-super` veto forces super-user activities off for every daemon connection. The client's `/path` part is relative inside the module and is rejected if absolute or if it contains `..`. Unknown modules are refused before any data moves (the run fails cleanly at the config handshake). An absolute destination and a module request against a non-daemon server are also refused. +- **`client owner` (client-chosen-ownership opt-in):** by default a daemon module refuses every request that would let the client pick an owner or ask for super-user activities — `--numeric-ids`, `--chown`, `--usermap`/`--groupmap`, `--fake-super`, `--copy-as`, and an explicit `--super` — at the config handshake (before `STATUS_OK`), because a daemon has no per-module opt-in for client-chosen ownership and any anonymous client could otherwise force arbitrary owner ids inside the module root. `client owner = yes` opts a single module in, allowing those requests within that module's root (the standalone listener and the SSH `--stdio` server always honor them for their single operator-authorized root). Without the opt-in the daemon also forces super-user **device** activity off for that connection — char/block device-node creation (`--devices`) and `--write-devices` — even under the default `AUTO` mode, so a non-opted module can never be made to `mknod` or write a raw device; those entries are skipped (not refused) so an ordinary `-a` push still succeeds without device nodes. The opt-in does **not** lift the privilege requirement: `--copy-as` still needs a root receiver, and the operator `--no-super` veto still forces super-user activities off for every connection. The daemon logs a prominent startup warning for each `client owner = yes` module so the operator's deliberate choice is visible. +- **`read only` safe default:** every network transfer FastSync currently supports is a push that writes under the module root, so a `read only` module refuses the connection (clear server log "module is read only"; the client exits non-zero, nothing is transferred). A future pull/list operation can be opened up when it exists; the knob is already stored. +- **`auth users` (A7 SCRAM-SHA-256 authentication):** a module that declares `auth users` requires the client to present credentials. The config frame carries ONLY the username; the daemon answers an auth-required module with `STATUS_AUTH_CHALLENGE` (PBKDF2 iteration count, 16-byte salt, 32-byte server nonce), the client answers with `STATUS_AUTH_RESPONSE` (fresh 32-byte client nonce + a 32-byte ClientProof), and the daemon accepts only when the proof verifies **and** the username is **on the module's `auth users` list** and has a store entry, replying `STATUS_AUTH_OK` with a 32-byte ServerSignature the client verifies before proceeding. Verification is constant-time over fixed 32-byte keys (the compare runs even for a miss), username membership uses a constant-time full-length scan, and an unknown/off-list user still receives a challenge and runs the same math against a dummy verifier: a deterministic per-username salt (`HMAC-SHA256(store dummy key, username)`), the store-wide uniform iteration count and dummy keys. Re-probing the same unknown username therefore yields an identical salt and iteration count while a different username yields a different salt, so there is no user-enumeration or timing oracle. The daemon logs the username but **never the password, proof or keys**. A module WITHOUT `auth users` stays open (legitimate rsync configuration); credentials sent to such a module are ignored. Read-only is orthogonal: even a correctly authenticated push to a `read only` module is still refused (all FastSync network transfers write). Fail-closed policy: a daemon whose config declares `auth users` on any module refuses to start unless a credential store was given (`--password-file` and/or `--early-input`); a missing or empty store is never silently treated as "open". A failed handshake (missing credentials, unknown/off-list user, wrong proof or malformed data) yields a single generic `STATUS_AUTH_FAILED` and the daemon closes before any data moves. The dummy key is persisted in an owner-only `.dummykey` sidecar (auto-created on first load, mode 0600) so the dummy salt stays stable across daemon restarts, closing the restart-gated enumeration channel. The sidecar is secret material and must be protected like the credential store (owner-only 0600, included with the store in backups and rotation). It must be preserved across restarts for that guarantee; if it cannot be created (a process-substitution/FIFO store path such as `/dev/fd/N`, a read-only filesystem, a missing directory, or a create/write/fsync/link/fchmod failure), the daemon logs a warning and uses a transient per-run key, so unknown-user challenges change across restarts and the cross-restart guarantee does not hold for that deployment. One residual is accepted: the store iteration count is observable pre-auth by design, since the miss path must match a hit. **Transport policy (hardening A7-3/S1):** an auth-required module accepts credentials only when either (a) the connection is an encrypted, verified TLS connection whose client certificate matches `--client-cn`, or (b) the connection is plaintext from a loopback TCP peer **and** the operator explicitly passed `--allow-unauthenticated`. A remote plaintext peer, and a loopback plaintext peer without that flag, are refused at the config gate before any challenge is sent; `--allow-unauthenticated` never permits remote plaintext auth (remote peers still require verified TLS). Daemon modules are a `--daemon`-only feature — the SSH `--stdio` path never loads a daemon config and is not an auth transport for them. Because the loopback allowance trusts whichever peer the kernel reports as `127.0.0.1`, it assumes nothing relays remote connections to the daemon: a local TCP forwarder or TLS-terminating proxy in front of an auth-module listener makes remote clients appear as loopback and bypasses the mutual-TLS identity check, so do not front an auth-module listener with such a relay. +- **Credential store format:** server `--password-file`/`--early-input` files are line-based `user:$fastsync$1$pbkdf2-sha256$$$$`, one per line (standard base64; 16-byte salt, 32-byte keys; `iters` in `[100000, 10000000]`, default 600000). Every entry in the resulting store must agree on `iters` (a store whose entries disagree, or where a layered `--early-input` disagrees with `--password-file`, is rejected). Generate lines with `fastsync-server --hash-credentials FILE [--iterations N]`; the emitted lines are secret material, so redirect them to an owner-only (mode 0600) file (the tool warns on stderr if stdout is a group/other-accessible regular file). Blank lines and lines starting with `#`/`;` are comments; the parser is strict (a malformed line fails the whole load, so a typo can never let a different set of users in). **The legacy `user:SHA256HEX` form is hard-rejected** with an actionable "legacy" error; there is no auto-upgrade, so a replayable bearer digest can never be loaded by a 2.19.0 daemon. The client `--password-file` holds `user:password` on its first meaningful line (the literal password, used only for the handshake then burned); keep both files readable only by their owner (mode 0600). Per-username wire length is bounded (256 chars) and every decoded salt/key length is validated. Loading the store also maintains an owner-only `.dummykey` sidecar (auto-created, mode 0600, exactly 32 bytes) holding the store-wide dummy key that shapes unknown-user challenges; persist it across daemon restarts so those challenges stay stable, and treat a sidecar with the wrong owner, a mode other than exactly 0600, the wrong size or the wrong type as a fatal load error (fail closed). If the sidecar cannot be created (e.g. a process-substitution store path such as `/dev/fd/N`, a read-only filesystem, a missing directory, or a create/write/fsync/link/fchmod failure), the daemon logs a warning and uses a transient per-run key, so the cross-restart stability guarantee does not hold there. +- **Plaintext caveat:** an auth-required module is refused, **before any challenge is sent**, unless the connection is encrypted and verified TLS whose client certificate matches the server's `--client-cn`, or it is plaintext from a loopback TCP peer **and** the operator passed `--allow-unauthenticated`. A remote plaintext peer, and a loopback plaintext peer without that flag, never receive a challenge, and `--allow-unauthenticated` never permits remote plaintext auth (remote peers still require verified TLS). On the loopback plaintext transport that remains permitted, a local sniffer could still read the challenge and response and mount an **offline dictionary attack** against a weak password, so use `--tls` for any real deployment. `--client-cn` matches the certificate CN only (not a subjectAltName), which is acceptable for a private CA. Clients sending daemon credentials with `--password-file` to a non-loopback daemon must use `--tls`; the client rejects such a destination before any network I/O. Unlike the old challenge-less exchange there is **no replay**: the proof is bound to the fresh per-connection server nonce, so a captured `STATUS_AUTH_RESPONSE` cannot be reused on another connection (an integration test proxies the daemon and proves this). TLS client-CN (`--client-cn`) is an independent transport identity check and composes with password auth; because `--tls` already mandates `--client-cn`, a TLS auth connection always verifies the client CN, so both checks necessarily apply together on such a connection. +- **Wire/protocol:** the config-frame auth block is now `[int present][str_redacted username]` (the old digest field is gone), and the frame stream gains the challenge/response (`STATUS_AUTH_CHALLENGE` → `STATUS_AUTH_RESPONSE` → `STATUS_AUTH_OK`/`STATUS_AUTH_FAILED`) between the config frame and the `STATUS_OK` ack. Both are wire-layout changes, so `PROTOCOL_VERSION` is bumped **2.18.0 → 2.19.0** (see the A7 note in `src/shared/config.h`); the strict same-version handshake keeps a 2.19 client and a 2.18 server from desynchronizing. +- **Client side:** `host::module/path` selects the TCP transport and connects to `--server-port`; `host:path` stays the SSH transport; plain paths stay local TCP. The daemon username comes from `--password-file` (first `user:password` line), and `--password-file` without a `host::module/path` destination is a client error (fail fast). A `user@host::module` form is rejected with a pointer to `--password-file`. The client's plaintext password is wiped from memory (`config_burn_auth`) at transfer teardown. +- **MOTD (Wave C):** a daemon configured with a global `motd file` sends that file's content as the first server→client string frame after the config-frame STATUS_OK ack (rsync sends the MOTD as the first thing from the server at the start of a daemon connection). Only the daemon listener path (`host::module`) gets a MOTD; the `--stdio` SSH path never sends or reads one. The server reads the file bounded to 4096 bytes and treats an absent/unreadable file as "no MOTD" (an empty frame, never an error). The exchange is server→client only and does **not** bump `PROTOCOL_VERSION`: every 2.15.0 daemon client reads the frame after the ack, so sender and receiver stay in lockstep (see the Wave C note in `src/shared/config.h`). `--no-motd` is the client-side suppression switch: the client still reads (consumes) the frame to keep the stream in sync but does not display it. The MOTD is printed to stdout with control bytes (ESC included) escaped octal-style while newlines/tabs are preserved, so a hostile server cannot inject terminal escape sequences. +- **Merge note:** the Wave A module bump (2.15.0) and the MOTD wave did not bump the version, but the A7 auth redesign is a genuine wire-layout change and owns the 2.18.0 → 2.19.0 bump (see the A7 note in `src/shared/config.h`). + +## 15. Safety & Security + +| Flag | Rsync Description | FastSync Status | Notes | +|------|-------------------|-----------------|-------| +| Path escape detection | Ensure files stay within root | ✅ Implemented | `has_path_traversal()` + realpath | +| Symlink-safe delete | Skip symlinks in delete walk | ✅ Implemented | `delete_extras_walk()` | +| Protocol version check | Verify compatible versions | ✅ Implemented | `config_receive()` | +| Max data/string/chunk sizes | Prevent OOM attacks | ✅ Implemented | Per-message limits | +| Per-connection memory limit | 1GB per connection | ✅ Implemented | `MAX_CONNECTION_MEMORY` | +| `--max-alloc=SIZE` | Limit a single memory allocation | ✅ Implemented | Caps the largest single allocation; binary units, default 1G | +| `--trust-sender` | Trust remote sender's file list | ✅ Implemented | Long-form-only, receiver-local policy that never crosses the wire. The receiver skips its redundant up-front re-validation of the incoming file list (empty/`..` path rejection and the escaping-symlink-target containment), trusting the sender instead of double-checking (fewer checks, faster, potentially unsafe, matching rsync). Off by default. The low-level fd-relative confinement primitives (`file_open_secure_parent`, the O_NOFOLLOW parent walk, leaf/destination confinement) are deliberately KEPT even under `--trust-sender`, so a hostile sender still cannot write or link outside the authorized root (see Phase-5 notes below) | +| `--old-args` | Disable modern arg protection | ✅ Implemented | SSH-only; accepted for CLI compatibility but is now a **documented no-op**: FastSync always single-quote-escapes the remote server path and each `--remote-option` value (`ssh_build_remote_command`), so a metacharacter-bearing `--rsync-path` can never be interpreted by the remote shell. The flag no longer disables that quoting (the old raw-construction behavior was an injection foot-gun and is removed); the safety-relevant behavior is identical either way | +| `--ignore-missing-args` | Ignore missing source args | ✅ Implemented | FastSync has a single source-root argument (which always exists), so the "explicitly requested source arguments" are the `--files-from` entries and the flags only ever apply there (inert without `--files-from`, like `-R`). Without the flag a listed-but-missing entry stays a hard pre-transfer error (nothing is transferred). With it each missing entry is skipped: nothing is sent for it, it never enters the keep-set, and the run succeeds for the rest — an all-missing non-empty list succeeds transferring nothing, matching rsync. `--dirs` + `--files-from` missing entries are skipped the same way. Every skipped entry is logged and a per-run warning names the count, so the handling is never a silent no-op. Divergences: an EMPTY `--files-from` file stays a hard error in every mode (no argument was requested at all; rsync likewise reports "no source files specified"); missing-arg skipping only applies to the pre-transfer list validation, so an entry that is present at preflight and vanishes mid-transfer still fails (matching rsync, whose flag "does not affect subsequent vanished-file errors"); `--no-ignore-missing-args` is not a supported negation | +| `--delete-missing-args` | Delete missing source args | ✅ Implemented | Implies `--ignore-missing-args` (order-independent) and additionally removes each missing entry's destination mirror receiver-side. The mirror is computed exactly like a present sibling's wire path: the bare relative entry under `-R`, otherwise the full source-mirror path below the destination root. rsync parity, verified against the man page: it does **not** imply `--delete` generally and is "independent of any other type of delete processing" — unrelated destination extras are untouched unless `--delete` is also present. Composition with `--delete` + timing: the exact-path deletions commit with the manifest, early for `--delete-before`/`--delete-during`, else only after a fully-successful transfer (delete-after/commit). A non-empty directory mirror is removed only when `--force` or `--delete` is in effect (otherwise it is left with a warning and the run continues, like rsync); an absent mirror is a no-op. An explicitly listed missing arg is a user request, not an excluded file: its deletion is never blocked by the filter-exclusion protection of excluded destination mirrors (a mirror sitting inside a filter-excluded directory is still removed). Safety/policy: gated by the server `--allow-delete` policy like `--delete`; the request paths cross the wire only in the delete-manifest frame and are confined by the same receiver validation as the keep-set (non-empty, relative, traversal-free, bounded by the per-section/per-frame manifest caps); the `--delay-updates` staging directory and basis snapshots are protected exactly as in the extras walker. Divergence: the missing-args deletions are not counted toward `--max-delete` (they are explicit per-path requests, not discovered extras). See the Phase-3 wire note below for the `PROTOCOL_VERSION` bump | + +## 16. Batch Operations + +| Flag | Rsync Description | FastSync Status | Notes | +|------|-------------------|-----------------|-------| +| `--write-batch=FILE` | Write batched update to file | ✅ Implemented | Phase-6 residual-batch (client-only): runs the normal live transfer AND additionally emits a self-contained single-file batch of the whole source tree. The batch is a magic/format-version header followed by length-prefixed `chunk_serialize` blobs (full file images), replayable byte-identically by `--read-batch` on another machine with no source/server. `--write-batch` drives the single-threaded transfer path (the multithreaded path consumes the config before the separate batch scan pass). See the Phase-6 batch note below | +| `--only-write-batch=FILE` | Write batch without updating dest | ✅ Implemented | Phase-6 residual-batch: emits the self-contained batch FILE only — NO destination update, NO server connection. Requires a source (scans it and serializes the full tree to FILE). Same single-file format as `--write-batch`, so the file is re-appliable via `--read-batch=FILE DEST`. See the Phase-6 batch note below | +| `--read-batch=FILE` | Read batched update from file | ✅ Implemented | Phase-6 residual-batch: applies a previously written batch FILE locally to the destination. NO source and NO server — positional args are the destination only. Reads the magic/version header, then length-prefixed records, `chunk_deserialize`, and applies each via the confined `file_save_to_disk_full` path (same O_NOFOLLOW / `..`-rejection / root-confinement as the network receiver, so an attacker-controlled batch cannot escape the destination root). Malformed/truncated/oversized/traversal records are rejected cleanly. See the Phase-6 batch note below | + +## 17. Advanced + +| Flag | Rsync Description | FastSync Status | Notes | +|------|-------------------|-----------------|-------| +| `--stop-after=MINS` | Stop after N minutes | ✅ Implemented | Client-only sender stop deadline (Phase 6): computing `--stop-after=MINS` (a positive minute count; 0/negative/garbage rejected) and `--stop-at=TIME` (`HH:MM`, `HH:MM:SS`, or `now+N[smhd]`; a past time stops immediately). The transfer stops ELEGANTLY at the next chunk boundary: everything already fully sent is kept and applied, the run returns 0, and --delete (late/delete-after timing) does NOT wipe the destination — when the scan is cut short the partial keep-set manifest is suppressed with a warning (the delete walk is skipped rather than acting on an incomplete keep-set, so unscanned source mirrors survive). `--delete-before`/`--delete-during` still run their complete pre-scan (which ignores the deadline). Local client-only fields: never serialized into the wire config frame, so no PROTOCOL_VERSION bump. `--stop-after` uses CLOCK_MONOTONIC; `--stop-at` uses the wall clock. Works single-threaded and under `-j`/`--threads` (multithreaded). Divergence: rsync computes `--stop-after` from the run start; FastSync likewise. When both are given, the earlier of the two deadlines wins (checked per iteration). See the Phase-6 stop notes below | +| `--stop-at=TIME` | Stop at specified time | ✅ Implemented | Same feature as `--stop-after` (deadline transfer stop), absolute wall-clock form (`HH:MM[:SS]` or `now+N[smhd]`). See the row above and the Phase-6 stop notes | +| `--fsync` | Fsync every written file before publication | ✅ Implemented | | +| `--protocol=NUM` | Force older protocol version | ✅ Implemented | Forces the wire protocol version for this transfer. FastSync has exactly ONE wire format (`PROTOCOL_VERSION`, currently 2.19.0) with no downgrade/backward-compat code paths, so `--protocol=2.19.0` is accepted (it sets the version claim the client sends, which the server already requires to match exactly) and **every other value is rejected up front** with a clear error before any connection — it does not and cannot speak an older or virtual wire format. Divergence from rsync (which negotiates a range and downgrades to an integer 0..31): FastSync's honest contract is force-to-the-one-supported-value; a genuine downgrade would require a per-version compatibility layer that does not exist. Client-only; the server-side exact-match check is unchanged. `--protocol=2.18.0`/`2.18`/`2.17.0`/`2.16.0`/`2.15.0`/`216`/`31`/garbage are all rejected. See the Phase-6 protocol note below | +| `--iconv=CONVERT_SPEC` | Charset conversion | ✅ Implemented | Charset conversion of FILE NAMES (not content) at the protocol boundary via iconv(3): `--iconv=LOCAL[,REMOTE]` — the sender converts each local filename LOCAL→REMOTE before transmitting, and the receiver converts each wire filename REMOTE→LOCAL before creating/writing. The full CONVERT_SPEC is serialized into the config frame as a new trailing string field so the peer knows the wire charset; **PROTOCOL_VERSION bumped 2.15.0 → 2.16.0**. `LOCAL[,REMOTE]` parse: single charset ⇒ LOCAL==REMOTE (identity both ways); garbage rejected up front. Validation probes BOTH directions (a spec that only opens one way is refused, as is a NUL-emitting target charset like utf-16/utf-32/ucs-2, since filenames cannot contain NUL). An unrepresentable name (EILSEQ/EINVAL) fails that path cleanly with a logged `--iconv: cannot convert file name ...` and is never written mangled/truncated. Conversion is applied at EVERY wire-path site (regular/MKDIR/hardlink path+target/symlink path+target/SPECIAL, the delete manifest, the incremental-check path, and the `-s`/`chunk_serialize` embedded blob path), on both client and server (`--iconv` is also a server/daemon option). Zero overhead when unset. See the Phase-6 iconv notes below | +| `--checksum-seed=NUM` | Set checksum seed | ✅ Implemented | Sets the seed for FastSync's whole-file xxHash64 digest (full 64-bit seed) and for the delta path's per-block xxHash32 strong checksum (low 32 bits of the seed). An explicit seed deterministically changes every computed digest on BOTH endpoints (sender and receiver share the seed via the config frame, protocol 2.10.0), so identical runs with the same seed skip the same files and a changed seed changes the digests — the explicit-seed path that makes xxHash comparisons deterministic. `--checksum-choice=md5` has no seed and ignores it (documented). The value is a strict decimal 0..2⁶⁴-1 (blank, signed, or non-numeric values are rejected). Like rsync, a seed only matters where a digest is actually computed (`--checksum` or a basis-dir run, or a delta transfer); it does not by itself enable `--checksum`/`--delta`. Divergence from rsync: the default is seed 0, and FastSync never randomizes the seed (rsync uses a random per-transfer seed when `--checksum-seed` is unset); FastSync's unset default therefore reproduces its historical byte-for-byte behavior | +| `--secluded-args`, `-s` | Use protocol to send args | ⛔ Impossible/Divergence | Accepted for CLI compatibility (including the rsync short `-s`, Phase 7 Wave A) but a documented **no-op / divergence**. rsync's `-s` protects arguments from shell expansion by shipping them over the protocol; FastSync never passes remote arguments through a shell expansion boundary in the first place — its SSH transport builds the remote argv as **single-quote-escaped shell words** (`ssh_build_remote_command`), so the injection/leak that `-s` guards against does not exist and there is nothing to "seclude". Implementing a true arg-send protocol would mean replacing the argv-based SSH launch with an in-band argument channel, a large redesign of the transport that buys no security here. Chunk serialization remains the long-only `--chunk-serialization`. | +| `--no-OPTION` | Turn off implied option | ✅ Supported | Supported boolean FastSync options and archive-implied options; unsafe or value-taking options are rejected. | + +--- + +## Implementation Difficulty Plan + +**Phase 5 notes (remote-option wave):** `--remote-option=OPT` (long form only) and `--trust-sender` landed here. +- `--remote-option` is CLIENT-only and never serialized into the binary config frame. On the SSH transport the client forwards each value to the remote server by appending it to the remote command line in `ssh_build_remote_command()`, after ` --stdio`, as an individually single-quoted shell word (`'...'` with `'\''` for embedded quotes). Values are validated at CLI parse time (non-empty; no ASCII control characters) and rejected otherwise, and a non-conforming value is refused again in the command builder, so shell metacharacters (`;`, `&`, `|`, backticks, `$()`, quotes) can never break out of the quoting to inject an unrelated remote command — including after a client-side `--` separator, whose arguments are never forwarded anyway. Because the remote options affect the *remote server invocation*, not the transmitted config, the wire frame layout is unchanged, but `PROTOCOL_VERSION` was bumped **2.13.0 → 2.14.0** as the Phase-5 lockstep release marker (a 2.14 client against a 2.13 server fails the version check cleanly rather than the old server rejecting an unfamiliar forwarded argv later). Divergence: rsync's short `-M` form of `--remote-option` was intentionally NOT implemented at that time because `-M` was FastSync metadata mode; **Phase 7 Wave A later freed `-M` for `--remote-option` and moved metadata to long-only `--preserve`** (see the Sending Options table). +- `--trust-sender` is a receiver-local policy: it never crosses the wire (the sender's value is never serialized, so a wire peer can never enable it). On the receiving process it skips the up-front re-validation of the incoming file list (empty/`..` path rejection and the escaping-symlink-target containment), trusting the sender's list instead of double-checking — fewer checks, faster, and potentially unsafe, matching rsync. It is OFF by default (`config.trust_sender`). As a deliberate safety floor, the low-level fd-relative confinement primitives are NOT disabled: `file_open_secure_parent()` (O_NOFOLLOW walk, `..` rejection, root containment) and leaf/destination confinement still hold, so even under `--trust-sender` a hostile sender cannot write or create a symlink outside the authorized root — the relaxation only removes the redundant list-layer double-checks, never the root-confinement guarantees. + +The estimates below cover the currently unimplemented features in this document. They assume one engineer familiar with the codebase, include implementation and focused tests, and exclude production rollout time. A feature should not be marked implemented until its behavior is tested in both local and SSH/TCP paths where applicable. + +> **Note:** This plan is a superset snapshot written while several of the listed features were still outstanding. The Summary matrix above is the authoritative record of what is already shipped (for example quiet/info/debug output, `--existing`, `--remove-source-files`, `-h`, and `--size-only` are now implemented on `dev`). Treat the phases as sequencing guidance for the work that remains unimplemented. + +| Effort | Typical duration | Meaning | +|--------|------------------|---------| +| XS | 0.5-1 day | CLI alias or a local formatting/validation change | +| S | 1-3 days | Isolated behavior with little or no protocol change | +| M | 3-7 days | Cross-cutting client, server, or scanner behavior | +| L | 1-3 weeks | Protocol, filesystem, privilege, or compatibility work | +| XL | 3+ weeks | New transfer mode, daemon subsystem, or broad interoperability effort | + +### Phase 1: Low-Risk CLI and Local Behavior + +These are the best first changes because they require limited wire-format work and can be tested with existing transfer fixtures. + +| Features | Effort | Implementation plan | +|----------|--------|--------------------| +| `--quiet`, `-q`; `--human-readable`, `-h`; `--8-bit-output`, `-8`; `--stderr=MODE`; `--info=FLAGS`; `--debug=FLAGS` | S | Extend logging and output formatting without changing transferred data. | +| `--no-OPTION`; `--old-args`; `--secluded-args`, `-s` | M | Add option implication/negation and safely serialize or protect remote arguments. `-s` currently has FastSync-specific semantics and needs a compatibility decision. | +| `-P`; `--del`; `--old-dirs`, `--old-d`; `--cc`; `--zc`; `--zl` | XS | Add aliases and composed behaviors after the underlying options exist. | +| `--whole-file`, `-W`; `--ignore-times`, `-I`; `--size-only`; `--modify-window`, `-@`; `--update`, `-u` | S | Extend the existing incremental comparison decision. | +| `--existing`; `--ignore-existing`; `--remove-source-files` | S | Add scanner/receiver eligibility checks and remove successfully synchronized source files. | +| `--executability`, `-E`; `--chmod=CHMOD` | M | Apply permission transformations safely while preserving current metadata behavior. | +| `--skip-compress=LIST`; `--compress-threads=NUM` | S | Make compression selection configurable and validate the thread setting against zstd behavior. | +| `--max-alloc=SIZE`; `--fsync` | S | Reuse existing allocation limits and add an explicit durability step after file writes. | + +### Phase 2: Filesystem Selection and Update Semantics + +These features are moderate because they affect traversal, temporary files, manifests, or the receiver's update policy. + +| Features | Effort | Implementation plan | +|----------|--------|--------------------| +| `--one-file-system`, `-x` | M | Track the source device during scanner traversal and skip mount-point crossings. | +| `--relative`, `-R`; `--no-implied-dirs`; `--dirs`, `-d`; `--mkpath` | M | Extend path-list construction and destination directory creation while preserving traversal safety. | +| `--temp-dir`, `-T` | M | Separate temporary-file placement from FastSync's timeout alias and define collision, permissions, and cleanup rules. | +| `--delay-updates` | L | Stage all successful updates and publish them at completion, including crash and cancellation cleanup. | +| `--files-from=FILE`; `--from0`, `-0`; `--filter=RULE`, `-f`; `-F`; `--cvs-exclude`, `-C` | L | Build a complete filter/parser layer and integrate it with scanner pruning, manifests, and delete behavior. `-f` conflicts with FastSync sendfile mode. | +| `--list-only`; `--itemize-changes`, `-i`; `--out-format=FORMAT`; `--log-file-format=FMT` | M | Add a structured change-event model so output modes share one source of truth. | + +### Phase 3: Deletion, Comparison, and Delta Compatibility + +These features require careful interaction with manifests, incremental checks, backups, and the existing delta protocol. + +| Features | Effort | Implementation plan | +|----------|--------|--------------------| +| `--delete-during`; `--delete-before`; `--delete-after`; `--delete-delay`; `--del` | L | Add deletion timing to the transfer state machine and ensure failures cannot remove files unexpectedly. | +| `--delete-excluded`; `--max-delete=NUM`; `--ignore-errors`; `--force`; `--prune-empty-dirs`, `-m` | M | Extend delete walks with policy limits, error handling, empty-directory pruning, and the `-m` short-flag conflict. | +| `--ignore-missing-args`; `--delete-missing-args` | M | Distinguish missing source arguments from traversal errors and apply explicit deletion policy. | +| `--compare-dest=DIR`; `--copy-dest=DIR`; `--link-dest=DIR` | L | Add alternate basis roots and hard-link handling, including metadata and cross-filesystem failures. | +| `--fuzzy`, `-y`; `--no-fuzzy` | L | Index candidate files and select a safe similar basis without making transfer time unbounded. | +| `--append`; `--append-verify` | M | Negotiate file length and verify the retained prefix before resuming. | +| `--checksum-choice=STR`, `--cc`; `--checksum-seed=NUM` | M | Negotiate checksum algorithms/seeds and preserve compatibility with existing xxHash checks. | + +### Phase 4: Metadata, Links, and Devices + +These features are platform-sensitive and need Linux permission, ACL, xattr, and special-file integration tests. + +| Features | Effort | Implementation plan | +|----------|--------|--------------------| +| `--numeric-ids`; `--usermap=STRING`; `--groupmap=STRING`; `--chown=USER:GROUP` | L | Define identity mapping, privilege failures, and wire representation before applying ownership. | +| `--open-noatime`; `--atimes`, `-U`; `--crtimes`, `-N`; `--omit-dir-times`, `-O`; `--omit-link-times`, `-J` | L | Extend metadata capture/apply with platform capability checks and explicit unsupported-attribute handling. | +| `--acls`, `-A`; `--xattrs`, `-X`; `--fake-super` | XL | Add portable serialization, size limits, privilege behavior, and security tests for ACL/xattr data. | +| `--hard-links`, `-H` | L | Preserve inode relationships across the file list and coordinate hard-link creation order. | +| `--munge-links`; `--copy-dirlinks`, `-k`; `--keep-dirlinks`, `-K` | L | Define symlink trust boundaries and receiver-side directory/link collision behavior. | +| `--devices`; `--specials`; `-D`; `--copy-devices`; `--write-devices` | XL | Add privileged special-file handling with strict type, path, and authorization checks. | +| `--super`; `--copy-as=USER[:GROUP]` | XL | Requires a deliberate privilege model, identity switching, and refusal paths; do not implement by blindly elevating the process. | +| `--preallocate` | S | Use platform allocation APIs before writes and fall back cleanly when unsupported. | + +### Phase 5: Connectivity and Daemon Compatibility + +These options affect process startup, authentication, sockets, and remote execution. They should follow the filesystem and protocol work rather than being added as parser-only flags. + +| Features | Effort | Implementation plan | +|----------|--------|--------------------| +| `--rsh=COMMAND`, `-e`; `--rsync-path=PROGRAM`; `--blocking-io`; `--outbuf=N\|L\|B` | M | ✅ Wave A implemented (see the Connectivity table above). SSH argv construction is generalized: `-e`/`--rsh` replaces the hardcoded `ssh` program (whitespace-split, so `-e "ssh -p 2222"` works), `--rsync-path` aliases the existing `fastsync_server_path`, `--blocking-io` drops the SSH socket timeouts, and `--outbuf` maps N/L/B onto `setvbuf`. All four are client-only launch concerns and never cross the wire. | +| `--address=ADDRESS`; `--ipv4`, `-4`; `--ipv6`, `-6`; `--sockopts=OPTIONS`; `--port=PORT` daemon semantics | M | Add explicit socket-family/bind configuration and validate it independently for TCP client and daemon modes. | + +**Phase 5, Wave B (socket/bind) shipping note:** `--sockopts` adds a strict allowlisted `OPT=VAL` socket-option layer applied with correct per-option value types; `--address` binds the outgoing client socket to a local source address; `-4`/`-6` pin the address family via `getaddrinfo` hints on both the client connect and the server bind; and the server bind now honors `--address` plus `-4`/`-6` (falling back to the historical IPv4 `INADDR_ANY` when none are given). All of these are local socket concerns and none cross the wire config frame (only `--port` maps to `server_port`). +| `--remote-option=OPT`, `-M`; `--trust-sender` | L | Add authenticated remote-option/config negotiation and reject unsafe sender-controlled values. `-M` conflicts with FastSync metadata mode. | +| `--daemon`; `--config=FILE`; `--dparam=OVERRIDE`; `--no-detach`; `--password-file=FILE`; `--early-input=FILE`; `--no-motd` | XL | Implement a real daemon lifecycle, module configuration, authentication, privilege separation, and process management. | + +**Phase 5, Wave A (rsh/ssh) shipping note:** the SSH transport no longer hardcodes `ssh`. `-e`/`--rsh=COMMAND` selects the remote-shell program (whitespace-split into the leading child argv words), `--rsync-path=PROGRAM` aliases `--fastsync-server-path`, `--blocking-io` removes the SSH-socketpair `SO_RCVTIMEO`/`SO_SNDTIMEO` timeouts (by default they now match the TCP transport so a wedged shell cannot hang forever), and `--outbuf=N|L|B` maps onto `setvbuf` (`_IONBF`/`_IOLBF`/`_IOFBF`, garbage rejected). All four are client-only launch concerns and never cross the wire. + +**Phase 5, Wave C (remote-option/trust-sender) shipping note (PROTOCOL 2.13.0 → 2.14.0):** `--remote-option=OPT` (long form only; the short `-M` is intentionally left as FastSync metadata mode — documented divergence) appends each validated value to the remote server invocation over SSH as an individually single-quote-escaped shell word, so shell metacharacters cannot break out and a `--` can never be turned into injection; options never cross the binary config frame. `--trust-sender` is a receiver-local policy (never serialized, so a wire peer can't enable it): when requested on the server (via `--remote-option=--trust-sender`), it removes only the redundant receiver/save-layer path re-checking; the low-level floor (`file_open_secure_parent`'s `..` rejection, the O_NOFOLLOW parent walk, leaf/destination confinement) stays enforced. Off by default. The wire config-frame layout is unchanged; the bump reflects that a 2.14 sender composing remote options requires a 2.14 receiver to honor them. + +### Phase 6: Batch, Encoding, and Protocol Interoperability + +These are the hardest compatibility items because they require durable formats or behavior that must interoperate with rsync itself. + +| Features | Effort | Implementation plan | +|----------|--------|--------------------| +| `--write-batch=FILE`; `--only-write-batch=FILE`; `--read-batch=FILE` | XL | ✅ Implemented (see the Batch Operations table and Phase-6 batch note below): a versioned self-contained single-file residual-batch format, persisted via the existing chunk codec, with replay, corruption, and partial-application safety tests | +| `--protocol=NUM` | XL | ✅ Implemented (see the Advanced table and Phase-6 protocol note below): protocol-version forcing without weakening current validation; FastSync's single lockstep wire format means only the current `PROTOCOL_VERSION` is accepted, and everything else is rejected up-front | +| `--iconv=CONVERT_SPEC` | L | ✅ Implemented (see the Advanced table and Phase-6 iconv notes below): filename charset conversion at the wire boundary with expansion/overflow safety and invalid-sequence test coverage | +| `--stop-after=MINS`; `--stop-at=TIME` | M | ✅ Implemented (see the Advanced table and Phase-6 stop notes below): deadline propagation and safe early stop with --delete safety | +| `--early-input=FILE`; `--password-file=FILE` | M | Securely read startup credentials/input with permission checks and no secret disclosure in logs. | + +**Phase 6, Wave A (stop deadline) shipping note:** `--stop-after=MINS` and `--stop-at=TIME` are client-only sender stop deadlines. `--stop-after` takes a positive minute count (0/negative/garbage rejected); `--stop-at` takes `HH:MM`, `HH:MM:SS`, or `now+N[smhd]` (a past time stops immediately, a garbage spec is rejected at parse time). The deadline is computed once at the start of the transfer (CLOCK_MONOTONIC for `--stop-after`, wall clock via `time()` for `--stop-at`) and checked at every chunk boundary in both the single-threaded `send_files` loop and the multithreaded `send_chunks_multithreaded` path, and inside the scanner loops so a busy scan itself stops. When it fires, the transfer stops ELEGANTLY: the in-flight chunk completes, the existing completion tail runs (summary, `disconnect`), and the run returns 0 — exactly like rsync's clean early stop. Because the deadline is client-only and never crosses the wire config frame, no PROTOCOL_VERSION bump is required. The safety-critical interaction is with `--delete`: FastSync streams while scanning, so a deadline can cut the source scan short and yield a PARTIAL keep-set manifest; committing that would make the receiver delete destination mirrors of source files not yet scanned. So the sender tracks `scan_stopped_early` and, when it is true on the late/delete-after (`--delete`/`--delete-after`/`--delete-delay`) path, SUPPRESSES the keep-set manifest (logs a warning) so no deletion happens from an incomplete set — this is the safe direction (preserves data; the delete simply does not run). `--delete-before`/`--delete-during` are unaffected: their complete pre-scan runs before any data and ignores the deadline (a stop can be exceeded by that pre-scan). Under `-j`/`--threads` the stop is symmetric and the scanner thread's still-in-progress manifest appends can never race the tail because the tail does not read the manifest on the early-stop path. + +**Phase 6, Wave B (iconv) shipping note (PROTOCOL 2.15.0 → 2.16.0):** `--iconv=LOCAL[,REMOTE]` converts file NAMES at the wire boundary (never content). The full CONVERT_SPEC is serialized into the config frame as a new trailing string field (empty→NULL canonicalized), so both ends share the same wire charset interpretation; this required the PROTOCOL bump because the frame is a strict ordered sequence and a peer that does not parse the new trailing field would desynchronize. Each end derives LOCAL (its own charset) and REMOTE (the wire charset): the sender opens LOCAL→REMOTE and converts every transmitted filename; the receiver opens REMOTE→LOCAL and converts every received filename before creating/writing. Conversion is applied at every wire-path site (regular/MKDIR/hardlink path+target/symlink path+target/SPECIAL, the delete manifest keep/protected/missing entries, the incremental-check path, and the embedded `-s`/chunk-blob path). A name it cannot convert (EILSEQ/EINVAL) is failed cleanly with a logged `--iconv: cannot convert file name ...` and is never written truncated/mangled. Validation probes both directions up front (both the sender local→remote and the receiver remote→local, and, for a server/daemon with its own `--iconv`, the client-REMOTE→server-LOCAL pair) so an unusable spec is rejected before the connection rather than mid-transfer, and NUL-emitting target charsets (utf-16/utf-32/ucs-2) are refused because filenames cannot contain NUL. Divergence documented upstream: the receiver does NOT half-swap; the wire charset always comes from the sender's REMOTE half, so a server whose local charset differs from the client's LOCAL must declare it with its own `--iconv`. Conversion is process-global and runs on a single thread per process (sender thread / receiver-loop thread), initialized before worker threads start and freed after they join. + +**Phase 6, Wave C (protocol-version) shipping note (no PROTOCOL_VERSION change):** `--protocol=NUM` lets the client force the wire protocol version for a transfer. FastSync's protocol is a single lockstep format: the config frame is a strict ordered sequence and the server requires the client's version string to equal `PROTOCOL_VERSION` exactly (`config_receive_with_validate`, src/shared/config.c) — there are no older-format code paths and no downgrade/negotiation machinery, so a lower/higher/virtual version can never be spoken. The honest contract is therefore: `--protocol=2.19.0` (the current `PROTOCOL_VERSION`, as of the A7 auth redesign) is accepted and stored into the client's `version` claim (which `config_send` already transmits), and every other value — `2.18.0`, `2.18`, `2.17.0`, `2.16.0`, `2.15.0`, `3.0.0`, rsync-integer spellings like `216`/`31`, garbage, empty — is rejected up front in `validate_config()` before any connection, with a clear error that FastSync supports only its current wire protocol and cannot speak an older or virtual one. Implementation is client-only: a server-side `--protocol` is intentionally not added because the server has no negotiation (it only enforces exact match), and it could only ever be the current version. This preserves (and slightly tightens) existing validation: the client now also refuses to launch with a version it cannot actually speak, rather than only the server rejecting it later. A genuine downgrade would require a per-version compatibility layer for every frame/feature added since (append 2.10, preallocate 2.11, hardlinks 2.12, devices/specials/symlink-trust/xattr 2.13, remote-option 2.14, daemon module/auth 2.15, iconv 2.16, dir/symlink times 2.17, privilege flags --super/--copy-as 2.18, SCRAM daemon auth 2.19) and is intentionally out of scope — documented divergences from rsync's integer-negotiated downgrade remain. + +**Phase-1/2 selection-and-update status correction (docs):** `-I/--ignore-times`, `--size-only`, `-@/--modify-window`, `--existing`, `--ignore-existing`, `-u/--update`, `-W/--whole-file`, and `--compress-threads` were previously listed as not-implemented in this document but are in fact fully implemented and tested on `dev`. This pass corrects the matrix to match the code. The realistic model of these is that FastSync is a *sender-driven* whole-tree copy, so the size+mtime quick-check and all three receiver-policy skips (`--existing`, `--ignore-existing`, `-u`) are evaluated against the **destination** on the receiver side, and their booleans cross the wire in the config frame. `-I`/`--size-only`/`--modify-window` modify the `--incremental` per-file `STATUS_CHECK` handshake's match predicate (`-I` disables the mtime leg and forces transfer; `--size-only` drops only the mtime leg; `--modify-window` adds tolerance to `metadata_mtime_matches`); they require `--incremental` (or a basis dir) to have a handshake to affect, mirroring how they only matter where a quick-check exists in rsync. `--existing`/`--ignore-existing`/`-u` are receiver write-time policies (skipping the write / newer-destination guard) applied across the regular-file, `--delay-updates`-staged, hardlink-sibling, and special/device paths; `-u` implies `-M` metadata and uses a second-then-nanosecond strict `>` newer check; both correctly influence `--remove-source-files` (a skipped source is not removed). `-W/--whole-file` disables block-level delta (opt-in via `--delta`), folded into the wire `use_delta` so no protocol bump was needed, and makes `--fuzzy` inert; `--append`/`--append-verify` are rejected with `-W`. `--compress-threads=NUM` (1..64, client-only, never crosses the wire) sizes the zstd compression worker pool. No code was changed by this correction; the implementation had landed in earlier merge waves (feat/ignore-times, feat/ignore-existing via the newer `file_to_disk_secure_no_replace`/`linkat EEXIST` path, feat/size-only, feat/modify-window, feat/whole-file, feat/update, compression-threads). + +**Phase 6, Wave D (batch) shipping note (no PROTOCOL_VERSION change):** FastSync batch mode is a **client-only, self-contained "residual batch"**: a single file `MAGIC "FSTRESBATCH" + format version 1 + metadata flag`, followed by length-prefixed `chunk_serialize` blobs that store full file images (regular files, dirs, symlinks, specials). It is NOT a raw capture of the live wire, because FastSync's protocol is per-file interactive (`STATUS_CHECK`/`STATUS_DELTA_SIGNATURE`/`STATUS_APPEND` handshake), so a raw sender-stream tee is not deterministically replayable against an arbitrary destination. Storing full residuals via the existing, fuzz-tested chunk codec makes `--read-batch` replay byte-identically by construction. `--write-batch=FILE` runs the normal live transfer AND emits the batch from a separate deterministic scan pass; `--only-write-batch=FILE` emits the batch only (no destination, no server); `--read-batch=FILE DEST` applies it locally (no source, no server; DEST is the only positional arg). Because batch is a local driver concern, it never crosses the wire: no new config-frame field and no `PROTOCOL_VERSION` bump (mirroring `--stop-after`/`--protocol`/`--compress-threads`). The READ side is hardened against untrusted/attacker-controlled batch files: magic+version validated before any record, per-record length bounds checked before allocation (64 MB cap), clean-EOF-after-prefix and truncated/oversized records rejected, and every applied path goes through the same confined `file_save_to_disk_full` machinery as the network receiver (O_NOFOLLOW fd-walk, `..`-rejection, root confinement — a malicious `../` or absolute/symlink path cannot escape the destination root; this was security-reviewed and valgrind/ASan-clean). Divergences from rsync: (1) the batch carries the FULL residual (complete file images) rather than rsync's update-only delta stream — always byte-correct but larger; (2) per-file data is capped at the chunk codec's ~64 MB (`BATCH_MAX_RECORD`), so very large files may be refused by the batch writer with a clean error (never a corrupt/truncated batch); (3) hard-links and xattr/ACL blocks are not represented by `chunk_serialize`, so `-H`/`-X`/`-A` are out of scope for batch; (4) there is no companion `.sh`/`.rsync_argvs` — the batch is invoked directly (`fastsync --read-batch=FILE DEST`, `--only-write-batch=FILE SOURCE`); (5) `--write-batch` drives the single-threaded transfer path. Integration/`-M` note: metadata is captured in the batch when `-M` is used and persisted in the header so it applies consistently regardless of the reading process's own `-M`. + +### Phase 7: CLI-Namespace Parity, Filesystem/Output Completion, and Privilege (Final) + +These are the last compatibility items and the closing phase toward rsync flag parity. Per the project decision: every rsync flag (short **and** long) that is *possible* gets real rsync-parity behavior; anything physically impossible becomes an explicit **Impossible/Divergence** status (accepted for CLI compatibility, safely inert, with coverage tests proving that); and the two privilege flags (`--super`, `--copy-as`) adopt the deliberately-scoped **safe-subset + clear-refusal** model rather than blind elevation. The remaining `⚠️ Partial`, `🔄 Compatibility No-op`, `🔀 Alt Arg`, and `❌ Not Implemented` rows in the Summary are this phase's scope. All Wave A renames are **client-side only** (the wire config fields `use_compression`/`use_metadata`/`use_sendfile`/`use_chunk_serialization` are unchanged), so they require **no `PROTOCOL_VERSION` bump**. + +**Wave A — CLI namespace parity (rename colliding FastSync short flags) — ✅ implemented.** This freed the short letters rsync needs and made the three `🔀 Alt Arg` rows real. `-c`→`--checksum`, `-m`→`--prune-empty-dirs`, `-M`→`--remote-option`, `-f`→`--filter`, `-s`→`--secluded-args`, `-p`→`--perms`, `-T`→`--temp-dir`, `-a`/`--archive`→real `-rlptgoD`. FastSync's own flags moved to long-form-only or new shorts: `-j`/`--threads` (multithreading), `--preserve` (metadata), `--sendfile`, `--chunk-serialization`, `--timeout`, `--ssh-port`. The server's independent little CLI keeps `-p` as its port. All client-side, no wire change, no `PROTOCOL_VERSION` bump. Unit tests 37/37, full integration 400 passed, cppcheck and clang-format clean. Known Wave-A limitation: `--no-perms`/`--no-compress`-style negation of the newly-aliased shorts is not wired into the negatable set (only the long-form `--preserve`/`--compress`/`--no-links` negations exist); `--archive --no-perms` is consequently not supported yet — a minor deviation from rsync, acceptable for Wave A. + +| FastSync flag today | rsync wants that name | Proposed rename | +|---------------------|----------------------|-----------------| +| `-c` / `--compress` | `-c` = `--checksum` | compression is already aliased as `-z`/`--compress` (rsync parity!) → drop the `-c` short, keep `--compress`/`-z` | +| `-m` / `--multithreading` | `-m` = `--prune-empty-dirs` | → `-j` / `--threads` | +| `-M` / `--preserve` | `-M` = `--remote-option` | → `--preserve` (long-only) | +| `-f` / `--sendfile` | `-f` = `--filter` | → `--sendfile` (long-only) | +| `-s` / `--chunk-serialization` | `-s` = `--secluded-args`/`--protect-args` | → `--chunk-serialization` (long-only) | +| `-p` (SSH port) | `-p` = `--perms` | → `--port` (long-only; `--server-port` already exists) | +| `-T` / `--timeout` | `-T` = `--temp-dir` | → `--timeout` (long-only) | +| `-a` / `--archive` (= `-c -m -M`) | `-a` = `-rlptgoD` | → becomes **real rsync `-a`** after the renames | + +**Wave B — Output & filesystem completion (✅ implemented).** `-S`/`--sparse` (`⚠️→✅`): real hole preservation — a sparse-aware writer (`write_all_sparse`) skips all-zero runs ≥ 4096 bytes with `lseek(SEEK_CUR)` and `ftruncate`s the final size, wired into both the atomic temp+rename store and `--inplace` receiver-side with **no wire change** (the full file image is already in memory; the ftruncate presize is kept). `-P` (`⚠️→✅`): interrupted-write retention — on a save failure after data reached the temp fd, `--partial` now renames the already-written temp to the destination path (best-effort; falls through to the normal unlink on failure, never retains when `--partial` is off) so a later `--append`/`--append-verify` run can resume. `--block-size=SIZE` (`⚠️→✅`): promoted after verification — `--block-size` is now an alias for `--delta-block`, both set `config->delta_block_size`, which the delta engine already honored end-to-end (`delta_signature_create_seeded` + `delta_apply`); out-of-range values keep the default. `--fake-super` (`⚠️→✅`): added `fake_super_restore_fd` to parse and re-apply the recorded `user.fastsync.stat` record fd-relative (fchown best-effort/non-root skipped, fchmod, futimens); a save under `--fake-super` now re-applies the recorded attrs instead of only recording them, with the recording format unchanged. `--stderr=client` (`⚠️→⛔ Impossible/Divergence`): FastSync has no rsync client-message channel, and `client` is rejected at CLI parse — the rejection is the documented behavior (unit-tested). `-N`/`--crtimes` (`⚠️→⛔ Impossible/Divergence`): birth-times cannot be set by any portable fs call (`utimensat` sets only atime/mtime); capture/transmit stays, setting is impossible, the flag is accepted and safely inert. Review-hardening (post-eval): fake-super replay applies the mode through the same sanitization as the normal metadata path (group/other write bits are never granted); `--sparse` takes precedence over `--preallocate` (posix_fallocate skipped so holes survive); `--partial` retention is disabled for `--no_replace` (ignore/existing) and only marks a write-attempt after the actual write begins; `--block-size=SIZE`/`--delta-block=SIZE` inline forms are accepted. + +**Wave C — Devices & special files (finalize statuses + tests) (✅ implemented).** The four special-file rows are finalized with coverage tests. `--devices`, `--copy-devices`, and `--write-devices` are **✅ Implemented**, each with a documented, safety-driven divergence: device-node creation is privilege-gated, so a receiver without `CAP_MKNOD` skips that entry with a warning (a per-entry skip, never a transfer failure); `--copy-devices` copies a device/FIFO's reported size into an ordinary regular file (a size-bounded safe divergence from rsync's unbounded dd-like read); `--write-devices` writes only into an existing char/block node under the confined receive root and skips every unusable target rather than clobbering or aborting. `--specials` is classified **⛔ Impossible/Divergence** for one reason only: **FIFO recreation works** (unprivileged `mkfifo`, asserted under CI), but **sockets cannot be recreated by any standard filesystem call**, so a source socket is skipped with an explicit note. Tests assert FIFO recreation, the safe socket skip, the regular-file result of `--copy-devices`, the skipped/missing and non-device `--write-devices` targets, and (root-gated) real device-node creation; a root runner additionally drops the receiver to an unprivileged user to assert the `CAP_MKNOD` skip is graceful. + +**Wave D — Times superstructure & arg-protection no-ops (✅ implemented, `--secluded-args` ⛔).** `-O`/`--omit-dir-times` and `-J`/`--omit-link-times` are now **real modifiers** (both `🔄 → ✅ Implemented`), reversing the old "never preserves directory/symlink times" divergence: + +- **Directory times.** The recursive scanner captures every traversed source directory's metadata (mtime, plus atime under `-U`) into a per-transfer list — two paths are covered: the sequential `DirectoryScanner` captures each opened directory (including the transfer root), and the parallel scanner captures both the root in `parallel_scanner_create_with_options` and each worker's subdirectories in `open_next_directory` (appends are guarded by a mutex shared with the sender's pipeline context). The sender transmits them in trailing `STATUS_DIR_TIMES` frames (each: int count + count × (wire path, metadata) pairs) sent **after all file data and after the optional delete manifest**, just before `STATUS_FINISHED`. A tree larger than `MAX_MANIFEST_ENTRIES` (1 048 576) directories is chunked into repeated frames, each within the receiver's per-frame bound. A dir-time entry is RECORD-ONLY (`file->dir_time_only`): `file_save_to_disk_full` returns `FILE_SAVE_SKIPPED` without creating anything, so a source directory that was empty (or pruned by `-m/--prune-empty-dirs`) is never resurrected. The receiver accumulates received directory metadata in a `DirTimeList` and applies it only at the very end — after the entire stream, after the commit-style `--delete` deletion, and after `--delay-updates` publication — because creating or removing a child bumps the parent's mtime. Application is fd-relative/walk-confined (`file_open_secure_parent` + `utimensat(..., AT_SYMLINK_NOFOLLOW)`) and best-effort per entry: an absent path (an intentionally uncreated empty dir) is skipped QUIETLY and only a real existing directory is stamped. `-O` (config boolean, already on the wire) makes the receiver skip the whole set. The single-threaded sink applies in `receiver_send_success_frame`; the `-j`/`--threads` sink accumulates in `write_thread` and server.c applies after both threads join and the deletion commits. +- **Symlink times/owner/mode.** `STATUS_SYMLINK` already carried metadata; the receiver now applies it with no-follow primitives only: `utimensat(..., AT_SYMLINK_NOFOLLOW)`, best-effort `fchmodat(..., AT_SYMLINK_NOFOLLOW)` (honest no-op where unsupported, e.g. Linux), and policy-gated `fchownat(..., AT_SYMLINK_NOFOLLOW)` via a new `identity_apply_ownership_link` that shares the identity resolver with the fd path. `-J` suppresses only the timestamps; ownership stays governed by the identity opt-in (`--numeric-ids`/`--usermap`/`--groupmap`/`--chown`) exactly like regular files. A symlink has no children, so this is applied immediately at creation. +- **Wire:** the shared `STATUS_DIR_TIMES` frame (and metadata on `STATUS_MKDIR` for `--dirs` entries) is a frame-sequence change, so `PROTOCOL_VERSION` was bumped **2.16.0 → 2.17.0**; every version-sensitive test (`--protocol` accepted/rejected values) was updated. The config-frame layout itself is unchanged (the omit booleans already crossed). Non-metadata and `--no-preserve` transfers send no `STATUS_DIR_TIMES` frame and no directory metadata, keeping them byte-identical. + +`--secluded-args` (`🔄 → ⛔ Impossible/Divergence`): a true arg-send protocol would replace the argv-based SSH launch with an in-band channel, and FastSync already builds the remote SSH argv injection-safe (single-quote-escaped shell words), so there is no argument-leak to close; the already-safe behavior is documented in the row and no transport change is made. + +**Wave E (LAST) — Privilege: `--super`/`--no-super` and `--copy-as=USER[:GROUP]` (✅ implemented).** FastSync adopts a **safe-subset + clear-refusal** privilege model: it never blind-elevates and never calls `setuid`/`seteuid`/`setgid`. All privileged operations remain fd-relative and confined below the authorized receive root. + +`--super`/`--no-super` set a receiver-side tri-state `Config->super_mode` (`SUPER_MODE_AUTO`/`ON`/`OFF`). `privilege_super_permitted()` / `privilege_super_mode_permitted()` (src/shared/identity.c) return true for `ON` and `AUTO` (AUTO preserves FastSync's historical best-effort attempt, where the kernel refuses an unprivileged call and the caller skips it) and false only for `OFF`. The gate covers every super-user activity FastSync performs: ownership application (`identity_apply_ownership`/`_link`), char/block device-node creation (`file_save_special_to_disk`), writes into an existing device (`--write-devices`), and the `--fake-super` owner replay. Unprivileged FIFO creation is deliberately unaffected. `--super` does **not** imply `--numeric-ids`: ownership is applied only when an explicit identity policy (`--usermap`/`--groupmap`/`--chown`/`--numeric-ids`/`--copy-as`) is also given. `--no-super` suppresses those activities even for a root receiver. A non-root receiver given `--super` logs one warning at activation (`identity_set_active`); each confined attempt is then refused by the kernel and skipped, never aborting. The confinement floor is unchanged (`file_open_secure_parent`, `O_NOFOLLOW`, root/path checks). Operator control: the server CLI accepts `--no-super`, a veto that forces `OFF` for every connection, refuses any client `--copy-as`, and neutralizes an explicit `--super` (the connection is accepted but no super-user activity is attempted). On a daemon, a module that has not opted in with `client owner = yes` additionally has super-user device activity forced off (see the Daemon Mode notes). + +`--copy-as=USER[:GROUP]` is the safe subset. FastSync's receiver is multithreaded, so a real credential switch is unsafe; instead the receiver forces the ownership of **every entry it writes** — regular files, symlinks, directories (including implicitly-created parents), and special nodes — to the resolved target ids through the confined fd-relative identity path. USER is resolved on the client (name, `@N`/bare N, or `*` = client euid); when `:GROUP` is omitted the user's primary gid is used (falling back to `gid == uid` for a numeric id with no local passwd entry). It requires a privileged (root) receiver: an unprivileged receiver refuses the whole transfer at the config handshake, before `STATUS_OK`, so no data is ever written with the wrong ownership. A `--copy-as` chown failure on a capability-restricted root is logged at ERROR (never silently downgraded). `--copy-as` implies metadata (`--no-preserve` is rejected) and `--fake-super` cannot override it. Daemon policy: a `--daemon` receiver refuses **every** client-chosen-ownership / super-user request — `--numeric-ids`, `--chown`, `--usermap`/`--groupmap`, `--fake-super`, `--copy-as`, and explicit `--super` — unless the selected module opts in with `client owner = yes`; without that per-module opt-in any client could force arbitrary ownership inside the module root (the standalone listener and the SSH-launched `--stdio` server, which each serve one operator-authorized root, honor these requests). A `--copy-as` chown failure on a capability-restricted root marks the entry as failed rather than reporting success with the wrong owner. + +**Wire:** two trailing config-frame blocks after the `--iconv` spec, in fixed order — `send_privilege_options`/`receive_privilege_options` (one `super_mode` int, validated `0..2`), then `send_copy_as_options`/`receive_copy_as_options` (presence int + two int32 ids, validated `>= 0`, with `copy_as_set ⇒ use_metadata`). `PROTOCOL_VERSION` bumped **2.17.0 → 2.18.0**. **Divergences from rsync:** rsync's `--super` elevates the receiver and `--copy-as` actually switches its credentials; FastSync never elevates and only permits/forwards confined attempts, and `--copy-as` forces ownership rather than switching identity. + +**Post-Phase-7 Summary (after Waves A–E).** ✅143 / 🔀0 / ⛔4 / ⚠️0 / 🔄0 / ❌0 = 147. The 3 `🔀 Alt Arg` rows (`-a`, `-p`, `-z`) are ✅ (Wave A). All 10 prior `⚠️ Partial` rows are resolved to ✅ (`-S`, `-P`, `--block-size`, `--fake-super`, `--devices`, `--copy-devices`, `--write-devices`) or ⛔ (`--stderr=client`, `-N/--crtimes`, `--specials` for the impossible socket case). The 3 `🔄 Compatibility No-op` rows are resolved: `-O`/`-J` are now real ✅ (Wave D), `--secluded-args` is ⛔. The **Impossible/Divergence** bucket holds the 4 physically-impossible/divergent flags: `--stderr=client`, `-N/--crtimes`, `--specials` (sockets), `--secluded-args`. The last two `❌ Not Implemented` rows — `--super` and `--copy-as=USER[:GROUP]` — are now ✅ (Wave E). **No `❌ Not Implemented` rows remain.** + +### Recommended Delivery Order + +1. Resolve short-option conflicts (`-m`, `-M`, `-T`, `-f`, `-s`) and define the compatibility contract. +2. Implement Phase 1 comparison, update, output, and alias features with unit and integration coverage. +3. Implement Phase 2 traversal/filtering and Phase 3 deletion semantics. +4. Implement metadata and link features that are safe on the supported platforms. +5. Treat daemon mode, special files, batch mode, and protocol-version compatibility as separate projects. + +The existing priority list below is a feature shortlist, not an implementation schedule; this plan supersedes it for effort and sequencing. + +--- + +## Recommendations: Top Features to Implement Next + +Ranked by user demand, implementation complexity, and interoperability impact (_status reflects current `dev`_): + +| Priority | Feature | Effort | Impact | +|----------|---------|--------|--------| +| 1 | `--whole-file` / `-W` | Low | High — users expect opt-out of delta — **✅ implemented** | +| 2 | `--ignore-times` / `-I` | Low | Medium — useful for forcing re-transfer — **✅ implemented** | +| 3 | `--size-only` | Low | Medium — common migration scenario — **✅ implemented** | +| 4 | `--ignore-existing` | Low | Medium — common sync patterns — **✅ implemented** | +| 5 | `--existing` | Low | Medium — common sync patterns — **✅ implemented** | +| 6 | `--remove-source-files` | Low | High — common for moves/backup | +| 7 | `--delete-during` | Medium | High — performance improvement | +| 8 | `--delay-updates` | Medium | High — atomic updates | +| 9 | `--chmod` | Low | Medium — permission flexibility | +| 10 | `--executability` / `-E` | Low | Low — simple flag | +| 11 | `--skip-compress` | Low | Medium — performance tuning | + +--- + +## FastSync-Specific Features (Not in rsync) + +| Feature | Description | +|---------|-------------| +| `-j` / `--threads` | Multithreaded pipeline (scanner/loader/sender) (renamed from `-m` in Phase 7 Wave A; `-m` is now rsync `--prune-empty-dirs`) | +| `--chunk-serialization` | Chunk serialization mode (long form only; `-s` is now rsync `--secluded-args`) | +| `--sendfile` | Zero-copy sendfile() syscall (TCP only) (long form only; `-f` is now rsync `--filter`) | +| `-z [level]` / `--compress` | zstd compression level (1-22) (`-c` is now rsync `--checksum`) | +| `--chunk-size` | Configurable chunk size | +| `--tls` | TLS encryption (mutual auth) | +| `--fastsync-server-path` | Path to fastsync-server binary | +| `--server-host` / `--server-port` | Direct TCP connection | +| Incremental sync | Skip unchanged files (size+mtime) | +| Delta transfer | Block-level delta for changed files | diff --git a/pytest.ini b/pytest.ini new file mode 100644 index 0000000..9cd9e47 --- /dev/null +++ b/pytest.ini @@ -0,0 +1,10 @@ +[pytest] +; Fast integration subset run on every pull request (see .gitea/workflows/ci.yaml). +markers = + ci: fast, representative integration tests run on the PR CI gate + setpriv: privilege-dependent tests (drop to an unprivileged user); excluded + from CI because their result depends on the runner/container uid and the + host mount permissions, but run locally as root + daemon_detach: real double-fork backgrounding path (--daemon without + --no-detach); slower/fragile, so it runs in the full suite but not the + fast PR gate diff --git a/src/client/change_list.c b/src/client/change_list.c new file mode 100644 index 0000000..5cddaf0 --- /dev/null +++ b/src/client/change_list.c @@ -0,0 +1,329 @@ +#include "change_list.h" +#include "utils.h" +#include +#include +#include +#include +#include +#include +#include + +/* Itemize code emitted for a transferred regular file. + * + * Layout (rsync-compatible 11-char item): `>f` marks a regular file that was + * transferred to the remote host; the trailing nine markers are, in order, + * c(hecksum) s(ize) t(ime) p(erms) o(wner) g(roup) u(ser/acl) a(ttrs) x(attrs). + * Every marker is `+` (FastSync does not compare each attribute on the + * receiving side, so a sent file is reported as fully updated). Files that + * are already up to date print no line at all, matching rsync's single -i + * which only itemizes changes. + * + * Because the scanner only yields regular-file transfer candidates, `>d` + * (directory) lines are never produced; directories are not transferred as + * items by FastSync. */ +#define ITEMIZE_SENT_FILE ">f+++++++++" + +typedef struct { + char* data; + size_t length; + size_t capacity; +} StrBuf; + +static void strbuf_free(StrBuf* buf) { + if (buf == NULL) + return; + free(buf->data); + buf->data = NULL; + buf->length = 0; + buf->capacity = 0; +} + +static bool strbuf_reserve(StrBuf* buf, size_t extra) { + if (buf->length > SIZE_MAX - extra - 1) + return false; + size_t need = buf->length + extra + 1; + if (need <= buf->capacity) + return true; + size_t capacity = buf->capacity > 0 ? buf->capacity : 32; + while (capacity < need) { + if (capacity > SIZE_MAX / 2) { + capacity = need; + break; + } + capacity *= 2; + } + char* grown = realloc(buf->data, capacity); + if (!grown) + return false; + buf->data = grown; + buf->capacity = capacity; + return true; +} + +static bool strbuf_append_char(StrBuf* buf, char c) { + if (!strbuf_reserve(buf, 1)) + return false; + buf->data[buf->length++] = c; + buf->data[buf->length] = '\0'; + return true; +} + +static bool strbuf_append(StrBuf* buf, const char* text) { + if (text == NULL) + return true; + size_t length = strlen(text); + if (!strbuf_reserve(buf, length)) + return false; + memcpy(buf->data + buf->length, text, length); + buf->length += length; + buf->data[buf->length] = '\0'; + return true; +} + +static bool strbuf_append_ull(StrBuf* buf, unsigned long long value) { + char digits[32]; + int written = snprintf(digits, sizeof(digits), "%llu", value); + if (written < 0 || (size_t)written >= sizeof(digits)) + return false; + return strbuf_append(buf, digits); +} + +static bool strbuf_append_longlong(StrBuf* buf, long long value) { + char digits[32]; + int written = snprintf(digits, sizeof(digits), "%lld", value); + if (written < 0 || (size_t)written >= sizeof(digits)) + return false; + return strbuf_append(buf, digits); +} + +bool change_list_enabled(const Config* config) { + return config != NULL && (config->itemize_changes || config->out_format != NULL || + (config->log_file != NULL && config->log_file_format != NULL)); +} + +char* change_render_itemize(const ChangeEvent* event) { + if (event == NULL || event->decision != CHANGE_SENT) + return str_dup(""); + const char* code = event->is_directory ? ">d+++++++++" : ITEMIZE_SENT_FILE; + StrBuf line = {0}; + bool ok = strbuf_append(&line, code) && strbuf_append(&line, " ") && + strbuf_append(&line, event->path != NULL ? event->path : ""); + if (!ok) { + strbuf_free(&line); + return NULL; + } + return line.data; +} + +static const char* leaf_name(const char* path) { + if (path == NULL) + return ""; + const char* slash = strrchr(path, '/'); + return slash != NULL && slash[1] != '\0' ? slash + 1 : path; +} + +char* change_render_format(const char* format, const ChangeEvent* event) { + if (format == NULL) + return NULL; + StrBuf line = {0}; + bool ok = true; + for (const char* p = format; *p != '\0' && ok;) { + if (*p != '%') { + ok = strbuf_append_char(&line, *p); + p++; + continue; + } + char token = p[1]; + if (token == '\0') { + ok = strbuf_append_char(&line, '%'); + break; + } + switch (token) { + case '%': + ok = strbuf_append_char(&line, '%'); + break; + case 'f': + ok = strbuf_append(&line, event->path != NULL ? event->path : ""); + break; + case 'n': + ok = strbuf_append(&line, leaf_name(event->path)); + break; + case 'l': + ok = strbuf_append_ull(&line, event->size); + break; + case 'b': + ok = strbuf_append_ull(&line, event->bytes_sent); + break; + case 'M': + ok = strbuf_append_longlong(&line, (long long)event->mtime_sec); + break; + default: + /* Unknown escape sequences are preserved verbatim. */ + ok = strbuf_append_char(&line, '%') && strbuf_append_char(&line, token); + break; + } + p += 2; + } + if (!ok) { + strbuf_free(&line); + return NULL; + } + if (line.data == NULL) { + line.data = str_dup(""); + if (!line.data) + return NULL; + } + return line.data; +} + +/* Format a mode as an `ls -l` permission string, e.g. `-rw-r--r--`. */ +static void mode_to_ls_string(mode_t mode, char out[11]) { + out[0] = S_ISDIR(mode) ? 'd' + : S_ISLNK(mode) ? 'l' + : S_ISCHR(mode) ? 'c' + : S_ISBLK(mode) ? 'b' + : S_ISFIFO(mode) ? 'p' + : S_ISSOCK(mode) ? 's' + : '-'; + mode_t bits = mode & 07777; + out[1] = (bits & S_IRUSR) ? 'r' : '-'; + out[2] = (bits & S_IWUSR) ? 'w' : '-'; + out[3] = (bits & S_IXUSR) ? (bits & S_ISUID ? 's' : 'x') : (bits & S_ISUID ? 'S' : '-'); + out[4] = (bits & S_IRGRP) ? 'r' : '-'; + out[5] = (bits & S_IWGRP) ? 'w' : '-'; + out[6] = (bits & S_IXGRP) ? (bits & S_ISGID ? 's' : 'x') : (bits & S_ISGID ? 'S' : '-'); + out[7] = (bits & S_IROTH) ? 'r' : '-'; + out[8] = (bits & S_IWOTH) ? 'w' : '-'; + out[9] = (bits & S_IXOTH) ? (bits & S_ISVTX ? 't' : 'x') : (bits & S_ISVTX ? 'T' : '-'); + out[10] = '\0'; +} + +char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime, + const char* path) { + char permission[11]; + mode_to_ls_string(mode, permission); + char date[32]; + struct tm broken_down; + if (localtime_r(&mtime, &broken_down) != NULL) { + if (strftime(date, sizeof(date), "%Y/%m/%d %H:%M:%S", &broken_down) == 0) + snprintf(date, sizeof(date), "?"); + } else { + snprintf(date, sizeof(date), "?"); + } + StrBuf line = {0}; + char size_field[32]; + int written = snprintf(size_field, sizeof(size_field), "%llu", size); + if (written < 0 || (size_t)written >= sizeof(size_field)) { + strbuf_free(&line); + return NULL; + } + bool ok = strbuf_append(&line, permission) && strbuf_append_char(&line, ' ') && + strbuf_append(&line, size_field) && strbuf_append_char(&line, ' ') && + strbuf_append(&line, date) && strbuf_append_char(&line, ' ') && + strbuf_append(&line, path != NULL ? path : ""); + if (!ok) { + strbuf_free(&line); + return NULL; + } + return line.data; +} + +static void print_escaped_line(FILE* stream, const char* line, bool eight_bit_output) { + char* escaped = output_escape(line, eight_bit_output); + if (escaped != NULL) { + fprintf(stream, "%s\n", escaped); + free(escaped); + } else { + fprintf(stream, "%s\n", line); + } + fflush(stream); +} + +void change_emit(const Config* config, const ChangeEvent* event) { + if (event == NULL || !change_list_enabled(config)) + return; + if (event->decision == CHANGE_UP_TO_DATE) + return; + bool to_stdout = config->itemize_changes || config->out_format != NULL; + bool to_log = config->log_file != NULL && config->log_file_format != NULL; + if (to_stdout) { + char* line = config->out_format != NULL ? change_render_format(config->out_format, event) + : change_render_itemize(event); + if (line != NULL) { + print_escaped_line(stdout, line, config->eight_bit_output); + free(line); + } + } + if (to_log) { + char* line = change_render_format(config->log_file_format, event); + if (line != NULL) { + print_escaped_line(config->log_file, line, config->eight_bit_output); + free(line); + } + } +} + +static bool format_uses_mtime(const char* format) { + if (format == NULL) + return false; + /* Mirror change_render_format's tokenizer: "%%" is a literal percent (so + * "%%M" does NOT expand %M) and unknown "%X" escapes consume both chars. + * This keeps the optional stat() fallback below in step with the renderer. */ + for (const char* p = format; *p != '\0';) { + if (*p != '%') { + p++; + continue; + } + char token = p[1]; + if (token == '\0') + break; + if (token == 'M') + return true; + p += 2; + } + return false; +} + +void change_emit_file_sent(const Config* config, const File* file) { + if (file == NULL || !change_list_enabled(config)) + return; + ChangeEvent event; + memset(&event, 0, sizeof(event)); + /* The displayed path is the one transmitted (with -R + --files-from this is + the bare relative destination path); the metadata fallback below still + stats the local absolute path. */ + event.path = file_wire_path(file); + event.decision = CHANGE_SENT; + event.is_directory = false; + event.size = file->data != NULL ? file->data->size : 0; + /* FastSync has no wire-byte counter yet, so %b reports the source length + * that had to be delivered (always equal to %l); the actual bytes written + * to the socket (compressed/delta) are not measured. */ + event.bytes_sent = event.size; + if (file->metadata != NULL) { + event.mtime_sec = file->metadata->mtime_sec; + } else if (format_uses_mtime(config->out_format) || format_uses_mtime(config->log_file_format)) { + /* Best-effort fallback for %M when no metadata was captured (no -M): the + * path is stat()ed just to fill the field, and any failure leaves 0. */ + struct stat st; + if (file->path != NULL && stat(file->path, &st) == 0) + event.mtime_sec = st.st_mtime; + } + change_emit(config, &event); +} + +/* Build and emit a CHANGE_SENT event for an explicit directory entry (-d). */ +void change_emit_dir_sent(const Config* config, const File* file) { + if (file == NULL || !change_list_enabled(config)) + return; + ChangeEvent event; + memset(&event, 0, sizeof(event)); + event.path = file_wire_path(file); + event.decision = CHANGE_SENT; + event.is_directory = true; + event.size = 0; + event.bytes_sent = 0; + if (file->metadata != NULL) + event.mtime_sec = file->metadata->mtime_sec; + change_emit(config, &event); +} diff --git a/src/client/change_list.h b/src/client/change_list.h new file mode 100644 index 0000000..9434026 --- /dev/null +++ b/src/client/change_list.h @@ -0,0 +1,78 @@ +#ifndef CHANGE_LIST_H +#define CHANGE_LIST_H + +#include "config.h" +#include "file_types.h" +#include +#include +#include + +/* + * Shared per-file change-event / output model (rsync --itemize-changes, + * --out-format, --log-file-format, and --list-only all render from here). + * + * FastSync is a push-style tool: the client sends files from the source tree + * to a server that writes them under the destination root. Events are + * emitted by whichever code path decides a file's fate (the single-threaded + * send loop and the `-m` sender thread both call the same per-file sender), so + * all change events are emitted by exactly one thread and itemize/out-format + * lines never interleave with each other. They may still interleave with + * legacy log messages (log.c) that share the same stdout/log-file stream. + */ + +typedef enum { + CHANGE_SENT, /* file data (full or delta) was transmitted */ + CHANGE_UP_TO_DATE, /* receiver already had an identical file; skipped */ +} ChangeDecision; + +typedef struct { + const char* path; /* full source path */ + ChangeDecision decision; + bool is_directory; + unsigned long long size; /* source file length in bytes */ + /* The number of bytes reported for a sent file. FastSync has no wire-byte + * counter, so this is always the source length (== size / %l); actual + * post-compression/delta bytes on the wire are not counted. */ + unsigned long long bytes_sent; + time_t mtime_sec; /* 0 when unknown */ +} ChangeEvent; + +/* True when any output mode is active and per-file events matter. */ +bool change_list_enabled(const Config* config); + +/* Render the rsync-style itemize line for a transferred file: + * `>f+++++++++ ` + * The 11-char code is `>f` (regular file transferred to the remote host) + * followed by c/s/t/p/o/g/u/a/x markers that are all `+` (value will be set + * / differs) because FastSync does not separately compare checksums, size, + * mtime, perms, owner, group, uid, acl, or xattr on the receiving side, so a + * sent file is reported as fully updated. Up-to-date files print no line + * (rsync single `-i` only shows changes). Caller frees the result. */ +char* change_render_itemize(const ChangeEvent* event); + +/* Expand an --out-format/--log-file-format template. Tokens: + * %f full source path %b "bytes sent" == the source length (%l); + * %n leaf (base) name actual post-compression/delta wire bytes + * %l file length in bytes are not counted + * %M mtime in whole seconds %% a literal percent sign + * Unknown %X sequences are preserved verbatim. Caller frees the result. */ +char* change_render_format(const char* format, const ChangeEvent* event); + +/* Render one --list-only long-listing entry: + * `-rw-r--r-- 12 2026/09/06 10:00:00 ` + * (ls -l style columns; mtime in the local time zone). Caller frees it. */ +char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime, const char* path); + +/* Emit an event to every active destination: + * stdout: --itemize-changes line, or the --out-format expansion when set; + * log file: the --log-file-format expansion (requires --log-file). + * CHANGE_UP_TO_DATE events produce no output. */ +void change_emit(const Config* config, const ChangeEvent* event); + +/* Build and emit a CHANGE_SENT event for a file the client just sent. */ +void change_emit_file_sent(const Config* config, const File* file); + +/* Build and emit a CHANGE_SENT event for an explicit directory entry (-d). */ +void change_emit_dir_sent(const Config* config, const File* file); + +#endif diff --git a/src/client/client_cli.c b/src/client/client_cli.c index de1fc3c..9c10610 100644 --- a/src/client/client_cli.c +++ b/src/client/client_cli.c @@ -1,17 +1,33 @@ #include "client_send.h" +#include "client_validation.h" +#include "charset.h" +#include "chmod.h" +#include "compression.h" #include "config.h" +#include "credentials.h" #include "delta.h" +#include "file.h" +#include "file_list.h" +#include "filter.h" +#include "identity.h" #include "log.h" #include "protocol.h" +#include "stop_condition.h" #include "transport_tcp.h" #include "transport_tls.h" +#include "usage.h" #include "utils.h" #include #include +#include +#include +#include +#include #include #include #include +#ifndef FASTSYNC_TEST_BUILD /* Parse environment variables for source/destination directories and save-to-disk flag. */ static void parse_environment(const char** out_env_source, const char** out_env_dest, bool* out_save_to_disk) { @@ -22,19 +38,7 @@ static void parse_environment(const char** out_env_source, const char** out_env_ if (env_save && (strcmp(env_save, "true") == 0 || strcmp(env_save, "1") == 0)) *out_save_to_disk = true; } - -/* Parse a string as a positive integer, returning true on success. */ -static bool parse_positive_int(const char* s, int* out_val) { - if (!s || *s == '\0') - return false; - char* endptr; - errno = 0; - long val = strtol(s, &endptr, 10); - if (errno != 0 || *endptr != '\0' || val <= 0 || val > INT_MAX) - return false; - *out_val = (int)val; - return true; -} +#endif /* Parse a string as a non-negative integer, returning true on success. */ static bool parse_nonneg_int(const char* s, int* out_val) { @@ -49,520 +53,1504 @@ static bool parse_nonneg_int(const char* s, int* out_val) { return true; } -static void print_usage(void); +/* Parse a string as a positive integer, returning true on success. */ +static bool parse_positive_int(const char* s, int* out_val) { + int temp; + if (!parse_nonneg_int(s, &temp)) { + return false; + } + if (temp == 0) { + return false; + } + *out_val = temp; + return true; +} + +/* Duplicate a string argument into *dest, freeing the old value. Returns 0 on success, -1 on + * failure. */ +static int set_string_option(char** dest, const char* value, const char* option_name) { + char* dup = str_dup(value); + if (!dup) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed for %s", option_name); + return -1; + } + free(*dest); + *dest = dup; + return 0; +} + +/* Parse a string as a positive integer into *dest. Returns 0 on success, -1 on error. */ +static int set_positive_int_option(int* dest, const char* value, const char* option_name) { + if (!parse_positive_int(value, dest)) { + log_message(LOG_LEVEL_ERROR, "%s must be a positive integer", option_name); + return -1; + } + return 0; +} + +/* Set and validate the compression algorithm selected by the client. */ +static int set_compression_choice(Config* config, const char* value) { + if (strcmp(value, "zstd") != 0 && strcmp(value, "none") != 0) { + log_message(LOG_LEVEL_ERROR, "--compress-choice must be zstd or none"); + return -1; + } + if (set_string_option(&config->compress_choice, value, "--compress-choice") != 0) + return -1; + config->use_compression = strcmp(value, "zstd") == 0; + return 0; +} + +/* Validate and store the --checksum-choice/--cc algorithm. Only the algorithms + * the engine genuinely supports are accepted (xxHash64 and md5); anything else + * is a clear error, never a silent no-op. "xxhash" is accepted as rsync's + * spelling of xxHash64. */ +static int set_checksum_choice(Config* config, const char* value) { + int algo = checksum_algo_from_name(value); + if (algo < 0) { + log_message(LOG_LEVEL_ERROR, "--checksum-choice must be xxh64 (or xxhash) or md5 (got '%s')", + value); + return -1; + } + config->checksum_algo = algo; + return 0; +} + +/* parse_ull_arg is defined later in this file; declared here for the seed + parser below. */ +static int parse_ull_arg(const char* val, unsigned long long* out, const char* optname); + +/* Parse --checksum-seed=NUM as a strict decimal 0..UINT64_MAX. A blank value, + * a sign, or any non-digit (which parse_ull_arg's strtoull would silently + * coerce) is rejected: an explicit seed must be an exact unsigned integer or + * the run fails with a clear error rather than quietly ignoring the value. */ +static int set_checksum_seed(Config* config, const char* value) { + if (!value || *value == '\0') { + log_message(LOG_LEVEL_ERROR, "--checksum-seed must be a non-negative integer"); + return -1; + } + for (const char* p = value; *p; p++) { + if (*p < '0' || *p > '9') { + log_message(LOG_LEVEL_ERROR, "--checksum-seed must be a non-negative integer"); + return -1; + } + } + unsigned long long seed; + if (parse_ull_arg(value, &seed, "--checksum-seed") != 0) + return -1; + config->checksum_seed = seed; + return 0; +} + +static int set_compression_threads_option(int* dest, const char* value) { + if (set_positive_int_option(dest, value, "--compress-threads") != 0) + return -1; + if (*dest > COMPRESSION_MAX_THREADS) { + log_message(LOG_LEVEL_ERROR, "--compress-threads must be between 1 and %d", + COMPRESSION_MAX_THREADS); + return -1; + } + return 0; +} + +/* Parse and validate --sockopts=OPTIONS into the config. The strict allowlist + * (config_sockopts_parse) rejects an unknown option name or an invalid value + * up front, so a typo never silently disables a socket option. */ +static int set_sockopts_option(Config* config, const char* value) { + SockOptEntry* entries = NULL; + int count = 0; + if (config_sockopts_parse(value, &entries, &count) != 0) { + log_message(LOG_LEVEL_ERROR, + "--sockopts must be a comma-separated OPT=VAL list of supported options " + "(TCP_NODELAY, SO_KEEPALIVE, SO_RCVBUF, SO_SNDBUF, SO_REUSEADDR)"); + return -1; + } + free(config->sockopts); + config->sockopt_count = count; + config->sockopts = entries; + return 0; +} + +/* Parse a string as a non-negative integer into *dest. Returns 0 on success, -1 on error. */ +static int set_nonneg_int_option(int* dest, const char* value, const char* option_name) { + if (!parse_nonneg_int(value, dest)) { + log_message(LOG_LEVEL_ERROR, "%s must be a non-negative integer", option_name); + return -1; + } + return 0; +} + +/* Forward decl: config_add_pattern is defined below, but the --remote-option + * helper above needs it. */ +static int config_add_pattern(char*** patterns, int* count, const char* value, const char* optname); + +/* Validate and append one --remote-option=OPT value. OPT is forwarded to the + * remote server invocation (over SSH) by appending it to the remote command + * line, so it must be a single safe shell word: it must be non-empty and must + * contain no control characters that could break the single-quoted command + * word ssh_build_remote_command wraps it in (newline/CR and other ASCII + * control chars are rejected up front). Ordinary shell metacharacters + * (; & | ` $ () etc.) need not be rejected because they are neutralized by the + * single-quoting boundary, but rejecting control characters keeps the + * quoting scheme airtight regardless of the remote shell. Returns 0 on + * success, -1 on a rejected value. */ +static int config_add_remote_option(Config* config, const char* value, const char* optname) { + if (!value || value[0] == '\0') { + log_message(LOG_LEVEL_ERROR, "%s requires a non-empty option value", optname); + return -1; + } + for (const unsigned char* p = (const unsigned char*)value; *p; p++) { + if (*p < 0x20 || *p == 0x7f) { + log_message(LOG_LEVEL_ERROR, + "%s value contains a control character that could break the remote shell " + "quoting; rejecting", + optname); + return -1; + } + } + if (config_add_pattern(&config->remote_options, &config->remote_option_count, value, optname) != + 0) + return -1; + return 0; +} + +/* Validate and append one --compare-dest/--copy-dest/--link-dest directory. + * The path is interpreted on the receiver relative to the destination root, + * so it must be a non-empty relative path with no "." / ".." components (an + * absolute or escaping path is rejected up front instead of failing on the + * server). Returns 0 on success, -1 on error. */ +static int set_basis_dest_option(Config* config, BasisDestType type, const char* value, + const char* option_name) { + if (!value || !value[0]) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", option_name); + return -1; + } + if (config_basis_append(config, type, value) != 0) { + log_message(LOG_LEVEL_ERROR, + "%s requires a non-empty relative directory name with no '.', '..', or absolute " + "path (resolved below the destination root)", + option_name); + return -1; + } + return 0; +} + +static int set_stderr_mode(const char* value) { + if (strcmp(value, "errors") == 0 || strcmp(value, "e") == 0) + log_set_stderr_mode(LOG_STDERR_ERRORS); + else if (strcmp(value, "all") == 0 || strcmp(value, "a") == 0) + log_set_stderr_mode(LOG_STDERR_ALL); + else if (strcmp(value, "client") == 0 || strcmp(value, "c") == 0) { + log_message(LOG_LEVEL_ERROR, + "--stderr=client is not supported: FastSync has no client message channel"); + return -1; + } else { + log_message(LOG_LEVEL_ERROR, "--stderr must be errors or all"); + return -1; + } + return 0; +} + +/* Parse --outbuf=N|L|B into the config's OutbufMode. N=none (unbuffered), + * L=line-buffered, B=block-buffered (the stdio default). Anything else is a + * clear error, never a silent fallback. */ +static int set_outbuf_option(Config* config, const char* value) { + if (strcmp(value, "N") == 0 || strcmp(value, "n") == 0) + config->outbuf = OUTBUF_NONE; + else if (strcmp(value, "L") == 0 || strcmp(value, "l") == 0) + config->outbuf = OUTBUF_LINE; + else if (strcmp(value, "B") == 0 || strcmp(value, "b") == 0) + config->outbuf = OUTBUF_BLOCK; + else { + log_message(LOG_LEVEL_ERROR, "--outbuf must be N (none), L (line), or B (block)"); + return -1; + } + return 0; +} + +#ifndef FASTSYNC_TEST_BUILD +/* Apply the parsed --outbuf style to stdout/stderr via setvbuf, matching stdio + * semantics: N -> _IONBF (unbuffered), L -> _IOLBF (line), B -> _IOFBF (block, + * the default). */ +static void apply_output_buffering(const Config* config) { + int mode = config->outbuf; + int stdio_mode = (mode == OUTBUF_NONE) ? _IONBF : (mode == OUTBUF_LINE) ? _IOLBF : _IOFBF; + setvbuf(stdout, NULL, stdio_mode, 0); + setvbuf(stderr, NULL, stdio_mode, 0); +} +#endif + static int read_patterns_from_file(const char* filepath, char*** patterns, int* count); +static int parse_debug_flags(const char* value, Config* config) { + if (!value || value[0] == '\0' || value[0] == ',' || value[strlen(value) - 1] == ',' || + strstr(value, ",,")) { + log_message(LOG_LEVEL_ERROR, "--debug requires at least one flag"); + return -1; + } + + char* flags = str_dup(value); + if (!flags) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed for --debug"); + return -1; + } + uint32_t parsed = (uint32_t)config->debug_level; + char* saveptr = NULL; + for (char* token = strtok_r(flags, ",", &saveptr); token != NULL; + token = strtok_r(NULL, ",", &saveptr)) { + uint32_t flag = 0; + if (strcmp(token, "help") == 0) { + print_debug_usage(); + free(flags); + return 1; + } else if (strcmp(token, "all") == 0) { + parsed = LOG_DEBUG_ALL; + continue; + } else if (strcmp(token, "none") == 0) { + parsed = 0; + continue; + } else if (strcmp(token, "io") == 0) { + flag = LOG_DEBUG_IO; + } else if (strcmp(token, "proto") == 0) { + flag = LOG_DEBUG_PROTO; + } else if (strcmp(token, "pack") == 0) { + flag = LOG_DEBUG_PACK; + } else if (strcmp(token, "util") == 0) { + flag = LOG_DEBUG_UTIL; + } else { + log_message(LOG_LEVEL_ERROR, "unsupported --debug flag: %s", token); + free(flags); + return -1; + } + parsed |= flag; + } + free(flags); + config->debug_level = (int)parsed; + set_log_debug_flags(parsed); + set_log_level(LOG_LEVEL_DEBUG); + return 0; +} + +static int parse_info_flags(const char* value, Config* config) { + if (!value || value[0] == '\0' || value[0] == ',' || value[strlen(value) - 1] == ',' || + strstr(value, ",,")) { + log_message(LOG_LEVEL_ERROR, "--info requires at least one flag"); + return -1; + } + char* flags = str_dup(value); + if (!flags) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed for --info"); + return -1; + } + + uint32_t parsed = (uint32_t)config->info_level; + char* saveptr = NULL; + for (char* token = strtok_r(flags, ",", &saveptr); token != NULL; + token = strtok_r(NULL, ",", &saveptr)) { + uint32_t flag = 0; + if (strcmp(token, "all") == 0) { + parsed = LOG_INFO_ALL; + continue; + } + if (strcmp(token, "none") == 0) { + parsed = 0; + continue; + } + if (strcmp(token, "copy") == 0) + flag = LOG_INFO_COPY; + else if (strcmp(token, "misc") == 0) + flag = LOG_INFO_MISC; + else if (strcmp(token, "skip") == 0) + flag = LOG_INFO_SKIP; + else if (strcmp(token, "stats") == 0) + flag = LOG_INFO_STATS; + else { + log_message(LOG_LEVEL_ERROR, "unsupported --info flag: %s", token); + free(flags); + return -1; + } + parsed |= flag; + } + free(flags); + config->info_level = (int)parsed; + set_log_info_flags(parsed); + return 0; +} + +/* Parse a string as an unsigned long long. Returns 0 on success, -1 on error. */ +static int parse_ull_arg(const char* val, unsigned long long* out, const char* optname) { + char* end; + errno = 0; + unsigned long long v = strtoull(val, &end, 10); + if (errno != 0 || *end != '\0') { + log_message(LOG_LEVEL_ERROR, "%s must be a non-negative integer", optname); + return -1; + } + *out = v; + return 0; +} + +/* Apply a --delta-block/--block-size value (both spellings and both the inline + * and separate argument forms share this one range check). An out-of-range + * value warns once and leaves the configured default untouched. Returns 0 on + * success, -1 on a non-numeric value. */ +static int set_delta_block_size(Config* config, const char* value) { + unsigned long long val; + if (parse_ull_arg(value, &val, "--block-size/--delta-block") != 0) + return -1; + if (val >= DELTA_BLOCK_SIZE_MIN && val <= DELTA_BLOCK_SIZE_MAX) + config->delta_block_size = (uint32_t)val; + else + log_message(LOG_LEVEL_WARNING, "block size value %llu out of range, using default", val); + return 0; +} + +/* Parse a byte count with an optional single-letter binary suffix (K/M/G/T/P/E). + * When allow_zero is false, a bare 0 is rejected (size limits use true, since 0 + * means "no limit"). Returns 0 on success, -1 on error. */ +static int parse_size_arg_allow_zero(const char* value, unsigned long long* out, bool allow_zero) { + if (!value || *value < '0' || *value > '9') + return -1; + char* end; + errno = 0; + unsigned long long number = strtoull(value, &end, 10); + if (errno != 0 || end == value) + return -1; + unsigned long long multiplier = 1; + if (*end != '\0') { + if (end[1] != '\0') + return -1; + switch (*end) { + case 'b': + case 'B': + break; + case 'k': + case 'K': + multiplier = 1024ULL; + break; + case 'm': + case 'M': + multiplier = 1024ULL * 1024; + break; + case 'g': + case 'G': + multiplier = 1024ULL * 1024 * 1024; + break; + case 't': + case 'T': + multiplier = 1024ULL * 1024 * 1024 * 1024; + break; + case 'p': + case 'P': + multiplier = 1024ULL * 1024 * 1024 * 1024 * 1024; + break; + case 'e': + case 'E': + multiplier = 1024ULL * 1024 * 1024 * 1024 * 1024 * 1024; + break; + default: + return -1; + } + } + if ((!allow_zero && number == 0) || number > ULLONG_MAX / multiplier) + return -1; + *out = number * multiplier; + return 0; +} + +static int parse_size_arg(const char* value, unsigned long long* out) { + return parse_size_arg_allow_zero(value, out, false); +} + +/* Append a duplicated pattern to a growable pattern array. Returns 0 on success, -1 on error. */ +static int config_add_pattern(char*** patterns, int* count, const char* value, + const char* optname) { + char** tmp = realloc(*patterns, (*count + 1) * sizeof(char*)); + if (!tmp) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed for %s", optname); + return -1; + } + *patterns = tmp; + char* dup = str_dup(value); + if (!dup) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed for %s", optname); + return -1; + } + (*patterns)[(*count)++] = dup; + return 0; +} + +/* Validate and append one --filter=RULE string. Returns 0 on success, -1 on error. */ +static int config_add_filter(Config* config, const char* rule) { + char err[160]; + FilterRule* parsed = filter_rule_parse(rule, err, sizeof(err)); + if (!parsed) { + log_message(LOG_LEVEL_ERROR, "invalid --filter rule '%s': %s", rule, err); + return -1; + } + filter_rule_free(parsed); + if (!config->filters) { + config->filters = array_list_create(free); + if (!config->filters) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed for --filter"); + return -1; + } + } + char* dup = str_dup(rule); + if (!dup || !array_list_add(config->filters, dup)) { + free(dup); + log_message(LOG_LEVEL_ERROR, "memory allocation failed for --filter"); + return -1; + } + return 0; +} + +static int parse_skip_compress(Config* config, const char* value) { + char* list = str_dup(value); + if (!list) + return -1; + config->skip_compress_set = true; + for (char* token = strtok(list, ","); token; token = strtok(NULL, ",")) { + while (*token == ' ' || *token == '\t') + token++; + size_t len = strlen(token); + while (len > 0 && (token[len - 1] == ' ' || token[len - 1] == '\t')) + token[--len] = '\0'; + if (len == 0) + continue; + if (config_add_pattern(&config->skip_compress_suffixes, &config->skip_compress_count, token, + "--skip-compress") != 0) { + free(list); + return -1; + } + } + free(list); + return 0; +} + +typedef enum { + OPT_FLAG, + OPT_NOOP, + OPT_STRING, + OPT_POS_INT, + OPT_NONNEG_INT, + OPT_ULL, +} OptKind; + +typedef struct { + const char* name; + const char* alias; + OptKind kind; + size_t offset; /* offsetof of the target field in Config, or 0 for OPT_NOOP */ +} OptionEntry; + +/* Options parsed directly into Config, plus compatibility options with no effect. */ +typedef struct { + const char* name; + const char* alias; + size_t offset; /* offsetof of the boolean target field in Config */ +} NegatableOption; + +/* Options that map directly onto a Config field with no side effects. */ +static const OptionEntry OPTION_TABLE[] = { + {"--dry-run", "-n", OPT_FLAG, offsetof(Config, dry_run)}, + {"--remove-source-files", NULL, OPT_FLAG, offsetof(Config, remove_source_files)}, + {"--delete", NULL, OPT_FLAG, offsetof(Config, use_delete)}, + {"--incremental", NULL, OPT_FLAG, offsetof(Config, use_incremental)}, + {"--size-only", NULL, OPT_FLAG, offsetof(Config, size_only)}, + {"--ignore-times", "-I", OPT_FLAG, offsetof(Config, ignore_times)}, + {"--modify-window", "-@", OPT_NONNEG_INT, offsetof(Config, modify_window)}, + {"--delta", NULL, OPT_FLAG, offsetof(Config, use_delta)}, + {"--whole-file", "-W", OPT_FLAG, offsetof(Config, whole_file)}, + {"--fuzzy", "-y", OPT_FLAG, offsetof(Config, fuzzy)}, + {"--save-to-disk", NULL, OPT_FLAG, offsetof(Config, save_to_disk)}, + {"--progress", NULL, OPT_FLAG, offsetof(Config, show_progress)}, + {"--tls", NULL, OPT_FLAG, offsetof(Config, use_tls)}, + {"--backup", NULL, OPT_FLAG, offsetof(Config, backup)}, + {"--stats", NULL, OPT_FLAG, offsetof(Config, stats)}, + {"--human-readable", "-h", OPT_FLAG, offsetof(Config, human_readable)}, + {"--partial", NULL, OPT_FLAG, offsetof(Config, partial)}, + {"--secluded-args", "-s", OPT_NOOP, 0}, + {"--update", "-u", OPT_FLAG, offsetof(Config, update)}, + {"--old-args", NULL, OPT_FLAG, offsetof(Config, old_args)}, + {"--rsh", "-e", OPT_STRING, offsetof(Config, rsh_command)}, + {"--blocking-io", NULL, OPT_FLAG, offsetof(Config, blocking_io)}, + {"--links", "-l", OPT_FLAG, offsetof(Config, follow_symlinks)}, + {"--copy-links", NULL, OPT_FLAG, offsetof(Config, copy_links)}, + {"--safe-links", NULL, OPT_FLAG, offsetof(Config, safe_links)}, + {"--copy-unsafe-links", NULL, OPT_FLAG, offsetof(Config, copy_unsafe_links)}, + {"--copy-dirlinks", "-k", OPT_FLAG, offsetof(Config, copy_dirlinks)}, + {"--keep-dirlinks", "-K", OPT_FLAG, offsetof(Config, keep_dirlinks)}, + {"--munge-links", NULL, OPT_FLAG, offsetof(Config, munge_links)}, + {"--hard-links", "-H", OPT_FLAG, offsetof(Config, preserve_hard_links)}, + {"--sparse", "-S", OPT_FLAG, offsetof(Config, preserve_sparse)}, + {"--inplace", NULL, OPT_FLAG, offsetof(Config, inplace)}, + {"--preallocate", NULL, OPT_FLAG, offsetof(Config, preallocate)}, + {"--append", NULL, OPT_FLAG, offsetof(Config, append)}, + {"--append-verify", NULL, OPT_FLAG, offsetof(Config, append_verify)}, + {"--fsync", NULL, OPT_FLAG, offsetof(Config, use_fsync)}, + {"--checksum", "-c", OPT_FLAG, offsetof(Config, checksum)}, + {"--8-bit-output", "-8", OPT_FLAG, offsetof(Config, eight_bit_output)}, + {"--itemize-changes", "-i", OPT_FLAG, offsetof(Config, itemize_changes)}, + {"--list-only", NULL, OPT_FLAG, offsetof(Config, list_only)}, + {"--out-format", NULL, OPT_STRING, offsetof(Config, out_format)}, + {"--log-file-format", NULL, OPT_STRING, offsetof(Config, log_file_format)}, + {"--existing", NULL, OPT_FLAG, offsetof(Config, existing)}, + {"--ignore-existing", NULL, OPT_FLAG, offsetof(Config, ignore_existing)}, + {"--delay-updates", NULL, OPT_FLAG, offsetof(Config, delay_updates)}, + {"--chmod", NULL, OPT_STRING, offsetof(Config, chmod_spec)}, + {"--dirs", "-d", OPT_FLAG, offsetof(Config, dirs)}, + {"--old-dirs", NULL, OPT_FLAG, offsetof(Config, dirs)}, + {"--old-d", NULL, OPT_FLAG, offsetof(Config, dirs)}, + {"--relative", "-R", OPT_FLAG, offsetof(Config, relative)}, + {"--mkpath", NULL, OPT_FLAG, offsetof(Config, mkpath)}, + /* --password-file: client-only path to a `user:password` secret file used + * to authenticate a daemon (host::module/path) destination. Stored as a + * path; main() reads it (after the destination form is known) and derives + * the wire credentials. Never crosses the wire. */ + {"--password-file", NULL, OPT_STRING, offsetof(Config, password_file)}, + /* --iconv (protocol 2.16.0): convert file-NAME charsets at the wire + * boundary. The CONVERT_SPEC (LOCAL[,REMOTE]) is validated for real iconv + * charsets at startup (client_validation.c) and the full spec rides the + * config frame so the receiver derives the wire charset symmetrically. */ + {"--iconv", NULL, OPT_STRING, offsetof(Config, iconv_spec)}, + /* --protocol=NUM: rsync-compatible flag that forces the wire protocol + * version to the current value. FastSync has exactly one wire format, so + * any value other than PROTOCOL_VERSION is rejected at validation, before + * any network I/O. Client-only: the server does not negotiate, it just + * enforces an exact match. */ + {"--protocol", NULL, OPT_STRING, offsetof(Config, version)}, + /* Phase 6 residual-batch (client-only): --write-batch=FILE runs the normal + * live transfer AND also emits the self-contained batch FILE; + * --only-write-batch=FILE emits FILE only (no destination, no server); + * --read-batch=FILE applies FILE to the destination (no source, no server). + * All three are LOCAL driver flags and never cross the wire. */ + {"--write-batch", NULL, OPT_STRING, offsetof(Config, write_batch)}, + {"--only-write-batch", NULL, OPT_STRING, offsetof(Config, only_write_batch)}, + {"--read-batch", NULL, OPT_STRING, offsetof(Config, read_batch)}, + {"--delete-before", NULL, OPT_FLAG, offsetof(Config, delete_before)}, + {"--delete-during", "--del", OPT_FLAG, offsetof(Config, delete_during)}, + {"--delete-delay", NULL, OPT_FLAG, offsetof(Config, delete_delay)}, + {"--delete-after", NULL, OPT_FLAG, offsetof(Config, delete_after)}, + {"--delete-excluded", NULL, OPT_FLAG, offsetof(Config, delete_excluded)}, + {"--max-delete", NULL, OPT_NONNEG_INT, offsetof(Config, max_delete)}, + {"--ignore-errors", NULL, OPT_FLAG, offsetof(Config, ignore_errors)}, + {"--force", NULL, OPT_FLAG, offsetof(Config, force_delete)}, + {"--prune-empty-dirs", "-m", OPT_FLAG, offsetof(Config, prune_empty_dirs)}, + {"--ignore-missing-args", NULL, OPT_FLAG, offsetof(Config, ignore_missing_args)}, + {"--delete-missing-args", NULL, OPT_FLAG, offsetof(Config, delete_missing_args)}, + + {"--source-dir", NULL, OPT_STRING, offsetof(Config, send_directory)}, + {"--dest-dir", NULL, OPT_STRING, offsetof(Config, receive_root_directory)}, + {"--server-host", NULL, OPT_STRING, offsetof(Config, server_host)}, + {"--cert", NULL, OPT_STRING, offsetof(Config, tls_cert)}, + {"--key", NULL, OPT_STRING, offsetof(Config, tls_key)}, + {"--ca", NULL, OPT_STRING, offsetof(Config, tls_ca)}, + {"--backup-dir", NULL, OPT_STRING, offsetof(Config, backup_dir)}, + {"--fastsync-server-path", NULL, OPT_STRING, offsetof(Config, fastsync_server_path)}, + /* --rsync-path is rsync's spelling for the same "server program path"; it + * is a pure alias for fastsync_server_path (never a distinct field). */ + {"--rsync-path", NULL, OPT_STRING, offsetof(Config, fastsync_server_path)}, + {"--temp-dir", "-T", OPT_STRING, offsetof(Config, temp_dir)}, + {"--partial-dir", NULL, OPT_STRING, offsetof(Config, partial_dir)}, + {"--suffix", NULL, OPT_STRING, offsetof(Config, suffix)}, + {"--compress-choice", "--zc", OPT_STRING, offsetof(Config, compress_choice)}, + {"--compress-level", "--zl", OPT_POS_INT, offsetof(Config, compression_level)}, + + {"--timeout", NULL, OPT_POS_INT, offsetof(Config, timeout)}, + {"--contimeout", NULL, OPT_POS_INT, offsetof(Config, contimeout)}, + {"--max-depth", NULL, OPT_NONNEG_INT, offsetof(Config, max_depth)}, + {"--address", NULL, OPT_STRING, offsetof(Config, address)}, + {"--ipv4", "-4", OPT_FLAG, offsetof(Config, ipv4)}, + {"--ipv6", "-6", OPT_FLAG, offsetof(Config, ipv6)}, + + {"--max-size", NULL, OPT_ULL, offsetof(Config, max_size)}, + {"--min-size", NULL, OPT_ULL, offsetof(Config, min_size)}, + {"--one-file-system", "-x", OPT_FLAG, offsetof(Config, one_file_system)}, + {"--from0", "-0", OPT_FLAG, offsetof(Config, from0)}, + {"--cvs-exclude", "-C", OPT_FLAG, offsetof(Config, cvs_exclude)}, + {"-F", NULL, OPT_FLAG, offsetof(Config, per_dir_filter)}, + {"--numeric-ids", NULL, OPT_FLAG, offsetof(Config, numeric_ids)}, + {"--atimes", "-U", OPT_FLAG, offsetof(Config, preserve_atimes)}, + {"--crtimes", "-N", OPT_FLAG, offsetof(Config, preserve_crtimes)}, + /* -D is handled separately (it implies both --devices and --specials). */ + {"--devices", NULL, OPT_FLAG, offsetof(Config, preserve_devices)}, + {"--specials", NULL, OPT_FLAG, offsetof(Config, preserve_specials)}, + {"--copy-devices", NULL, OPT_FLAG, offsetof(Config, copy_devices)}, + {"--write-devices", NULL, OPT_FLAG, offsetof(Config, write_devices)}, + {"--omit-dir-times", "-O", OPT_FLAG, offsetof(Config, omit_dir_times)}, + {"--omit-link-times", "-J", OPT_FLAG, offsetof(Config, omit_link_times)}, + {"--open-noatime", NULL, OPT_FLAG, offsetof(Config, open_noatime)}, + {"--xattrs", "-X", OPT_FLAG, offsetof(Config, preserve_xattrs)}, + {"--acls", "-A", OPT_FLAG, offsetof(Config, preserve_acls)}, + {"--fake-super", NULL, OPT_FLAG, offsetof(Config, fake_super)}, + /* rsync's -M/--remote-option: -M is now the short alias for --remote-option + * (metadata mode is long-only --preserve), handled in the parse loop where + * --remote-option is parsed. --trust-sender is a local receiver policy and + * never travels to the remote peer. */ + {"--trust-sender", NULL, OPT_FLAG, offsetof(Config, trust_sender)}, +}; + +/* Only boolean options with no required argument are safe to negate. */ +static const NegatableOption NEGATABLE_OPTIONS[] = { + {"dry-run", "n", offsetof(Config, dry_run)}, + {"delete", NULL, offsetof(Config, use_delete)}, + {"incremental", NULL, offsetof(Config, use_incremental)}, + {"delta", NULL, offsetof(Config, use_delta)}, + {"fuzzy", NULL, offsetof(Config, fuzzy)}, + {"save-to-disk", NULL, offsetof(Config, save_to_disk)}, + {"progress", NULL, offsetof(Config, show_progress)}, + {"tls", NULL, offsetof(Config, use_tls)}, + {"backup", NULL, offsetof(Config, backup)}, + {"stats", NULL, offsetof(Config, stats)}, + {"partial", NULL, offsetof(Config, partial)}, + {"links", "l", offsetof(Config, follow_symlinks)}, + {"copy-links", NULL, offsetof(Config, copy_links)}, + {"safe-links", NULL, offsetof(Config, safe_links)}, + {"copy-unsafe-links", NULL, offsetof(Config, copy_unsafe_links)}, + {"hard-links", "H", offsetof(Config, preserve_hard_links)}, + {"sparse", "S", offsetof(Config, preserve_sparse)}, + {"inplace", NULL, offsetof(Config, inplace)}, + {"preallocate", NULL, offsetof(Config, preallocate)}, + {"checksum", "c", offsetof(Config, checksum)}, + {"from0", NULL, offsetof(Config, from0)}, + {"cvs-exclude", NULL, offsetof(Config, cvs_exclude)}, + + /* These options are also implied by --archive or handled outside the table. */ + {"compress", NULL, offsetof(Config, use_compression)}, + {"compress", "z", offsetof(Config, use_compression)}, + {"multithreading", "j", offsetof(Config, use_multithreading)}, + {"preserve", NULL, offsetof(Config, use_metadata)}, + {"sendfile", NULL, offsetof(Config, use_sendfile)}, + {"chunk-serialization", NULL, offsetof(Config, use_chunk_serialization)}, + {"xattrs", "X", offsetof(Config, preserve_xattrs)}, + {"acls", "A", offsetof(Config, preserve_acls)}, + {"fake-super", NULL, offsetof(Config, fake_super)}, +}; + +static bool opt_is(const char* arg, const char* name, const char* alias) { + return strcmp(arg, name) == 0 || (alias && strcmp(arg, alias) == 0); +} + +static const OptionEntry* find_table_option(const char* arg) { + for (size_t i = 0; i < sizeof(OPTION_TABLE) / sizeof(OPTION_TABLE[0]); i++) + if (opt_is(arg, OPTION_TABLE[i].name, OPTION_TABLE[i].alias)) + return &OPTION_TABLE[i]; + return NULL; +} + +/* Match a "--opt=value" argument against table options that take a value. Flags, + * no-ops, and unsupported options do not accept an inline "=" value. */ +static const OptionEntry* find_table_option_with_equals(const char* arg, const char** value) { + const char* equals = strchr(arg, '='); + if (!equals || equals == arg) + return NULL; + size_t name_len = (size_t)(equals - arg); + for (size_t i = 0; i < sizeof(OPTION_TABLE) / sizeof(OPTION_TABLE[0]); i++) { + const OptionEntry* entry = &OPTION_TABLE[i]; + if ((strlen(entry->name) == name_len && strncmp(arg, entry->name, name_len) == 0) || + (entry->alias && strlen(entry->alias) == name_len && + strncmp(arg, entry->alias, name_len) == 0)) { + if (entry->kind == OPT_STRING || entry->kind == OPT_POS_INT || + entry->kind == OPT_NONNEG_INT || entry->kind == OPT_ULL) { + *value = equals + 1; + return entry; + } + } + } + return NULL; +} + +static const NegatableOption* find_negatable_option(const char* name) { + for (size_t i = 0; i < sizeof(NEGATABLE_OPTIONS) / sizeof(NEGATABLE_OPTIONS[0]); i++) + if (strcmp(name, NEGATABLE_OPTIONS[i].name) == 0 || + (NEGATABLE_OPTIONS[i].alias && strcmp(name, NEGATABLE_OPTIONS[i].alias) == 0)) + return &NEGATABLE_OPTIONS[i]; + return NULL; +} + +static int apply_negation(Config* config, const char* arg) { + const char* name = arg + strlen("--no-"); + if (*name == '\0') { + fprintf(stderr, "Cannot negate an empty option name: %s\n", arg); + return -1; + } + const NegatableOption* entry = find_negatable_option(name); + if (!entry) { + fprintf(stderr, "Cannot negate unsupported or unsafe option: %s\n", arg); + return -1; + } + *(bool*)((char*)config + entry->offset) = false; + if (entry->offset == offsetof(Config, use_metadata)) + config->metadata_explicitly_disabled = true; + return 0; +} + +static int apply_table_option(Config* config, const OptionEntry* entry, const char* value) { + if (entry->kind == OPT_NOOP) + return 0; + void* field = (char*)config + entry->offset; + switch (entry->kind) { + case OPT_FLAG: + *(bool*)field = true; + if (entry->offset == offsetof(Config, update)) + config->use_metadata = true; + return 0; + case OPT_NOOP: + return 0; + case OPT_STRING: + return set_string_option((char**)field, value, entry->name); + case OPT_POS_INT: + return set_positive_int_option((int*)field, value, entry->name); + case OPT_NONNEG_INT: + return set_nonneg_int_option((int*)field, value, entry->name); + case OPT_ULL: { + unsigned long long v; + /* Size-limit options accept rsync-style suffixes (e.g. --max-size=2G); a + * plain byte count, including 0 ("no limit"), stays valid. */ + if (parse_size_arg_allow_zero(value, &v, true) != 0) { + log_message(LOG_LEVEL_ERROR, "%s must be a non-negative size (B, K, M, G, T, P, or E)", + entry->name); + return -1; + } + *(unsigned long long*)field = v; + return 0; + } + } + return -1; +} + /* Parse CLI arguments into config. Returns 0 on success, -1 on error, 1 for help/clean-exit. */ -static int parse_args(Config* config, int argc, char* argv[], int* positional_args, - int* positional_count) { +int parse_args(Config* config, int argc, char* argv[], int* positional_args, + int* positional_count) { + bool verbose = false; + /* Explicit --no-delta / --no-incremental seen on the command line: the user + switched part of the delta machinery off, so the --fuzzy implication must + not silently turn it back on. */ + bool no_delta = false; + bool no_incremental = false; + protocol_set_8_bit_output(config->eight_bit_output); + + /* Apply output controls before processing other options so their order is irrelevant. */ for (int i = 1; i < argc; i++) { - if (strcmp(argv[i], "--help") == 0) { + if (strcmp(argv[i], "-v") == 0 || strcmp(argv[i], "--verbose") == 0) { + set_log_level(LOG_LEVEL_DEBUG); + } else if (strncmp(argv[i], "--info=", 7) == 0) { + if (parse_info_flags(argv[i] + 7, config) != 0) + return -1; + } else if (strcmp(argv[i], "--info") == 0) { + if (i + 1 >= argc || parse_info_flags(argv[++i], config) != 0) + return -1; + } + } + + for (int i = 1; i < argc; i++) { + if (strcmp(argv[i], "-P") == 0) { + config->partial = true; + config->show_progress = true; + continue; + } + /* "--no-implied-dirs" is a real rsync option name, not a negation of + * "--implied-dirs", so it must be handled before the generic --no-* + * negation branch. */ + if (strcmp(argv[i], "--no-implied-dirs") == 0) { + config->no_implied_dirs = true; + continue; + } + /* "--no-motd" is a real rsync option name (client-side daemon MOTD display + * suppression), not a negation of a "--motd" flag, so it is handled before + * the generic --no-* negation branch. */ + if (strcmp(argv[i], "--no-motd") == 0) { + config->no_motd = true; + continue; + } + /* "--super" / "--no-super" are real rsync option names controlling the + * receiver's super-user activity policy (ownership, device nodes), not a + * Boolean pair for the generic --no-* negation branch: both map onto the + * Config->super_mode tri-state. Handle them explicitly (exact match only, + * so a malformed "--super=x" still falls through to the unknown-option + * error) before the generic negation branch would mis-reject "--no-super". */ + if (strcmp(argv[i], "--super") == 0) { + config->super_mode = SUPER_MODE_ON; + continue; + } + if (strcmp(argv[i], "--no-super") == 0) { + config->super_mode = SUPER_MODE_OFF; + continue; + } + if (strncmp(argv[i], "--no-", strlen("--no-")) == 0) { + if (strcmp(argv[i], "--no-delta") == 0) + no_delta = true; + else if (strcmp(argv[i], "--no-incremental") == 0) + no_incremental = true; + if (apply_negation(config, argv[i]) != 0) + return -1; + continue; + } + const char* modify_window_prefix = "--modify-window="; + if (strncmp(argv[i], modify_window_prefix, strlen(modify_window_prefix)) == 0) { + if (set_nonneg_int_option(&config->modify_window, argv[i] + strlen(modify_window_prefix), + "--modify-window") != 0) + return -1; + continue; + } + if (strncmp(argv[i], "-@", 2) == 0 && argv[i][2] != '\0') { + if (set_nonneg_int_option(&config->modify_window, argv[i] + 2, "-@") != 0) + return -1; + continue; + } + /* --stop-after/--stop-at are client-only sender-side stop deadlines. They + * are parsed by stop_condition (so the unit tests exercise the same validate + * that production uses) and never serialized into the config frame. */ + if (strncmp(argv[i], "--stop-after=", 13) == 0) { + if (!stop_parse_after_minutes(argv[i] + 13, &config->stop_after_mins)) { + log_message(LOG_LEVEL_ERROR, "--stop-after must be a positive number of minutes"); + return -1; + } + continue; + } + if (strcmp(argv[i], "--stop-after") == 0) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for --stop-after"); + return -1; + } + if (!stop_parse_after_minutes(argv[++i], &config->stop_after_mins)) { + log_message(LOG_LEVEL_ERROR, "--stop-after must be a positive number of minutes"); + return -1; + } + continue; + } + if (strncmp(argv[i], "--stop-at=", 10) == 0) { + if (!stop_parse_at_time(argv[i] + 10, time(NULL), &config->stop_at)) { + log_message(LOG_LEVEL_ERROR, "--stop-at must be HH:MM[:SS] or now+N[smhd]"); + return -1; + } + config->stop_at_set = true; + continue; + } + if (strcmp(argv[i], "--stop-at") == 0) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for --stop-at"); + return -1; + } + if (!stop_parse_at_time(argv[++i], time(NULL), &config->stop_at)) { + log_message(LOG_LEVEL_ERROR, "--stop-at must be HH:MM[:SS] or now+N[smhd]"); + return -1; + } + config->stop_at_set = true; + continue; + } + const char* threads_prefix = "--compress-threads="; + if (strncmp(argv[i], threads_prefix, strlen(threads_prefix)) == 0) { + if (set_compression_threads_option(&config->compression_threads, + argv[i] + strlen(threads_prefix)) != 0) + return -1; + continue; + } + if (strncmp(argv[i], "--max-alloc=", 12) == 0 || strcmp(argv[i], "--max-alloc") == 0) { + const char* value = strcmp(argv[i], "--max-alloc") == 0 ? "" : argv[i] + 12; + if (*value == '\0') { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for --max-alloc"); + return -1; + } + value = argv[++i]; + } + if (parse_size_arg(value, &config->max_alloc) != 0) { + log_message(LOG_LEVEL_ERROR, + "--max-alloc must be a positive size (B, K, M, G, T, P, or E)"); + return -1; + } + continue; + } + + const OptionEntry* entry = find_table_option(argv[i]); + const char* inline_value = NULL; + if (!entry) + entry = find_table_option_with_equals(argv[i], &inline_value); + if (entry) { + const char* value = NULL; + if (entry->kind != OPT_FLAG) { + value = inline_value; + if (!value && i + 1 < argc) + value = argv[++i]; + if (!value) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", entry->name); + return -1; + } + if (strcmp(entry->name, "--compress-choice") == 0) { + if (set_compression_choice(config, value) != 0) + return -1; + } else { + if (apply_table_option(config, entry, value) != 0) + return -1; + if (strcmp(entry->name, "--compress-level") == 0 && + (config->compression_level < 1 || config->compression_level > 22)) { + log_message(LOG_LEVEL_ERROR, "--compress-level must be between 1 and 22"); + return -1; + } + if (entry->offset == offsetof(Config, chmod_spec)) { + mode_t ignored; + if (!chmod_apply(0, config->chmod_spec, &ignored)) { + log_message(LOG_LEVEL_ERROR, "--chmod has invalid permission changes"); + return -1; + } + config->use_metadata = true; + } + } + } else if (apply_table_option(config, entry, NULL) != 0) { + return -1; + } + if (entry->offset == offsetof(Config, eight_bit_output)) + protocol_set_8_bit_output(true); + /* A delete-timing flag selects when --delete removes extras, so it + implies --delete exactly like the rsync options do. */ + if (entry->offset == offsetof(Config, delete_before) || + entry->offset == offsetof(Config, delete_during) || + entry->offset == offsetof(Config, delete_delay) || + entry->offset == offsetof(Config, delete_after)) + config->use_delete = true; + /* --delete-missing-args implies --ignore-missing-args (missing entries + are skipped for deletion instead of failing the run). The implication + is order-independent because it is applied over the final parsed + config. */ + if (entry->offset == offsetof(Config, delete_missing_args)) + config->ignore_missing_args = true; + /* -U/--atimes and -N/--crtimes carry their times inside the metadata + payload, which is only transmitted when use_metadata is set, so either + one implies metadata transmission. This is FastSync's broad -M bundle + (mode/mtime travel too); it does NOT enable ownership application, + which stays opt-in via the identity flags. */ + if (entry->offset == offsetof(Config, preserve_atimes) || + entry->offset == offsetof(Config, preserve_crtimes)) + config->use_metadata = true; + if (entry->offset == offsetof(Config, preserve_xattrs) || + entry->offset == offsetof(Config, preserve_acls)) { + config->use_metadata = true; + config->use_xattrs = config->preserve_acls || config->preserve_xattrs; + } + if (entry->offset == offsetof(Config, fake_super)) + config->use_metadata = true; + continue; + } + + if (strncmp(argv[i], "--chmod=", 8) == 0) { + if (set_string_option(&config->chmod_spec, argv[i] + 8, "--chmod") != 0) + return -1; + mode_t ignored; + if (!chmod_apply(0, config->chmod_spec, &ignored)) { + log_message(LOG_LEVEL_ERROR, "--chmod has invalid permission changes"); + return -1; + } + config->use_metadata = true; + continue; + } + + if (opt_is(argv[i], "--help", NULL)) { print_usage(); return 1; - } else if (strcmp(argv[i], "-a") == 0 || strcmp(argv[i], "--archive") == 0) { - config->use_compression = true; - config->use_multithreading = true; + } else if (opt_is(argv[i], "-V", "--version")) { + printf("fastsync version %s\n", PROTOCOL_VERSION); + return 1; + } else if (opt_is(argv[i], "-D", NULL)) { + /* rsync -D == --devices --specials. -D is otherwise unassigned in + FastSync (verified: no collision), so it is free to imply both. */ + config->preserve_devices = true; + config->preserve_specials = true; + log_info_message(LOG_INFO_MISC, "Enabled preservation of device and special files (-D)"); + } else if (opt_is(argv[i], "-a", "--archive")) { + /* Real rsync archive (-rlptgoD). FastSync is always recursive and always + * preserves hard-link/other transfer semantics per its own flags, so -a + * implies links, full metadata (perms/times/group/owner as FastSync's + * broad bundle), devices and specials. Compression and multithreading + * are NOT implied (they are no longer part of archive mode). */ + config->follow_symlinks = true; config->use_metadata = true; - log_message(LOG_LEVEL_INFO, "Enabled archive mode (-c -m -M)"); - } else if (strcmp(argv[i], "-n") == 0 || strcmp(argv[i], "--dry-run") == 0) { - config->dry_run = true; - } else if (strcmp(argv[i], "-p") == 0 && i + 1 < argc) { - if (!parse_positive_int(argv[++i], &config->ssh_port)) { - fprintf(stderr, "Error: invalid --port/-p value: %s\n", argv[i]); + config->preserve_devices = true; + config->preserve_specials = true; + log_info_message(LOG_INFO_MISC, + "Enabled archive mode (-rlptgoD: links, metadata, devices, specials)"); + } else if (opt_is(argv[i], "-p", "--perms")) { + /* rsync -p/--perms: preserve permission bits. Folded into FastSync's + * broad metadata bundle (mode/mtime travel together). */ + config->use_metadata = true; + log_info_message(LOG_INFO_MISC, "Enabled permission preservation"); + } else if (opt_is(argv[i], "--ssh-port", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); return -1; } - } else if (strcmp(argv[i], "--delete") == 0) { - config->use_delete = true; - } else if (strcmp(argv[i], "--exclude") == 0 && i + 1 < argc) { - char** tmp = realloc(config->exclude_patterns, (config->exclude_count + 1) * sizeof(char*)); - if (!tmp) { - fprintf(stderr, "Error: memory allocation failed for --exclude\n"); + if (set_positive_int_option(&config->ssh_port, argv[++i], "--ssh-port") != 0) + return -1; + if (config->ssh_port > 65535) { + log_message(LOG_LEVEL_ERROR, "SSH port must be 1-65535"); return -1; } - config->exclude_patterns = tmp; - char* dup = str_dup(argv[++i]); - if (!dup) { - fprintf(stderr, "Error: memory allocation failed for --exclude\n"); + } else if (strncmp(argv[i], "--ssh-port=", 11) == 0) { + if (set_positive_int_option(&config->ssh_port, argv[i] + 11, "--ssh-port") != 0) + return -1; + if (config->ssh_port > 65535) { + log_message(LOG_LEVEL_ERROR, "SSH port must be 1-65535"); return -1; } - config->exclude_patterns[config->exclude_count++] = dup; - } else if (strcmp(argv[i], "--include") == 0 && i + 1 < argc) { - char** tmp = realloc(config->include_patterns, (config->include_count + 1) * sizeof(char*)); - if (!tmp) { - fprintf(stderr, "Error: memory allocation failed for --include\n"); + } else if (opt_is(argv[i], "--exclude", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); return -1; } - config->include_patterns = tmp; - char* dup = str_dup(argv[++i]); - if (!dup) { - fprintf(stderr, "Error: memory allocation failed for --include\n"); + if (config_add_pattern(&config->exclude_patterns, &config->exclude_count, argv[++i], + "--exclude") != 0) + return -1; + } else if (opt_is(argv[i], "--include", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); return -1; } - config->include_patterns[config->include_count++] = dup; - } else if (strcmp(argv[i], "--max-size") == 0 && i + 1 < argc) { - config->max_size = strtoull(argv[++i], NULL, 10); - } else if (strcmp(argv[i], "--min-size") == 0 && i + 1 < argc) { - config->min_size = strtoull(argv[++i], NULL, 10); - } else if (strcmp(argv[i], "--incremental") == 0) { - config->use_incremental = true; - } else if (strcmp(argv[i], "--delta") == 0) { - config->use_delta = true; - } else if (strcmp(argv[i], "--delta-block") == 0 && i + 1 < argc) { - unsigned long long val = strtoull(argv[++i], NULL, 10); - if (val >= DELTA_BLOCK_SIZE_MIN && val <= DELTA_BLOCK_SIZE_MAX) - config->delta_block_size = (uint32_t)val; - else - fprintf(stderr, "Warning: --delta-block value %llu out of range, using default\n", val); - } else if (strcmp(argv[i], "--delta-max") == 0 && i + 1 < argc) { - unsigned long long val = strtoull(argv[++i], NULL, 10); + if (config_add_pattern(&config->include_patterns, &config->include_count, argv[++i], + "--include") != 0) + return -1; + } else if (strncmp(argv[i], "--delta-block=", 14) == 0) { + if (set_delta_block_size(config, argv[i] + 14) != 0) + return -1; + } else if (strncmp(argv[i], "--block-size=", 13) == 0) { + if (set_delta_block_size(config, argv[i] + 13) != 0) + return -1; + } else if (opt_is(argv[i], "--delta-block", "--block-size")) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); + return -1; + } + if (set_delta_block_size(config, argv[++i]) != 0) + return -1; + } else if (opt_is(argv[i], "--delta-max", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); + return -1; + } + unsigned long long val; + if (parse_ull_arg(argv[++i], &val, "--delta-max") != 0) + return -1; if (val >= DELTA_MIN_FILE_SIZE) config->delta_max_file_size = val; else - fprintf(stderr, "Warning: --delta-max value %llu too small, using default\n", val); - } else if (strcmp(argv[i], "-c") == 0 || strcmp(argv[i], "-z") == 0) { - config->use_compression = true; - log_message(LOG_LEVEL_INFO, "Enabled Compression"); + log_message(LOG_LEVEL_WARNING, "--delta-max value %llu too small, using default", val); + } else if (opt_is(argv[i], "-z", "--compress")) { + config->use_compression = + !config->compress_choice || strcmp(config->compress_choice, "zstd") == 0; + log_info_message(LOG_INFO_MISC, "Enabled Compression"); if (i + 1 < argc) { char* end_ptr; long level = strtol(argv[i + 1], &end_ptr, 10); if (*end_ptr == '\0') { + if (level < 1 || level > 22) { + log_message(LOG_LEVEL_ERROR, "compression level must be 1-22"); + return -1; + } config->compression_level = (int)level; - log_message(LOG_LEVEL_INFO, "Set Compression level to %ld", level); + log_info_message(LOG_INFO_MISC, "Set Compression level to %ld", level); i++; } } - } else if (strcmp(argv[i], "--source-dir") == 0 && i + 1 < argc) { - char* dup = str_dup(argv[++i]); - if (!dup) { - fprintf(stderr, "Error: memory allocation failed for --source-dir\n"); - return -1; - } - free(config->send_directory); - config->send_directory = dup; - } else if (strcmp(argv[i], "--dest-dir") == 0 && i + 1 < argc) { - char* dup = str_dup(argv[++i]); - if (!dup) { - fprintf(stderr, "Error: memory allocation failed for --dest-dir\n"); - return -1; - } - free(config->receive_root_directory); - config->receive_root_directory = dup; - } else if (strcmp(argv[i], "--save-to-disk") == 0) { - config->save_to_disk = true; - } else if (strcmp(argv[i], "-M") == 0 || strcmp(argv[i], "--preserve") == 0) { + } else if (opt_is(argv[i], "--preserve", NULL)) { config->use_metadata = true; - log_message(LOG_LEVEL_INFO, "Enabled metadata preservation"); - } else if (strcmp(argv[i], "-f") == 0 || strcmp(argv[i], "--sendfile") == 0) { + log_info_message(LOG_INFO_MISC, "Enabled metadata preservation"); + } else if (opt_is(argv[i], "-E", "--executability")) { + config->use_metadata = true; + config->use_executability = true; + log_info_message(LOG_INFO_MISC, "Enabled executable permission preservation"); + } else if (opt_is(argv[i], "--sendfile", NULL)) { config->use_sendfile = true; - log_message(LOG_LEVEL_INFO, "Enabled sendfile"); - } else if (strcmp(argv[i], "-m") == 0) { + log_info_message(LOG_INFO_MISC, "Enabled sendfile"); + } else if (opt_is(argv[i], "-j", "--threads")) { config->use_multithreading = true; - log_message(LOG_LEVEL_INFO, "Enabled Multithreading"); - } else if (strcmp(argv[i], "-s") == 0) { + log_info_message(LOG_INFO_MISC, "Enabled Multithreading"); + } else if (opt_is(argv[i], "--chunk-serialization", NULL)) { config->use_chunk_serialization = true; - log_message(LOG_LEVEL_INFO, "Enabled Chunk Serialization"); - } else if (strcmp(argv[i], "--server-host") == 0 && i + 1 < argc) { - char* dup = str_dup(argv[++i]); - if (!dup) { - fprintf(stderr, "Error: memory allocation failed for --server-host\n"); + log_info_message(LOG_INFO_MISC, "Enabled Chunk Serialization"); + } else if (opt_is(argv[i], "--server-port", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); return -1; } - free(config->server_host); - config->server_host = dup; - } else if (strcmp(argv[i], "--server-port") == 0 && i + 1 < argc) { if (!parse_positive_int(argv[++i], &config->server_port)) { - fprintf(stderr, "Error: invalid --server-port value: %s\n", argv[i]); + char* escaped = output_escape(argv[i], false); + log_message(LOG_LEVEL_ERROR, "invalid --server-port value: %s", + escaped ? escaped : ""); + free(escaped); return -1; } - } else if (strcmp(argv[i], "--bwlimit") == 0 && i + 1 < argc) { - char* end; - errno = 0; - unsigned long long kbps = strtoull(argv[++i], &end, 10); - if (errno != 0 || *end != '\0' || kbps == 0) { - fprintf(stderr, "Error: --bwlimit must be a positive integer\n"); + if (config->server_port > 65535) { + log_message(LOG_LEVEL_ERROR, "server port must be 1-65535"); + return -1; + } + } else if (opt_is(argv[i], "--bwlimit", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); + return -1; + } + unsigned long long kbps; + if (parse_ull_arg(argv[++i], &kbps, "--bwlimit") != 0) + return -1; + if (kbps == 0) { + log_message(LOG_LEVEL_ERROR, "--bwlimit must be a positive integer"); return -1; } if (kbps > ULLONG_MAX / 1024) { - fprintf(stderr, "Error: --bwlimit value too large\n"); + log_message(LOG_LEVEL_ERROR, "--bwlimit value too large"); return -1; } io_set_bwlimit(kbps * 1024); - log_message(LOG_LEVEL_INFO, "Set bandwidth limit to %llu KB/s", kbps); - } else if (strcmp(argv[i], "--progress") == 0) { - config->show_progress = true; - } else if (strcmp(argv[i], "--chunk-size") == 0 && i + 1 < argc) { - unsigned long long val = strtoull(argv[++i], NULL, 10); - if (val > 0) - config->chunk_size = val; - } else if (strcmp(argv[i], "--tls") == 0) { - config->use_tls = true; - } else if (strcmp(argv[i], "--cert") == 0 && i + 1 < argc) { - char* dup = str_dup(argv[++i]); - if (!dup) { - fprintf(stderr, "Error: memory allocation failed for --cert\n"); + log_info_message(LOG_INFO_MISC, "Set bandwidth limit to %llu KB/s", kbps); + } else if (opt_is(argv[i], "--chunk-size", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); return -1; } - free(config->tls_cert); - config->tls_cert = dup; - } else if (strcmp(argv[i], "--key") == 0 && i + 1 < argc) { - char* dup = str_dup(argv[++i]); - if (!dup) { - fprintf(stderr, "Error: memory allocation failed for --key\n"); + unsigned long long val; + if (parse_ull_arg(argv[++i], &val, "--chunk-size") != 0) + return -1; + if (val == 0) { + log_message(LOG_LEVEL_ERROR, "--chunk-size must be a positive integer"); return -1; } - free(config->tls_key); - config->tls_key = dup; - } else if (strcmp(argv[i], "--ca") == 0 && i + 1 < argc) { - char* dup = str_dup(argv[++i]); - if (!dup) { - fprintf(stderr, "Error: memory allocation failed for --ca\n"); + config->chunk_size = val; + } else if (opt_is(argv[i], "--log-file", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); return -1; } - free(config->tls_ca); - config->tls_ca = dup; - } else if (strcmp(argv[i], "--timeout") == 0 && i + 1 < argc) { - int val; - if (!parse_positive_int(argv[++i], &val)) { - fprintf(stderr, "Error: --timeout must be a positive integer\n"); - return -1; + if (config->log_file) { + fclose(config->log_file); + config->log_file = NULL; + log_set_file(NULL); } - config->timeout = val; - } else if (strcmp(argv[i], "--contimeout") == 0 && i + 1 < argc) { - int val; - if (!parse_positive_int(argv[++i], &val)) { - fprintf(stderr, "Error: --contimeout must be a positive integer\n"); - return -1; - } - config->contimeout = val; - } else if (strcmp(argv[i], "-q") == 0 || strcmp(argv[i], "--quiet") == 0 || - strcmp(argv[i], "--silent") == 0) { - config->quiet = true; - } else if (strcmp(argv[i], "--backup") == 0) { - config->backup = true; - } else if (strcmp(argv[i], "--backup-dir") == 0 && i + 1 < argc) { - char* dup = str_dup(argv[++i]); - if (!dup) { - fprintf(stderr, "Error: memory allocation failed for --backup-dir\n"); - return -1; - } - free(config->backup_dir); - config->backup_dir = dup; - } else if (strcmp(argv[i], "--stats") == 0) { - config->stats = true; - } else if (strcmp(argv[i], "--max-depth") == 0 && i + 1 < argc) { - if (!parse_nonneg_int(argv[++i], &config->max_depth)) { - fprintf(stderr, "Error: --max-depth must be a non-negative integer\n"); - return -1; - } - } else if (strcmp(argv[i], "--log-file") == 0 && i + 1 < argc) { FILE* lf = fopen(argv[++i], "a"); if (!lf) { - fprintf(stderr, "Error: could not open log file '%s': %s\n", argv[i], strerror(errno)); + char* escaped = output_escape(argv[i], false); + log_message(LOG_LEVEL_ERROR, "could not open log file '%s': %s", + escaped ? escaped : "", strerror(errno)); + free(escaped); return -1; } config->log_file = lf; log_set_file(lf); - } else if (strcmp(argv[i], "--queue-size") == 0 && i + 1 < argc) { - int val; - if (!parse_positive_int(argv[++i], &val)) { - fprintf(stderr, "Error: --queue-size must be a positive integer\n"); + } else if (strncmp(argv[i], "--stderr=", 9) == 0) { + if (set_stderr_mode(argv[i] + 9) != 0) + return -1; + } else if (opt_is(argv[i], "--stderr", NULL)) { + if (i + 1 >= argc || set_stderr_mode(argv[++i]) != 0) + return -1; + } else if (opt_is(argv[i], "--exclude-from", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); return -1; } - config->queue_size = val; - } else if (strcmp(argv[i], "--exclude-from") == 0 && i + 1 < argc) { if (read_patterns_from_file(argv[++i], &config->exclude_patterns, &config->exclude_count) != 0) return -1; - } else if (strcmp(argv[i], "--include-from") == 0 && i + 1 < argc) { + } else if (opt_is(argv[i], "--include-from", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); + return -1; + } if (read_patterns_from_file(argv[++i], &config->include_patterns, &config->include_count) != 0) return -1; - } else if (strcmp(argv[i], "--partial") == 0) { - config->partial = true; - } else if (strcmp(argv[i], "--fastsync-server-path") == 0 && i + 1 < argc) { - char* dup = str_dup(argv[++i]); - if (!dup) { - fprintf(stderr, "Error: memory allocation failed for --fastsync-server-path\n"); + } else if (strncmp(argv[i], "--filter=", 9) == 0) { + if (config_add_filter(config, argv[i] + 9) != 0) + return -1; + } else if (strncmp(argv[i], "-f=", 3) == 0) { + if (config_add_filter(config, argv[i] + 3) != 0) + return -1; + } else if (opt_is(argv[i], "--filter", "-f")) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); return -1; } - free(config->fastsync_server_path); - config->fastsync_server_path = dup; - } else if (strcmp(argv[i], "-v") == 0 || strcmp(argv[i], "--verbose") == 0) { + if (config_add_filter(config, argv[++i]) != 0) + return -1; + } else if (strncmp(argv[i], "--files-from=", 13) == 0) { + if (set_string_option(&config->files_from, argv[i] + 13, "--files-from") != 0) + return -1; + } else if (opt_is(argv[i], "--files-from", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); + return -1; + } + if (set_string_option(&config->files_from, argv[++i], "--files-from") != 0) + return -1; + } else if (opt_is(argv[i], "-v", "--verbose")) { + verbose = true; set_log_level(LOG_LEVEL_DEBUG); - } else if (strcmp(argv[i], "-l") == 0 || strcmp(argv[i], "--links") == 0) { - config->follow_symlinks = true; - } else if (strcmp(argv[i], "--copy-links") == 0) { - config->copy_links = true; - } else if (strcmp(argv[i], "--safe-links") == 0) { - config->safe_links = true; - } else if (strcmp(argv[i], "--copy-unsafe-links") == 0) { - config->copy_unsafe_links = true; - } else if (strcmp(argv[i], "-H") == 0 || strcmp(argv[i], "--hard-links") == 0) { - config->preserve_hard_links = true; - } else if (strcmp(argv[i], "-A") == 0 || strcmp(argv[i], "--acls") == 0) { - config->preserve_acls = true; - } else if (strcmp(argv[i], "-X") == 0 || strcmp(argv[i], "--xattrs") == 0) { - config->preserve_xattrs = true; - } else if (strcmp(argv[i], "-D") == 0 || strcmp(argv[i], "--devices") == 0) { - config->preserve_devices = true; - } else if (strcmp(argv[i], "-S") == 0 || strcmp(argv[i], "--sparse") == 0) { - config->preserve_sparse = true; - } else if (strcmp(argv[i], "-i") == 0 || strcmp(argv[i], "--itemize-changes") == 0) { - config->itemize_changes = true; - } else if (strcmp(argv[i], "--out-format") == 0 && i + 1 < argc) { - char* dup = str_dup(argv[++i]); - if (!dup) + } else if (opt_is(argv[i], "-q", "--quiet")) { + config->quiet = true; + } else if (strncmp(argv[i], "--debug=", 8) == 0) { + int debug_ret = parse_debug_flags(argv[i] + 8, config); + if (debug_ret != 0) + return debug_ret; + } else if (opt_is(argv[i], "--debug", NULL)) { + if (i + 1 >= argc) + return parse_debug_flags(NULL, config); + int debug_ret = parse_debug_flags(argv[++i], config); + if (debug_ret != 0) + return debug_ret; + } else if (strncmp(argv[i], "--info=", 7) == 0) { + if (parse_info_flags(argv[i] + 7, config) != 0) return -1; - free(config->out_format); - config->out_format = dup; - } else if (strcmp(argv[i], "--info") == 0 && i + 1 < argc) { - config->info_level = atoi(argv[++i]); - } else if (strcmp(argv[i], "--debug") == 0 && i + 1 < argc) { - config->debug_level = atoi(argv[++i]); - } else if (strcmp(argv[i], "--list-only") == 0) { - config->list_only = true; - } else if (strcmp(argv[i], "-h") == 0 || strcmp(argv[i], "--human-readable") == 0) { - config->human_readable = true; - } else if (strcmp(argv[i], "-u") == 0 || strcmp(argv[i], "--update") == 0) { - config->update = true; - } else if (strcmp(argv[i], "--inplace") == 0) { - config->inplace = true; - } else if (strcmp(argv[i], "--append") == 0) { - config->append = true; - } else if (strcmp(argv[i], "--append-verify") == 0) { - config->append_verify = true; - } else if (strcmp(argv[i], "--delete-excluded") == 0) { - config->delete_excluded = true; - } else if (strcmp(argv[i], "--delete-after") == 0) { - config->delete_after = true; - } else if (strcmp(argv[i], "--max-delete") == 0 && i + 1 < argc) { - int val; - if (!parse_nonneg_int(argv[++i], &val)) { - fprintf(stderr, "Error: --max-delete must be a non-negative integer\n"); + } else if (opt_is(argv[i], "--info", NULL)) { + if (i + 1 >= argc || parse_info_flags(argv[++i], config) != 0) + return -1; + } else if (strncmp(argv[i], "--skip-compress=", 16) == 0) { + if (parse_skip_compress(config, argv[i] + 16) != 0) + return -1; + } else if (opt_is(argv[i], "--skip-compress", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); return -1; } - config->max_delete = val; - } else if (strcmp(argv[i], "--filter") == 0 && i + 1 < argc) { - if (!config->filters) - config->filters = array_list_create(free); - char* dup = str_dup(argv[++i]); - if (!dup) + if (parse_skip_compress(config, argv[++i]) != 0) return -1; - array_list_add(config->filters, dup); - } else if (strcmp(argv[i], "--files-from") == 0 && i + 1 < argc) { - char* dup = str_dup(argv[++i]); - if (!dup) - return -1; - free(config->files_from); - config->files_from = dup; - } else if (strcmp(argv[i], "--cvs-exclude") == 0) { - config->cvs_exclude = true; - } else if (strcmp(argv[i], "--prune-empty-dirs") == 0) { - config->prune_empty_dirs = true; - } else if (strcmp(argv[i], "-R") == 0 || strcmp(argv[i], "--relative") == 0) { - config->relative = true; - } else if (strcmp(argv[i], "-e") == 0 || strcmp(argv[i], "--rsh") == 0) { - if (i + 1 < argc) { - char* dup = str_dup(argv[++i]); - if (!dup) - return -1; - free(config->rsh_command); - config->rsh_command = dup; - } else { - fprintf(stderr, "Error: -e/--rsh requires a command argument\n"); + } else if (opt_is(argv[i], "--compress-threads", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); return -1; } - } else if (strcmp(argv[i], "--rsync-path") == 0 && i + 1 < argc) { - char* dup = str_dup(argv[++i]); - if (!dup) + if (set_compression_threads_option(&config->compression_threads, argv[++i]) != 0) return -1; - free(config->rsync_path); - config->rsync_path = dup; - } else if (strcmp(argv[i], "--temp-dir") == 0 && i + 1 < argc) { - char* dup = str_dup(argv[++i]); - if (!dup) + } else if (opt_is(argv[i], "--checksum-choice", "--cc")) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); return -1; - free(config->temp_dir); - config->temp_dir = dup; - } else if (strcmp(argv[i], "--compare-dest") == 0 && i + 1 < argc) { - char* dup = str_dup(argv[++i]); - if (!dup) + } + if (set_checksum_choice(config, argv[++i]) != 0) return -1; - free(config->compare_dest); - config->compare_dest = dup; - } else if (strcmp(argv[i], "--copy-dest") == 0 && i + 1 < argc) { - char* dup = str_dup(argv[++i]); - if (!dup) + } else if (strncmp(argv[i], "--checksum-choice=", 18) == 0) { + if (set_checksum_choice(config, argv[i] + 18) != 0) return -1; - free(config->copy_dest); - config->copy_dest = dup; - } else if (strcmp(argv[i], "--link-dest") == 0 && i + 1 < argc) { - char* dup = str_dup(argv[++i]); - if (!dup) + } else if (strncmp(argv[i], "--cc=", 5) == 0) { + if (set_checksum_choice(config, argv[i] + 5) != 0) + return -1; + } else if (strncmp(argv[i], "--checksum-seed=", 16) == 0) { + if (set_checksum_seed(config, argv[i] + 16) != 0) + return -1; + } else if (opt_is(argv[i], "--checksum-seed", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for --checksum-seed"); + return -1; + } + if (set_checksum_seed(config, argv[++i]) != 0) + return -1; + } else if (strncmp(argv[i], "--sockopts=", 11) == 0) { + if (set_sockopts_option(config, argv[i] + 11) != 0) + return -1; + } else if (opt_is(argv[i], "--sockopts", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for --sockopts"); + return -1; + } + if (set_sockopts_option(config, argv[++i]) != 0) + return -1; + } else if (strncmp(argv[i], "--remote-option=", 16) == 0) { + if (config_add_remote_option(config, argv[i] + 16, "--remote-option") != 0) + return -1; + } else if (strncmp(argv[i], "-M=", 3) == 0) { + if (config_add_remote_option(config, argv[i] + 3, "-M") != 0) + return -1; + } else if (opt_is(argv[i], "--remote-option", "-M")) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for --remote-option"); + return -1; + } + if (config_add_remote_option(config, argv[++i], "--remote-option") != 0) + return -1; + } else if (strncmp(argv[i], "--compare-dest=", 15) == 0) { + if (set_basis_dest_option(config, BASIS_DEST_COMPARE, argv[i] + 15, "--compare-dest") != 0) + return -1; + } else if (opt_is(argv[i], "--compare-dest", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); + return -1; + } + if (set_basis_dest_option(config, BASIS_DEST_COMPARE, argv[++i], "--compare-dest") != 0) + return -1; + } else if (strncmp(argv[i], "--copy-dest=", 12) == 0) { + if (set_basis_dest_option(config, BASIS_DEST_COPY, argv[i] + 12, "--copy-dest") != 0) + return -1; + } else if (opt_is(argv[i], "--copy-dest", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); + return -1; + } + if (set_basis_dest_option(config, BASIS_DEST_COPY, argv[++i], "--copy-dest") != 0) + return -1; + } else if (strncmp(argv[i], "--link-dest=", 12) == 0) { + if (set_basis_dest_option(config, BASIS_DEST_LINK, argv[i] + 12, "--link-dest") != 0) + return -1; + } else if (opt_is(argv[i], "--link-dest", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); + return -1; + } + if (set_basis_dest_option(config, BASIS_DEST_LINK, argv[++i], "--link-dest") != 0) + return -1; + } else if (strncmp(argv[i], "--usermap=", 10) == 0) { + if (identity_parse_map(config, argv[i] + 10, false) != 0) + return -1; + config->use_metadata = true; + } else if (opt_is(argv[i], "--usermap", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); + return -1; + } + if (identity_parse_map(config, argv[++i], false) != 0) + return -1; + config->use_metadata = true; + } else if (strncmp(argv[i], "--groupmap=", 11) == 0) { + if (identity_parse_map(config, argv[i] + 11, true) != 0) + return -1; + config->use_metadata = true; + } else if (opt_is(argv[i], "--groupmap", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); + return -1; + } + if (identity_parse_map(config, argv[++i], true) != 0) + return -1; + config->use_metadata = true; + } else if (strncmp(argv[i], "--chown=", 8) == 0) { + if (identity_parse_chown(config, argv[i] + 8) != 0) + return -1; + config->use_metadata = true; + } else if (opt_is(argv[i], "--chown", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); + return -1; + } + if (identity_parse_chown(config, argv[++i]) != 0) + return -1; + config->use_metadata = true; + } else if (strncmp(argv[i], "--copy-as=", 10) == 0) { + if (identity_parse_copy_as(config, argv[i] + 10) != 0) + return -1; + } else if (opt_is(argv[i], "--copy-as", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); + return -1; + } + if (identity_parse_copy_as(config, argv[++i]) != 0) + return -1; + } else if (strncmp(argv[i], "--outbuf=", 9) == 0) { + if (set_outbuf_option(config, argv[i] + 9) != 0) + return -1; + } else if (opt_is(argv[i], "--outbuf", NULL)) { + if (i + 1 >= argc || set_outbuf_option(config, argv[++i]) != 0) return -1; - free(config->link_dest); - config->link_dest = dup; } else if (argv[i][0] == '-') { - fprintf(stderr, "Unknown option: %s\n", argv[i]); + char* escaped = output_escape(argv[i], false); + fprintf(stderr, "Unknown option: %s\n", escaped ? escaped : ""); + free(escaped); print_usage(); return -1; } else { if (*positional_count < 2) positional_args[(*positional_count)++] = i; else { - fprintf(stderr, "Unexpected argument: %s\n", argv[i]); + char* escaped = output_escape(argv[i], false); + fprintf(stderr, "Unexpected argument: %s\n", escaped ? escaped : ""); + free(escaped); print_usage(); return -1; } } } - return 0; -} + set_log_level(config->quiet ? LOG_LEVEL_ERROR : (verbose ? LOG_LEVEL_DEBUG : LOG_LEVEL_WARNING)); + if (config->compress_choice) + config->use_compression = strcmp(config->compress_choice, "zstd") == 0; -/* Validate config after parsing. Returns true if valid. */ -static bool validate_config(const Config* config) { - if (!config->send_directory || !config->receive_root_directory) { - fprintf(stderr, "Error: source and destination directories are required\n"); - print_usage(); - return false; - } - if (config->use_sendfile && (config->use_chunk_serialization || config->use_compression)) { - fprintf(stderr, "Error: -f/--sendfile cannot be combined with -c (compression) or -s (chunk " - "serialization)\n"); - return false; - } - if (config->transport == TRANSPORT_SSH && config->use_sendfile) { - fprintf(stderr, "Error: -f/--sendfile is not supported with SSH transport\n"); - return false; - } - if (config->use_incremental && config->use_chunk_serialization) { - fprintf(stderr, "Error: --incremental is not supported with -s (chunk serialization)\n"); - return false; - } - if (config->use_delta && !config->use_incremental) { - fprintf(stderr, "Error: --delta requires --incremental\n"); - return false; - } - if (config->use_delta && config->use_chunk_serialization) { - fprintf(stderr, "Error: --delta cannot be combined with -s (chunk serialization)\n"); - return false; - } - if (config->use_delta && config->use_sendfile) { - fprintf(stderr, "Error: --delta cannot be combined with -f (sendfile)\n"); - return false; - } - if (config->use_tls) { - if (!config->tls_cert || !config->tls_key) { - fprintf(stderr, "Error: --tls requires --cert and --key\n"); - return false; + /* --files-from is loaded after every argument is seen so that -0/--from0 may + * appear anywhere on the command line. A missing or unreadable file, and + * invalid (absolute / traversal) entries, are hard CLI errors. */ + if (config->files_from) { + char err[256]; + FileListSet* set = file_list_load(config->files_from, config->from0, err, sizeof(err)); + if (!set) { + log_message(LOG_LEVEL_ERROR, "--files-from: %s", err); + return -1; } + file_list_destroy((FileListSet*)config->files_from_set); + config->files_from_set = set; } - return true; -} -static void print_usage(void) { - printf("Usage:\n"); - printf(" fastsync [options] \n"); - printf(" fastsync [options] --source-dir --dest-dir \n"); - printf("\n"); - printf("Destination formats:\n"); - printf(" user@host:/path SSH transport (rsync-style)\n"); - printf(" host:/path SSH transport (current user)\n"); - printf(" /local/path TCP transport (requires server on localhost:8080)\n"); - printf("\n"); - printf("Options:\n"); - printf(" -c [level] Enable compression (level 1-22, default 5)\n"); - printf(" -z [level] Alias for -c\n"); - printf(" -a, --archive Archive mode (-c -m -M)\n"); - printf(" -n, --dry-run Show what would be transferred\n"); - printf(" -p SSH port (default: 22)\n"); - printf(" --progress Show transfer progress\n"); - printf(" --delete Delete files on receiver not in source\n"); - printf(" --exclude Exclude files matching pattern\n"); - printf(" --include Only include files matching pattern\n"); - printf(" --exclude-from Read exclude patterns from file\n"); - printf(" --include-from Read include patterns from file\n"); - printf(" --max-size Skip files larger than n bytes\n"); - printf(" --min-size Skip files smaller than n bytes\n"); - printf(" --incremental Skip files unchanged since last transfer\n"); - printf(" --delta Delta transfer for changed files (requires --incremental)\n"); - printf(" --delta-block Delta block size in bytes (default: %d)\n", - DELTA_BLOCK_SIZE_DEFAULT); - printf(" --delta-max Max file size for delta transfer (default: %llu)\n", - DELTA_MAX_FILE_SIZE); - printf(" -m Enable multithreading\n"); - printf(" -s Enable chunk serialization\n"); - printf(" -f Enable sendfile (TCP only, not with -c or -s)\n"); - printf(" -v, --verbose Enable debug logging\n"); - printf(" -M, --preserve Preserve file metadata\n"); - printf(" --chunk-size Chunk size in bytes (default: %d)\n", DEFAULT_CHUNK_SIZE); - printf(" --source-dir Source directory\n"); - printf(" --dest-dir Destination directory\n"); - printf(" --save-to-disk Write received files to disk\n"); - printf(" --server-host Server IP address (default: 127.0.0.1)\n"); - printf(" --server-port Server port (default: 8080)\n"); - printf(" --bwlimit Bandwidth limit in kilobytes per second\n"); - printf(" --tls Enable TLS encryption\n"); - printf(" --cert TLS certificate file (PEM)\n"); - printf(" --key TLS private key file (PEM)\n"); - printf(" --ca TLS CA certificate file (PEM)\n"); - printf(" --timeout I/O timeout in seconds (default: 30)\n"); - printf(" --contimeout Connection timeout in seconds (default: 10)\n"); - printf(" -q, --quiet Suppress non-error output\n"); - printf(" --silent Alias for --quiet\n"); - printf(" --backup Backup existing files before overwriting\n"); - printf(" --backup-dir Directory for backups (requires --backup)\n"); - printf(" --stats Print transfer statistics at end\n"); - printf(" --max-depth Maximum directory depth (0=unlimited)\n"); - printf(" --log-file Write log messages to file\n"); - printf(" --queue-size Queue capacity for multithreaded mode (default: 100)\n"); - printf(" --partial Keep partial files on interrupted transfer\n"); - printf(" --fastsync-server-path \n"); - printf(" Path to fastsync-server on remote (default: fastsync-server)\n"); - printf(" -l, --links Copy symlinks as symlinks\n"); - printf(" --copy-links Transform symlinks into referent files\n"); - printf(" --safe-links Skip symlinks that point outside transfer tree\n"); - printf(" --copy-unsafe-links Only transform unsafe symlinks into referent files\n"); - printf(" -H, --hard-links Preserve hard links\n"); - printf(" -A, --acls Preserve ACLs\n"); - printf(" -X, --xattrs Preserve extended attributes\n"); - printf(" -D, --devices Preserve device files\n"); - printf(" -S, --sparse Handle sparse files efficiently\n"); - printf(" -i, --itemize-changes Show per-file change summary\n"); - printf(" --out-format Custom output format string\n"); - printf(" --info Info verbosity level\n"); - printf(" --debug Debug verbosity level\n"); - printf(" --list-only List files without transferring\n"); - printf(" -h, --human-readable Human-readable numbers\n"); - printf(" -u, --update Skip files newer on destination\n"); - printf(" --inplace Update files in-place (no temp+rename)\n"); - printf(" --append Append data to shorter files\n"); - printf(" --append-verify Append with verify\n"); - printf(" --delete-excluded Also delete excluded files\n"); - printf(" --delete-after Delete after transfer, not before\n"); - printf(" --max-delete Maximum number of files to delete\n"); - printf(" --filter Add file filtering rule\n"); - printf(" --files-from Read file list from file\n"); - printf(" --cvs-exclude Auto-ignore CVS files\n"); - printf(" --prune-empty-dirs Omit empty directories from transfer\n"); - printf(" -R, --relative Use relative paths\n"); - printf(" -e, --rsh Specify remote shell\n"); - printf(" --rsync-path Path to remote binary\n"); - printf(" --temp-dir Temporary directory for files\n"); - printf(" --compare-dest Compare destination\n"); - printf(" --copy-dest Copy destination\n"); - printf(" --link-dest Link destination\n"); - printf(" --help Show this help\n"); + /* Device/special preservation recreates a node from its metadata mode (whose + S_IFMT bits carry the node kind), so --devices/--specials/-D imply metadata + transmission. --copy-devices/--write-devices treat the entry as data but a + mtime/mode-preserving transfer still benefits from metadata, so all four + imply it (FastSync's broad -M bundle; ownership stays opt-in). */ + if (config->preserve_devices || config->preserve_specials || config->copy_devices || + config->write_devices) + config->use_metadata = true; + + /* The "unchanged" decision for --compare-dest/--copy-dest/--link-dest must + * be made on the receiver against the basis directories, which requires the + * per-file STATUS_CHECK handshake: basis-dir options therefore imply + * --incremental (and, via the block below, metadata) on the sender. */ + if (config_has_basis(config)) + config->use_incremental = true; + + /* -y/--fuzzy reuses an existing similar-named destination file as the delta + * basis, so it is meaningless without the receiver-driven delta path: + * imply --incremental and --delta unless --whole-file or an explicit + * --no-delta / --no-incremental switched the machinery off. FastSync has + * delta OFF by default (unlike rsync), so a bare --fuzzy must turn it on or + * it would be a silent no-op. -W/--no-delta/--no-incremental therefore + * leave fuzzy inert, matching rsync where --whole-file makes fuzzy + * irrelevant (note: unlike the basis-dir options, --fuzzy honors an + * explicit --no-incremental instead of forcing the handshake back on). */ + if (config->fuzzy) { + if (!no_incremental) + config->use_incremental = true; + /* Delta needs the incremental per-file handshake, so an explicit + * --no-incremental also suppresses the delta implication. */ + if (!config->whole_file && !no_delta && !no_incremental) + config->use_delta = true; + } + + /* --append / --append-verify resume a shorter existing destination file. + * The receiver must run the per-file STATUS_CHECK handshake to learn the + * destination length and reply STATUS_APPEND, so an append mode forces + * --incremental on (exactly like the basis-dir options: the handshake is + * required, not optional). The resume itself is a dedicated tail-only + * exchange, not the block delta, so no delta implication is made. When both + * spelling are given the safer --append-verify semantics win. */ + if (config->append || config->append_verify) { + config->use_incremental = true; + } + + /* Incremental and delta transfers need metadata unless the user disabled it. */ + if ((config->use_incremental || config->use_delta) && !config->use_metadata && + !config->metadata_explicitly_disabled) { + log_message(LOG_LEVEL_INFO, "Enabling metadata preservation for incremental/delta transfer"); + config->use_metadata = true; + } + /* Recompute the derived xattr flag from the FINAL preserve flags (after any + * --no-xattrs/--no-acls negation) so the sender's wire gate always matches + * the flags the receiver will recompute from the received config. */ + config->use_xattrs = config->preserve_acls || config->preserve_xattrs; + return 0; } static int read_patterns_from_file(const char* filepath, char*** patterns, int* count) { FILE* fp = fopen(filepath, "r"); if (!fp) { - fprintf(stderr, "Error: could not open pattern file '%s': %s\n", filepath, strerror(errno)); + char* escaped = output_escape(filepath, false); + log_message(LOG_LEVEL_ERROR, "could not open pattern file '%s': %s", + escaped ? escaped : "", strerror(errno)); + free(escaped); return -1; } - char line[4096]; - while (fgets(line, sizeof(line), fp)) { + char* line = NULL; + size_t line_size = 0; + ssize_t n; + while ((n = getline(&line, &line_size, fp)) != -1) { char* p = line; while (*p == ' ' || *p == '\t') p++; @@ -573,36 +1561,64 @@ static int read_patterns_from_file(const char* filepath, char*** patterns, int* p[--len] = '\0'; if (len == 0) continue; - char** tmp = realloc(*patterns, (*count + 1) * sizeof(char*)); - if (!tmp) { - fprintf(stderr, "Error: memory allocation failed for pattern file\n"); + if (config_add_pattern(patterns, count, p, "pattern file") != 0) { + free(line); fclose(fp); return -1; } - *patterns = tmp; - char* dup = str_dup(p); - if (!dup) { - fprintf(stderr, "Error: memory allocation failed for pattern file\n"); - fclose(fp); - return -1; - } - (*patterns)[(*count)++] = dup; } + free(line); fclose(fp); return 0; } +#ifndef FASTSYNC_TEST_BUILD +/* Daemon auth (A7, protocol 2.19.0): read --password-file and keep the + * username plus the LITERAL password (client-only, never serialized). Runs + * once the destination form is known: the credentials only make sense for a + * daemon (host::module/path) destination, so a --password-file without one is a + * hard error here rather than a silently-ignored flag. The password is handed + * to the SCRAM challenge/response in config_send and burned by + * config_burn_auth/config_delete at teardown. Returns 0 on success, -1 on + * error (the reason is logged; the password is never logged). */ +static int load_daemon_credentials(Config* config) { + if (!config->password_file) + return 0; + if (!config->module || config->module[0] == '\0') { + log_message(LOG_LEVEL_ERROR, + "--password-file requires a daemon destination (host::module/path)"); + return -1; + } + char err[512]; + char* user = NULL; + char* password = NULL; + if (credentials_read_secret_file(config->password_file, &user, &password, err, sizeof(err)) != + 0) { + log_message(LOG_LEVEL_ERROR, "%s", err); + return -1; + } + + config_burn_auth(config); + config->auth_user = user; + config->auth_password = password; + log_info_message(LOG_INFO_MISC, "Loaded daemon credentials for user '%s'", config->auth_user); + return 0; +} + int main(int argc, char* argv[]) { + /* The server may close a connection mid-stream (e.g. when it rejects an + oversized delta). Ignore SIGPIPE so that a broken TCP connection + surfaces as a clean write error instead of killing the client. */ + signal(SIGPIPE, SIG_IGN); const char* env_source = NULL; const char* env_dest = NULL; bool save_to_disk = false; parse_environment(&env_source, &env_dest, &save_to_disk); int exit_code = 0; - bool config_owned_by_pipeline = false; Config* config = config_create(); if (!config) { - fprintf(stderr, "Error: failed to allocate config\n"); + log_message(LOG_LEVEL_ERROR, "failed to allocate config"); return 1; } config->save_to_disk = save_to_disk; @@ -623,28 +1639,50 @@ int main(int argc, char* argv[]) { free(config->receive_root_directory); config->send_directory = str_dup(argv[positional_args[0]]); if (!config->send_directory) { - fprintf(stderr, "Error: memory allocation failed\n"); + log_message(LOG_LEVEL_ERROR, "memory allocation failed"); exit_code = 1; goto cleanup; } config->receive_root_directory = str_dup(argv[positional_args[1]]); if (!config->receive_root_directory) { - fprintf(stderr, "Error: memory allocation failed\n"); + log_message(LOG_LEVEL_ERROR, "memory allocation failed"); exit_code = 1; goto cleanup; } config->save_to_disk = true; - config_parse_ssh_dest(config); } else if (positional_count == 1) { - fprintf(stderr, "Error: missing destination argument\n"); - print_usage(); - exit_code = 1; - goto cleanup; + if (config->read_batch) { + /* --read-batch= : the single positional is the destination + (there is no source). */ + free(config->receive_root_directory); + config->receive_root_directory = str_dup(argv[positional_args[0]]); + if (!config->receive_root_directory) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed"); + exit_code = 1; + goto cleanup; + } + config->save_to_disk = true; + } else if (config->only_write_batch) { + /* --only-write-batch= : the single positional is the + source (there is no destination). */ + free(config->send_directory); + config->send_directory = str_dup(argv[positional_args[0]]); + if (!config->send_directory) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed"); + exit_code = 1; + goto cleanup; + } + } else { + log_message(LOG_LEVEL_ERROR, "missing destination argument"); + print_usage(); + exit_code = 1; + goto cleanup; + } } else { if (!config->send_directory && env_source) { config->send_directory = str_dup(env_source); if (!config->send_directory) { - fprintf(stderr, "Error: memory allocation failed\n"); + log_message(LOG_LEVEL_ERROR, "memory allocation failed"); exit_code = 1; goto cleanup; } @@ -652,48 +1690,92 @@ int main(int argc, char* argv[]) { if (!config->receive_root_directory && env_dest) { config->receive_root_directory = str_dup(env_dest); if (!config->receive_root_directory) { - fprintf(stderr, "Error: memory allocation failed\n"); + log_message(LOG_LEVEL_ERROR, "memory allocation failed"); exit_code = 1; goto cleanup; } } } + /* Resolve the destination's transport form after the source/destination are + * final (positional, --dest-dir, or the FASTSYNC_DEST_DIR env fallback): + * host::module[/path] selects the daemon TCP transport, host:path the SSH + * transport, anything else stays local TCP. An invalid daemon destination + * already logged its reason and is a hard error here. */ + if (config_parse_transport_dest(config) < 0) { + exit_code = 1; + goto cleanup; + } + + /* Daemon auth: read --password-file (if any) into the wire credentials now + * that the destination's module is known. */ + if (load_daemon_credentials(config) != 0) { + exit_code = 1; + goto cleanup; + } + if (!validate_config(config)) { exit_code = 1; goto cleanup; } - /* Enable implicit flags */ - if (config->use_incremental && !config->use_metadata) { - log_message(LOG_LEVEL_INFO, "Enabling metadata preservation for --incremental"); - config->use_metadata = true; - } - if (config->use_delta && !config->use_metadata) { - log_message(LOG_LEVEL_INFO, "Enabling metadata preservation for --delta"); - config->use_metadata = true; + /* --iconv: install the sender-side local->wire conversion before any path is + scanned or serialized (the scanner and the chunk/data path read windows are + all driven from this process, so one global initialization covers every + send site). */ + if (!charset_wire_init_sender(config->iconv_spec)) { + log_message(LOG_LEVEL_ERROR, + "--iconv has an invalid CONVERT_SPEC or an unsupported charset name"); + exit_code = 1; + goto cleanup; } + /* Apply the requested --outbuf style now that the mode is parsed. */ + apply_output_buffering(config); + + /* --open-noatime is a sender-side policy: install it for every source read + (scan + data path) without touching the receiver. */ + file_set_open_noatime(config->open_noatime); + /* Initialize TLS if needed */ if (config->use_tls) tls_global_init(); tcp_set_timeouts(config->timeout, config->contimeout); + /* Phase 6 residual-batch driver modes. --read-batch / --only-write-batch are + purely local (apply a batch file, or emit one from a scan): neither connects + to nor transfers to a server. --write-batch runs the normal live transfer + AND then emits the batch FILE from a separate deterministic scan pass. It + drives the single-threaded transfer so the config outlives the run for that + second pass (the -m path takes ownership of the config). */ + if (config->read_batch) { + exit_code = apply_batch_to_dest(config, config->read_batch, config->receive_root_directory); + goto cleanup; + } + if (config->only_write_batch) { + exit_code = write_batch_from_source(config, config->only_write_batch); + goto cleanup; + } + /* Execute transfer */ - if (config->use_multithreading) { - config_owned_by_pipeline = true; - exit_code = send_files_multithreaded(config); + if (config->write_batch) { + exit_code = send_files(config); + if (exit_code == 0 && write_batch_from_source(config, config->write_batch) != 0) { + log_message(LOG_LEVEL_ERROR, "live transfer succeeded but batch emission failed"); + exit_code = 1; + } + } else if (config->use_multithreading) { + exit_code = send_files_multithreaded(&config); } else { exit_code = send_files(config); } cleanup: + charset_wire_free(); if (config) { - if (config->log_file) - fclose(config->log_file); - if (!config_owned_by_pipeline) - config_delete(config); + config_delete(config); } return exit_code; } +#endif /* FASTSYNC_TEST_BUILD */ diff --git a/src/client/client_send.c b/src/client/client_send.c index d295a82..f4db82e 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -1,81 +1,938 @@ #include "client_send.h" #include "array_list.h" +#include "batch.h" +#include "change_list.h" +#include "charset.h" #include "chunk.h" #include "compression.h" #include "config.h" #include "data.h" #include "delta.h" #include "file.h" +#include "file_list.h" +#include "filter.h" +#include "hardlink.h" #include "metadata.h" +#include "motd.h" #include "log.h" #include "multiprocessing.h" #include "protocol.h" #include "queue.h" #include "scanner.h" +#include "stop_condition.h" #include "transport_tcp.h" #include "transport_ssh.h" #include "transport_tls.h" #include "utils.h" +#include "xattr.h" +#include +#include #include #include #include #include #include +#include #include #define STREAM_THRESHOLD (64ULL * 1024 * 1024) -/* Print dry-run manifest showing files that would be transferred. Returns 0 on success. */ -static int send_dry_run_manifest(Config* config) { - DirectoryScanner* scanner = directory_scanner_create( - config->send_directory, config->use_metadata, config->chunk_size, config->exclude_patterns, - config->exclude_count, config->include_patterns, config->include_count, config->max_size, - config->min_size, config->max_depth, config->follow_symlinks, config->copy_links, - config->safe_links, config->copy_unsafe_links); +/* Forward declaration for progress-reporting thread used in multithreaded send. */ +static int progress_thread_fn(void* arg); + +static const char* display_bytes(unsigned long long bytes, bool human_readable, char* buffer, + size_t buffer_size) { + if (human_readable && format_human_bytes(bytes, buffer, buffer_size)) + return buffer; + snprintf(buffer, buffer_size, "%.1f MB", bytes / 1048576.0); + return buffer; +} + +/* Compiled scanner inputs that are shared read-only across scanner instances + * and, in -m mode, across worker threads. `base_filters` owns the compiled + * command-line + -C rules; the FileListSet allow-set lives in the Config. + * `hardlinks` owns the --hard-links/-H link-group detection table (NULL when + * off) and is shared (mutex-guarded) across every scanner/worker of one scan. */ +typedef struct { + ScannerOptions options; + FilterRuleList* base_filters; /* owned; may be NULL */ + HardLinkTable* hardlinks; /* owned; may be NULL */ +} PreparedScanner; + +/* Build the scanner options for one scan. Returns false and logs on failure. */ +static bool prepare_scanner(const Config* config, int num_threads, PreparedScanner* out) { + if (!out) + return false; + out->base_filters = NULL; + out->hardlinks = NULL; + memset(&out->options, 0, sizeof(out->options)); + + int rule_count = config->filters ? config->filters->size : 0; + const char** texts = NULL; + if (rule_count > 0) { + texts = malloc((size_t)rule_count * sizeof(char*)); + if (!texts) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed for filter rules"); + return false; + } + for (int i = 0; i < rule_count; i++) + texts[i] = (const char*)config->filters->items[i]; + } + if (rule_count > 0 || config->cvs_exclude) { + char err[160]; + out->base_filters = filter_base_build(texts, rule_count, config->cvs_exclude, err, sizeof(err)); + free(texts); + if (!out->base_filters) { + log_message(LOG_LEVEL_ERROR, "invalid filter rule: %s", err); + return false; + } + } else { + free(texts); + } + + ScannerOptions* options = &out->options; + options->use_metadata = config->use_metadata; + options->preserve_atimes = config->preserve_atimes; + options->preserve_crtimes = config->preserve_crtimes; + options->preserve_xattrs = config->preserve_xattrs; + options->preserve_acls = config->preserve_acls; + options->chunk_size = config->chunk_size; + options->exclude_patterns = config->exclude_patterns; + options->exclude_count = config->exclude_count; + options->include_patterns = config->include_patterns; + options->include_count = config->include_count; + options->max_size = config->max_size; + options->min_size = config->min_size; + options->max_depth = config->max_depth; + options->num_threads = num_threads; + options->follow_symlinks = config->follow_symlinks; + options->copy_links = config->copy_links; + options->safe_links = config->safe_links; + options->copy_unsafe_links = config->copy_unsafe_links; + options->copy_dirlinks = config->copy_dirlinks; + options->munge_links = config->munge_links; + options->checksum = config->checksum; + options->one_file_system = config->one_file_system; + options->preserve_devices = config->preserve_devices; + options->preserve_specials = config->preserve_specials; + options->copy_devices = config->copy_devices; + options->file_list = (const FileListSet*)config->files_from_set; + options->base_filters = out->base_filters; + options->per_dir_filters = config->per_dir_filter; + options->dirs = config->dirs; + options->relative = config->relative; + options->prune_empty_dirs = config->prune_empty_dirs; + options->ignore_io_errors = config->ignore_errors; + options->ignore_missing_args = config->ignore_missing_args || config->delete_missing_args; + options->excluded_paths = NULL; + options->excluded_mutex = NULL; + options->hardlinks = NULL; + /* P7 Wave D: capture source directory times whenever metadata rides the + wire. Whether they are APPLIED is decided receiver-side (-O skips). */ + options->capture_dir_times = config->use_metadata; + options->dir_entries = NULL; + options->dir_entries_mutex = NULL; + if (config->preserve_hard_links) { + out->hardlinks = hardlink_table_create(); + if (!out->hardlinks) { + filter_rule_list_free(out->base_filters); + out->base_filters = NULL; + return false; + } + options->hardlinks = out->hardlinks; + } + return true; +} + +static void prepared_scanner_destroy(PreparedScanner* prepared) { + if (!prepared) + return; + filter_rule_list_free(prepared->base_filters); + prepared->base_filters = NULL; + hardlink_table_destroy(prepared->hardlinks); + prepared->hardlinks = NULL; +} + +/* True when some --files-from entry is an ancestor-or-equal directory of + * `rel` (an empty entry -- the whole tree "." -- counts as the root). */ +static bool file_list_ancestor_listed(const FileListSet* set, const char* rel) { + if (!set) + return true; + for (int i = 0; i < set->count; i++) { + const char* listed = set->entries[i]; + if (listed[0] == '\0') + return true; + size_t n = strlen(listed); + if (strncmp(rel, listed, n) == 0 && (rel[n] == '/' || rel[n] == '\0')) + return true; + } + return false; +} + +/* --no-implied-dirs (meaningful only with -R + --files-from): a listed file + * may only be placed when its parent directory (or one of its ancestors) is + * itself an explicitly listed entry. rsync omits a file whose implied parent + * directory is suppressed, and an explicitly listed file that cannot be placed + * fails the transfer; FastSync fails the whole run up front with a clear error + * (it has no per-entry skip channel). Without -R or --files-from the option + * has no effect. */ +static bool no_implied_dirs_files_from_valid(const Config* config) { + if (!config->no_implied_dirs || !config->relative) + return true; + const FileListSet* set = (const FileListSet*)config->files_from_set; + if (!set) + return true; + for (int i = 0; i < set->count; i++) { + const char* entry = set->entries[i]; + if (entry[0] == '\0') + continue; + char* full = path_cat(config->send_directory, entry); + if (!full) + return false; + struct stat st; + bool is_file = lstat(full, &st) == 0 && S_ISREG(st.st_mode); + free(full); + if (!is_file) + continue; + const char* slash = strrchr(entry, '/'); + if (!slash) + continue; /* top-level file: its parent is the receive root */ + size_t parent_len = (size_t)(slash - entry); + if (parent_len == 0) + continue; + char* parent = malloc(parent_len + 1); + if (!parent) + return false; + memcpy(parent, entry, parent_len); + parent[parent_len] = '\0'; + bool listed = file_list_ancestor_listed(set, parent); + if (!listed) { + log_message(LOG_LEVEL_ERROR, + "--no-implied-dirs: cannot place file '%s': parent directory '%s' is not " + "explicitly listed (list the directory or drop --no-implied-dirs)", + entry, parent); + } + free(parent); + if (!listed) + return false; + } + return true; +} + +/* The destination-relative mirror path for a missing --files-from entry: where + a PRESENT entry with the same name would have been written. With -R that is + the entry's bare relative path (the bare wire path the receiver uses); + otherwise it is the full source mirror below the destination root + (`send_directory` joined to the entry, leading '/' stripped), exactly the + path the manifest records for a present sibling. Returns an owned string, or + NULL on allocation failure. */ +static char* files_from_missing_dest_path(const Config* config, const char* entry) { + if (config->relative) + return str_dup(entry); + char* joined = path_cat(config->send_directory, entry); + if (!joined) + return NULL; + const char* rel = *joined == '/' ? joined + 1 : joined; + char* dup = str_dup(rel); + free(joined); + return dup; +} + +/* --files-from semantics: every listed entry must resolve under the source + * root, otherwise rsync reports a hard error instead of silently transferring + * nothing. An empty list is also an error. An entry of "." (the whole tree) + * and listed-but-empty directories are valid. With --ignore-missing-args + * (implied by --delete-missing-args) a listed-but-missing entry is instead + * skipped: nothing is transferred for it, it never enters the keep-set and the + * run succeeds for the rest (an all-missing non-empty list succeeds + * transferring nothing, matching rsync). With --delete-missing-args + * `missing_dest` (when non-NULL) collects the entry's destination-relative + * mirror for the receiver's exact-deletion request. An empty list stays a + * hard error in every mode (nothing was requested at all). Runs before any + * transfer so the failure/skip is surfaced uniformly in the single-threaded, + * -m, dry-run and --list-only paths. */ +static bool files_from_list_check(const Config* config, ArrayList* missing_dest, int* skipped_out) { + *skipped_out = 0; + const FileListSet* set = (const FileListSet*)config->files_from_set; + if (!set) + return true; + if (!config->send_directory) { + log_message(LOG_LEVEL_ERROR, "--files-from requires a source directory"); + return false; + } + if (set->count == 0) { + log_message(LOG_LEVEL_ERROR, "--files-from file '%s' contains no entries; nothing to transfer", + config->files_from ? config->files_from : ""); + return false; + } + bool ignore = config->ignore_missing_args || config->delete_missing_args; + for (int i = 0; i < set->count; i++) { + const char* entry = set->entries[i]; + if (entry[0] == '\0') + continue; /* "." == list the whole tree */ + char* full = path_cat(config->send_directory, entry); + if (!full) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed while validating --files-from"); + return false; + } + struct stat st; + if (lstat(full, &st) != 0) { + free(full); + if (ignore) { + (*skipped_out)++; + log_info_message(LOG_INFO_MISC, "skipping missing --files-from entry '%s'", entry); + if (config->delete_missing_args && missing_dest) { + char* mirror = files_from_missing_dest_path(config, entry); + if (!mirror || !array_list_add(missing_dest, mirror)) { + free(mirror); + log_message(LOG_LEVEL_ERROR, "memory allocation failed while validating --files-from"); + return false; + } + } + continue; + } + log_message(LOG_LEVEL_ERROR, "--files-from entry '%s' not found in source '%s'", entry, + config->send_directory); + return false; + } + free(full); + } + if (*skipped_out > 0) { + if (config->delete_missing_args) { + /* --list-only never deletes and a --dry-run only shows intent, so the + summary must not claim a real deletion happened in those modes. */ + if (config->list_only) + log_message(LOG_LEVEL_WARNING, + "--delete-missing-args: %d missing --files-from entr%s skipped (--list-only " + "never deletes)", + *skipped_out, *skipped_out == 1 ? "y" : "ies"); + else if (config->dry_run) + log_message(LOG_LEVEL_WARNING, + "--delete-missing-args: %d missing --files-from entr%s would be deleted from " + "the destination (dry run)", + *skipped_out, *skipped_out == 1 ? "y" : "ies"); + else + log_message( + LOG_LEVEL_WARNING, + "--delete-missing-args: %d missing --files-from entr%s will be deleted from the " + "destination", + *skipped_out, *skipped_out == 1 ? "y" : "ies"); + } else if (config->ignore_missing_args) + log_message(LOG_LEVEL_WARNING, + "--ignore-missing-args: ignored %d missing --files-from entr%s", *skipped_out, + *skipped_out == 1 ? "y" : "ies"); + } + return no_implied_dirs_files_from_valid(config); +} + +/* Basis directories are honored by the receiver's per-file incremental check, + which (like every whole-file payload path in FastSync) is bounded by + MAX_RECEIVE_WHOLE_FILE_SIZE. rsync would apply basis dirs to files of any + size; FastSync cannot, so when basis dirs are requested this preflight scan + refuses the run up front with a clear diagnostic instead of letting the + receiver abort the whole transfer mid-stream with no client explanation. + Returns true when the tree can be transferred. */ +static bool basis_oversize_preflight(const Config* config) { + PreparedScanner prepared; + if (!prepare_scanner(config, 0, &prepared)) + return false; + DirectoryScanner* scanner = + directory_scanner_create_with_options(config->send_directory, &prepared.options); + prepared_scanner_destroy(&prepared); if (!scanner) + return false; + bool ok = true; + Chunk* chunk; + while ((chunk = directory_scanner_next(scanner)) != NULL) { + for (int i = 0; i < chunk->element_count; i++) { + File* f = chunk->items[i]; + if (f == NULL || f->is_dir || f->data == NULL || f->data->size <= MAX_RECEIVE_WHOLE_FILE_SIZE) + continue; + char* escaped = output_escape(file_wire_path(f), config->eight_bit_output); + log_message(LOG_LEVEL_ERROR, + "%s is %llu bytes, larger than the %llu-byte whole-file transfer limit; " + "--compare-dest/--copy-dest/--link-dest cannot sync files above this limit", + escaped ? escaped : "", (unsigned long long)f->data->size, + (unsigned long long)MAX_RECEIVE_WHOLE_FILE_SIZE); + free(escaped); + ok = false; + break; + } + chunk_destroy(chunk); + if (!ok) + break; + } + if (directory_scanner_failed(scanner) || directory_scanner_had_io_error(scanner)) + ok = false; + directory_scanner_destroy(scanner); + return ok; +} + +/* Read the daemon's MOTD frame and, unless --no-motd, display it on stdout. + * + * The daemon sends the MOTD as the first thing after the config-frame STATUS_OK + * on a host::module/path connection (rsync semantics), so this runs immediately + * after config_send succeeds. The frame is ALWAYS consumed for a daemon + * connection -- even with --no-motd -- so the byte stream stays in sync; the + * flag only suppresses the display. A non-daemon (local TCP / SSH) connection + * has no MOTD frame. The text is rendered through motd_render so a hostile + * server cannot inject terminal escape sequences. A read failure is not fatal + * here: the transfer that follows surfaces the real connection error. */ +static void receive_daemon_motd(Client* client, const Config* config) { + if (!config->module || config->module[0] == '\0') + return; + char* motd = motd_receive(client->file_descriptor); + if (!motd) + return; + if (!config->no_motd && motd[0] != '\0') { + char* rendered = motd_render(motd, config->eight_bit_output); + if (rendered) { + fputs(rendered, stdout); + size_t length = strlen(rendered); + if (length == 0 || rendered[length - 1] != '\n') + fputc('\n', stdout); + fflush(stdout); + free(rendered); + } + } + free(motd); +} + +/* Select the configured transport for both transfer execution paths. */ +static Client* connect_transfer_client(const Config* config) { + if (config->transport == TRANSPORT_SSH) { + if (config->use_sendfile) { + log_message(LOG_LEVEL_ERROR, "-f/--sendfile is not supported with SSH transport"); + return NULL; + } + return client_connect_ssh(config->ssh_destination, config->ssh_port, + config->fastsync_server_path, config->old_args, config->rsh_command, + config->blocking_io, config->remote_options, + config->remote_option_count); + } + + Client* client = client_create(); + if (!client) + return NULL; + /* Socket/connect concerns that never cross the wire: --address (source bind), + * -4/-6 (family pinning), and --sockopts. Passed straight to the TCP layer. */ + TcpConnectOptions connect_opts; + connect_opts.bind_address = config->address; + connect_opts.family = tcp_connect_family(config->ipv4, config->ipv6); + connect_opts.sockopts = config->sockopts; + connect_opts.sockopt_count = config->sockopt_count; + bool connected; + if (config->use_tls) { + connected = + client_connect_tls_ex(client, config->server_host, config->server_port, config->tls_cert, + config->tls_key, config->tls_ca, &connect_opts); + } else { + connected = client_connect_ex(client, config->server_host, config->server_port, &connect_opts); + } + if (!connected) { + client_disconnect(client); + client_delete(client); + return NULL; + } + return client; +} + +static void disconnect_transfer_client(Client* client) { + if (!client) + return; + client_disconnect(client); + client_delete(client); +} + +static bool add_chunk_to_manifest(ArrayList* manifest, const Chunk* chunk) { + if (!manifest) + return true; + for (int i = 0; i < chunk->element_count; i++) { + const char* path = file_wire_path(chunk->items[i]); + if (*path == '/') + path++; + char* entry = str_dup(path); + if (!entry) { + log_message(LOG_LEVEL_ERROR, "Failed to allocate manifest entry"); + return false; + } + if (!array_list_add(manifest, entry)) { + free(entry); + return false; + } + } + return true; +} + +/* (finalize_transfer is defined after the SourceFile helpers below.) */ + +typedef struct SourceFile { + char* path; + dev_t device; + ino_t inode; + bool skipped; /* receiver reported the file was not written */ +} SourceFile; + +static void source_file_destroy(void* item) { + SourceFile* source = item; + if (source) { + free(source->path); + free(source); + } +} + +/* Remove only the same regular source file that was sent. */ +static void remove_transferred_sources(const Config* config, ArrayList* paths) { + if (!config->remove_source_files || !paths) + return; + for (int i = 0; i < paths->size; i++) { + SourceFile* source = paths->items[i]; + if (source->skipped) + continue; + const char* slash = strrchr(source->path, '/'); + const char* leaf = slash ? slash + 1 : source->path; + char parent[PATH_MAX]; + if (slash) { + size_t parent_length = (size_t)(slash - source->path); + if (parent_length == 0) + parent_length = 1; + if (parent_length >= sizeof(parent)) + continue; + memcpy(parent, source->path, parent_length); + parent[parent_length] = '\0'; + } else { + (void)snprintf(parent, sizeof(parent), "."); + } + + int dirfd = open(parent, O_RDONLY | O_DIRECTORY | O_CLOEXEC); + if (dirfd < 0) + continue; + struct stat st; + if (fstatat(dirfd, leaf, &st, AT_SYMLINK_NOFOLLOW) != 0 || !S_ISREG(st.st_mode) || + st.st_dev != source->device || st.st_ino != source->inode) { + close(dirfd); + continue; + } + if (unlinkat(dirfd, leaf, 0) != 0) + log_message(LOG_LEVEL_WARNING, "Could not remove source file %s", source->path); + close(dirfd); + } +} + +static SourceFile* source_file_create(const File* file) { + if (!file || !file->path) + return NULL; + struct stat st; + if (lstat(file->path, &st) != 0 || !S_ISREG(st.st_mode)) + return NULL; + SourceFile* source = malloc(sizeof(*source)); + if (!source) + return NULL; + source->path = str_dup(file->path); + source->device = st.st_dev; + source->inode = st.st_ino; + source->skipped = false; + if (!source->path) { + source_file_destroy(source); + return NULL; + } + return source; +} + +static bool remember_source_file(ArrayList* paths, const File* file) { + if (!paths || !file || !file->path) + return true; + SourceFile* source = source_file_create(file); + if (!source) + return true; + if (!array_list_add(paths, source)) { + source_file_destroy(source); + return false; + } + return true; +} + +static void mark_sender_done(PipelineContextSender* context) { + mtx_lock(&context->mutex_progress); + context->sender_done = true; + mtx_unlock(&context->mutex_progress); +} + +/* Send the final STATUS_FINISHED frame and await the receiver's verdict. + When --remove-source-files is active the receiver acknowledges each data + file it processed, in send order: STATUS_NEXT means the file was written, + STATUS_OK means the file was skipped/unchanged. Skipped sources are marked + so the later removal pass keeps them. */ +static bool finalize_transfer(Client* client, const Config* config, ArrayList* remove_sources) { + if (!send_status(client->file_descriptor, STATUS_FINISHED)) + return false; + if (config->remove_source_files && remove_sources) { + for (int i = 0; i < remove_sources->size; i++) { + Status per_file; + if (!receive_status(client->file_descriptor, &per_file)) + return false; + if (per_file == STATUS_ERROR) + return false; + if (per_file == STATUS_OK) { + ((SourceFile*)remove_sources->items[i])->skipped = true; + } else if (per_file != STATUS_NEXT) { + log_message(LOG_LEVEL_ERROR, "Unexpected per-file status from receiver"); + return false; + } + } + } + Status status; + return receive_status(client->file_descriptor, &status) && status == STATUS_OK; +} + +static void pipeline_cancel(PipelineContextSender* context) { + mtx_lock(&context->mutex_scanner); + mtx_lock(&context->mutex_loader); + atomic_store(&context->cancelled, true); + context->scanner_done = true; + context->loader_done = true; + cnd_broadcast(&context->condition_not_full_scanner); + cnd_broadcast(&context->condition_not_empty_scanner); + cnd_broadcast(&context->condition_not_full_loader); + cnd_broadcast(&context->condition_not_empty_loader); + mtx_unlock(&context->mutex_loader); + mtx_unlock(&context->mutex_scanner); +} + +/* Print dry-run manifest showing files that would be transferred. Returns 0 on success. */ +static int send_dry_run_manifest(const Config* config) { + int skipped = 0; + ArrayList* missing_dest = NULL; + if (config->delete_missing_args) { + missing_dest = array_list_create(free); + if (!missing_dest) + return -1; + } + if (!files_from_list_check(config, missing_dest, &skipped)) { + if (missing_dest) + array_list_delete(missing_dest); return -1; + } + PreparedScanner prepared; + if (!prepare_scanner(config, 0, &prepared)) { + if (missing_dest) + array_list_delete(missing_dest); + return -1; + } + DirectoryScanner* scanner = + directory_scanner_create_with_options(config->send_directory, &prepared.options); + if (!scanner) { + prepared_scanner_destroy(&prepared); + if (missing_dest) + array_list_delete(missing_dest); + return -1; + } Chunk* chunk; int file_count = 0; unsigned long long total_bytes = 0; - printf("Dry run: files to be transferred\n"); + char size_buffer[32]; + if (!config->quiet) + printf("Dry run: files to be transferred\n"); while ((chunk = directory_scanner_next(scanner)) != NULL) { for (int i = 0; i < chunk->element_count; i++) { - printf(" %s (%zu bytes)\n", chunk->items[i]->path, chunk->items[i]->data->size); + if (!config->quiet) { + char* escaped_path = + output_escape(file_wire_path(chunk->items[i]), config->eight_bit_output); + if (!escaped_path) { + chunk_destroy(chunk); + directory_scanner_destroy(scanner); + prepared_scanner_destroy(&prepared); + if (missing_dest) + array_list_delete(missing_dest); + return -1; + } + if (config->human_readable) + printf( + " %s (%s)\n", escaped_path, + display_bytes(chunk->items[i]->data->size, true, size_buffer, sizeof(size_buffer))); + else + printf(" %s (%zu bytes)\n", escaped_path, chunk->items[i]->data->size); + free(escaped_path); + } total_bytes += chunk->items[i]->data->size; file_count++; } chunk_destroy(chunk); } directory_scanner_destroy(scanner); - printf("Total: %d files, %.1f MB\n", file_count, total_bytes / 1048576.0); + prepared_scanner_destroy(&prepared); + /* --delete-missing-args: the missing entries' destination mirrors render as + would-be deletions (rsync's dry-run also lists its *deleting lines). */ + if (missing_dest && !config->quiet) { + for (int i = 0; i < missing_dest->size; i++) { + char* escaped = output_escape((char*)missing_dest->items[i], config->eight_bit_output); + printf(" %s (missing; would be deleted)\n", escaped ? escaped : ""); + free(escaped); + } + } + if (missing_dest) + array_list_delete(missing_dest); + if (!config->quiet) { + if (config->human_readable) + printf("Total: %d files, %s\n", file_count, + display_bytes(total_bytes, true, size_buffer, sizeof(size_buffer))); + else + printf("Total: %d files, %.1f MB\n", file_count, total_bytes / 1048576.0); + } return 0; } -/* Send the delete manifest (list of files) to the server. Returns 0 on success, -1 on failure. */ -static int send_delete_manifest(int fd, ArrayList* manifest) { +typedef struct { + char* path; + mode_t mode; + unsigned long long size; + time_t mtime; +} ListEntry; + +static void list_entries_destroy(ListEntry* entries, size_t count) { + if (entries == NULL) + return; + for (size_t i = 0; i < count; i++) + free(entries[i].path); + free(entries); +} + +static int compare_list_entries(const void* left, const void* right) { + const ListEntry* a = (const ListEntry*)left; + const ListEntry* b = (const ListEntry*)right; + return strcmp(a->path, b->path); +} + +/* --list-only: print an ls-style listing of the files that WOULD be + * transferred and exit without contacting the server or writing anything. + * Directory lines are not printed because the scanner only yields regular + * transfer candidates. Returns 0 on success, 1 on error. */ +static int send_list_only(const Config* config) { + int skipped = 0; + if (!files_from_list_check(config, NULL, &skipped)) + return 1; + PreparedScanner prepared; + if (!prepare_scanner(config, 0, &prepared)) + return 1; + prepared.options.use_metadata = true; /* capture mode + mtime for the listing */ + DirectoryScanner* scanner = + directory_scanner_create_with_options(config->send_directory, &prepared.options); + if (!scanner) { + prepared_scanner_destroy(&prepared); + return 1; + } + ListEntry* entries = NULL; + size_t count = 0; + size_t capacity = 0; + Chunk* chunk; + bool oom = false; + while ((chunk = directory_scanner_next(scanner)) != NULL) { + for (int i = 0; i < chunk->element_count; i++) { + File* f = chunk->items[i]; + if (f == NULL) + continue; + if (count == capacity) { + size_t new_capacity = capacity > 0 ? capacity * 2 : 64; + if (new_capacity <= capacity) { + oom = true; + break; + } + ListEntry* grown = realloc(entries, new_capacity * sizeof(ListEntry)); + if (!grown) { + oom = true; + break; + } + entries = grown; + capacity = new_capacity; + } + char* path = str_dup(file_wire_path(f)); + if (!path) { + oom = true; + break; + } + mode_t mode = 0; + time_t mtime = 0; + if (f->metadata != NULL) { + mode = f->metadata->mode; + mtime = f->metadata->mtime_sec; + } else { + struct stat st; + if (stat(f->path, &st) == 0) { + mode = st.st_mode; + mtime = st.st_mtime; + } + } + entries[count].path = path; + entries[count].mode = mode; + entries[count].mtime = mtime; + entries[count].size = f->data != NULL ? f->data->size : 0; + count++; + } + chunk_destroy(chunk); + if (oom) + break; + } + bool failed = oom || directory_scanner_failed(scanner) || directory_scanner_had_io_error(scanner); + directory_scanner_destroy(scanner); + prepared_scanner_destroy(&prepared); + if (failed) { + list_entries_destroy(entries, count); + if (oom) + log_message(LOG_LEVEL_ERROR, "memory allocation failed while listing"); + return 1; + } + if (count > 1) + qsort(entries, count, sizeof(ListEntry), compare_list_entries); + for (size_t i = 0; i < count; i++) { + char* line = change_render_list_line(entries[i].mode, entries[i].size, entries[i].mtime, + entries[i].path); + if (line != NULL) { + char* escaped = output_escape(line, config->eight_bit_output); + printf("%s\n", escaped != NULL ? escaped : line); + free(escaped); + free(line); + } + } + list_entries_destroy(entries, count); + return 0; +} + +/* Send the delete manifest (keep-set paths plus the protected excluded + prefixes and the --delete-missing-args exact-delete paths) to the server. + Returns 0 on success, -1 on failure. When --delete-excluded is given + `protected` is empty: excluded destination mirrors are then ordinary extras + and are removed. When --delete-missing-args is active `missing_args` holds + the destination mirrors of missing --files-from entries: each is an explicit + receiver-side deletion request, independent of the extras walk. A NULL + keep-set / protected / missing list transmits an empty section. All three + sections are unbounded on the sender; the receiver enforces + MAX_MANIFEST_ENTRIES per section and a single MAX_MANIFEST_BYTES budget + shared across the sections, rejecting (with STATUS_ERROR) an over-budget + frame. A heavily filtered source whose exclusion list is large therefore + fails the run cleanly on the receiver rather than being truncated. */ +static int send_delete_manifest(int fd, ArrayList* manifest, ArrayList* protected_prefixes, + ArrayList* missing_args) { if (!send_status(fd, STATUS_MANIFEST)) return -1; - if (!send_int(fd, manifest->size)) + int keep_count = manifest ? manifest->size : 0; + if (!send_int(fd, keep_count)) return -1; - for (int i = 0; i < manifest->size; i++) { - if (!send_str(fd, (char*)manifest->items[i])) + for (int i = 0; i < keep_count; i++) { + if (!send_wire_str(fd, (char*)manifest->items[i])) + return -1; + } + int protected_count = protected_prefixes ? protected_prefixes->size : 0; + if (!send_int(fd, protected_count)) + return -1; + for (int i = 0; i < protected_count; i++) { + if (!send_wire_str(fd, (char*)protected_prefixes->items[i])) + return -1; + } + int missing_count = missing_args ? missing_args->size : 0; + if (!send_int(fd, missing_count)) + return -1; + for (int i = 0; i < missing_count; i++) { + if (!send_wire_str(fd, (char*)missing_args->items[i])) return -1; } return 0; } -static int incremental_check(Client* client, File* file, DeltaSignature** out_sig) { +/* Transmit the keep-set manifest and wait for the receiver's verdict. Used by + --delete-before/--delete-during, where the extras are removed on the receiver + BEFORE the first byte of file data is sent: the receiver acknowledges with + STATUS_OK once the bounded delete committed, or STATUS_ERROR if it could not + (in which case the sender aborts without streaming any data). The ACK may + take much longer than an ordinary per-message round trip because the receiver + performs the whole bounded deletion walk (up to MAX_SERVER_DELETE_COUNT + unlinks) before replying, so the wait uses a generous explicit deadline + instead of the default 60 s receive window. */ +#define DELETE_ACK_TIMEOUT_SEC 3600 + +static bool send_delete_manifest_early(Client* client, ArrayList* manifest, + ArrayList* protected_prefixes, ArrayList* missing_args) { + if (!client || !manifest) + return false; + if (send_delete_manifest(client->file_descriptor, manifest, protected_prefixes, missing_args) != + 0) + return false; + Status ack; + if (!receive_status_timed(client->file_descriptor, &ack, DELETE_ACK_TIMEOUT_SEC)) + return false; + if (ack != STATUS_OK) { + log_message(LOG_LEVEL_ERROR, "Server failed to delete files before the transfer"); + return false; + } + return true; +} + +/* Walk the whole source tree once collecting only destination-relative wire + paths, loading and sending nothing. --delete-before/--delete-during need the + complete keep-set manifest before the first data byte, so it is built by a + dedicated pre-scan pass and transmitted early; the data pass then re-scans + with a fresh scanner. A source I/O error is fatal unless the options carry + --ignore-errors, in which case the scan continues past the unreadable + directory and *io_error_out reports it (the caller still performs the + deletion but reports the run as errored). */ +static bool scan_paths_only(const Config* config, const ScannerOptions* options, + ArrayList* manifest, bool* io_error_out) { + if (io_error_out) + *io_error_out = false; + DirectoryScanner* scanner = + directory_scanner_create_with_options(config->send_directory, options); + if (!scanner) + return false; + bool ok = true; + Chunk* chunk; + while ((chunk = directory_scanner_next(scanner)) != NULL) { + if (!add_chunk_to_manifest(manifest, chunk)) { + ok = false; + chunk_destroy(chunk); + break; + } + chunk_destroy(chunk); + } + if (ok && directory_scanner_failed(scanner)) + ok = false; + if (io_error_out) + *io_error_out = directory_scanner_had_io_error(scanner); + directory_scanner_destroy(scanner); + return ok; +} + +static int incremental_check(Client* client, File* file, const Config* config, + DeltaSignature** out_sig, unsigned long long* resume_offset) { *out_sig = NULL; + if (resume_offset) + *resume_offset = 0; if (!send_status(client->file_descriptor, STATUS_CHECK)) return -1; - if (!send_str(client->file_descriptor, file->path)) + if (!send_wire_str(client->file_descriptor, file_wire_path(file))) return -1; unsigned long long fsize = file->data->size; long long mtime = file->metadata ? file->metadata->mtime_sec : 0; + long long mtime_nsec = file->metadata ? file->metadata->mtime_nsec : 0; if (!send_n_data(client->file_descriptor, &fsize, sizeof(fsize))) return -1; if (!send_n_data(client->file_descriptor, &mtime, sizeof(mtime))) return -1; + if (!send_n_data(client->file_descriptor, &mtime_nsec, sizeof(mtime_nsec))) + return -1; + /* With alternate basis directories the receiver must be able to verify the + * content of every candidate basis file, so the sender supplies its whole-file + * digest (computed with the negotiated --checksum-choice algorithm and + * --checksum-seed) for every file even when --checksum was not requested. */ + if (config->checksum || config_has_basis(config)) { + uint8_t digest[CHECKSUM_MAX_DIGEST_LEN]; + size_t digest_len = 0; + if (!file_checksum(file, (ChecksumAlgo)config->checksum_algo, config->checksum_seed, digest, + sizeof(digest), &digest_len)) + return -1; + uint8_t wire_len = (uint8_t)digest_len; + if (!send_n_data(client->file_descriptor, &wire_len, sizeof(wire_len)) || + !send_n_data(client->file_descriptor, digest, wire_len)) + return -1; + } Status s; if (!receive_status(client->file_descriptor, &s)) return -1; @@ -87,26 +944,47 @@ static int incremental_check(Client* client, File* file, DeltaSignature** out_si return 1; if (s == STATUS_DELTA_SIGNATURE) { Data* sig_data = receive_data(client->file_descriptor); - if (!sig_data) + if (!sig_data) { + send_status(client->file_descriptor, STATUS_ERROR); return -1; + } DeltaSignature* sig = delta_signature_deserialize(sig_data); data_destroy(sig_data); - if (!sig) + if (!sig) { + send_status(client->file_descriptor, STATUS_ERROR); return -1; + } *out_sig = sig; return 2; } + if (s == STATUS_APPEND) { + /* --append / --append-verify tail resume: the receiver found an existing + destination SHORTER than the source and wants only the tail from this + offset (the bytes it already holds). */ + unsigned long long offset; + if (!receive_n_data(client->file_descriptor, &offset, sizeof(offset))) { + send_status(client->file_descriptor, STATUS_ERROR); + return -1; + } + if (resume_offset) + *resume_offset = offset; + return 3; + } if (s != STATUS_NEXT) { log_message(LOG_LEVEL_ERROR, "Unexpected server status"); + send_status(client->file_descriptor, STATUS_ERROR); return -1; } return 0; } static int send_delta(Client* client, File* file, DeltaSignature* sig, Config* config) { - Delta* delta = delta_compute(file->data->data, file->data->size, sig, config->delta_block_size); + Delta* delta = delta_compute_seeded(file->data->data, file->data->size, sig, + config->delta_block_size, (uint32_t)config->checksum_seed); + /* The receiver is blocked after sending the signature. Every local + fallback therefore needs the explicit NEXT response before full data. */ if (!delta) - return 1; + return send_status(client->file_descriptor, STATUS_NEXT) ? 1 : -1; if (!delta_is_worthwhile(delta, file->data->size)) { delta_destroy(delta); @@ -118,14 +996,17 @@ static int send_delta(Client* client, File* file, DeltaSignature* sig, Config* c Data* delta_data = delta_serialize(delta); delta_destroy(delta); if (!delta_data) - return -1; + return send_status(client->file_descriptor, STATUS_NEXT) ? 1 : -1; Data* to_send = delta_data; - if (config->use_compression) { - to_send = data_compress(delta_data, config->compression_level); + int skip_count = config->skip_compress_set ? config->skip_compress_count : -1; + if (config->use_compression && !compression_should_skip_with_suffixes( + file->path, config->skip_compress_suffixes, skip_count)) { + to_send = data_compress_with_threads(delta_data, config->compression_level, + config->compression_threads); data_destroy(delta_data); if (!to_send) - return -1; + return send_status(client->file_descriptor, STATUS_NEXT) ? 1 : -1; } bool ok = send_status(client->file_descriptor, STATUS_DELTA_DATA) && @@ -134,24 +1015,180 @@ static int send_delta(Client* client, File* file, DeltaSignature* sig, Config* c if (ok && config->use_metadata) ok = metadata_send(client->file_descriptor, file->metadata); + if (ok && config->use_xattrs) + ok = xattr_send(client->file_descriptor, file->xattrs); + data_destroy(to_send); return ok ? 0 : -1; } -typedef bool (*file_send_fn)(File*, int, bool, int, bool); +/* --append / --append-verify tail resume. The receiver learned the existing + * destination is SHORTER than the source and replied STATUS_APPEND with the + * resume offset (prefix bytes it already holds). For plain --append we send + * the tail immediately (the prefix is not content-verified, matching rsync). + * For --append-verify we first send the source prefix xxHash64; the receiver + * compares it to the retained prefix and replies STATUS_APPEND_OK (send the + * tail) or STATUS_NEXT (prefix mismatch -> full transfer, never corrupt). + * Returns 0 on success, 1 when a full transfer was done instead, -1 on error. */ +static int send_append(const Client* client, File* file, Config* config, + unsigned long long offset) { + int fd = client->file_descriptor; + const unsigned long long fsize = file->data->size; + if (offset >= fsize) { + send_status(fd, STATUS_ERROR); + return -1; + } + size_t off = (size_t)offset; + size_t tail_len = (size_t)(fsize - off); + int compression_level = config->use_compression ? config->compression_level : 0; + int skip_count = config->skip_compress_set ? config->skip_compress_count : -1; + bool compress = compression_level > 0 && + !compression_should_skip_with_suffixes(file->path, config->skip_compress_suffixes, + skip_count); + + /* --append-verify: exchange the source prefix checksum and await the verdict. */ + if (config->append_verify) { + uint64_t prefix_hash = delta_xxhash64(file->data->data, off); + if (!send_status(fd, STATUS_APPEND_SIG) || !send_n_data(fd, &prefix_hash, sizeof(prefix_hash))) + return -1; + Status resp; + if (!receive_status(fd, &resp)) + return -1; + if (resp == STATUS_NEXT) { + /* Retained prefix does not match the source: fall back to the atomic full + transfer (byte-identical, never a corrupt prefix+tail blend). */ + int rc = file_send_single_calls_with_skip(file, fd, config->use_metadata, compression_level, + false, config->skip_compress_suffixes, skip_count, + config->compression_threads, config->use_xattrs) + ? 1 + : -1; + return rc; + } + if (resp != STATUS_APPEND_OK) { + send_status(fd, STATUS_ERROR); + return -1; + } + } + + if (!send_status(fd, STATUS_APPEND_DATA)) { + return -1; + } + if (config->use_metadata && !metadata_send(fd, file->metadata)) { + return -1; + } + if (config->use_xattrs && !xattr_send(fd, file->xattrs)) { + return -1; + } + bool ok; + if (compress) { + /* Compression needs an owned copy of the tail to compress. */ + Data* tail = data_create_empty(tail_len); + if (!tail) { + send_status(fd, STATUS_ERROR); + return -1; + } + memcpy(tail->data, (const char*)file->data->data + off, tail_len); + Data* comp = data_compress_with_threads(tail, compression_level, config->compression_threads); + data_destroy(tail); + if (!comp) { + send_status(fd, STATUS_ERROR); + return -1; + } + ok = send_data(fd, comp); + data_destroy(comp); + } else { + /* Uncompressed: send directly from the source buffer (no per-file copy; + send_data is synchronous, so the view outlives the call). */ + Data tail_view; + tail_view.data = (char*)file->data->data + off; + tail_view.size = tail_len; + tail_view.protocol_charge = 0; + ok = send_data(fd, &tail_view); + } + return ok ? 0 : -1; +} // Send a single file directly (non-incremental path). -static bool send_file_direct(File* file, int fd, bool use_metadata, int compression_level) { +static bool send_file_direct(File* file, int fd, bool use_metadata, int compression_level, + const Config* config) { if (!send_status(fd, STATUS_NEXT)) return false; - return file_send_single_calls(file, fd, use_metadata, compression_level, true); + int skip_count = config->skip_compress_set ? config->skip_compress_count : -1; + return file_send_single_calls_with_skip(file, fd, use_metadata, compression_level, true, + config->skip_compress_suffixes, skip_count, + config->compression_threads, config->use_xattrs); +} + +/* Transmit one explicit directory entry (--dirs): a STATUS_MKDIR frame whose + payload is the destination path and, when metadata is negotiated, the + directory's metadata frame. The receiver validates the path, creates the + directory under the receive root, and (metadata case) defers applying its + times to the end of the transfer so -O/--omit-dir-times is honored. */ +static bool send_directory_entry(const Client* client, File* file, const Config* config) { + if (!file || !file_wire_path(file)) + return false; + if (!send_status(client->file_descriptor, STATUS_MKDIR) || + !send_wire_str(client->file_descriptor, file_wire_path(file))) + return false; + return !config->use_metadata || metadata_send(client->file_descriptor, file->metadata); +} + +/* P7 Wave D: transmit every captured source directory's metadata in terminal + STATUS_DIR_TIMES frames (count, then (path, metadata) pairs) after all file + data and the optional delete manifest. The receiver applies them at the END + of its own transfer (after deletion and --delay-updates publication) so a + directory's mtime is not clobbered by writing its children. A non-metadata + transfer (or an empty set) sends nothing, keeping the stream byte-identical. + + The receiver rejects a frame whose count exceeds MAX_MANIFEST_ENTRIES, so a + huge tree is CHUNKED into repeated frames of at most that many entries each + (the receiver's loop handles repeated STATUS_DIR_TIMES frames). Every frame + stays within the receiver's bound, and a frame that would exceed it is never + emitted. */ +static bool send_dir_times(const Client* client, const Config* config, ArrayList* dir_entries) { + if (!client || !config || !config->use_metadata || !dir_entries || dir_entries->size == 0) + return true; + int fd = client->file_descriptor; + int index = 0; + while (index < dir_entries->size) { + int remaining = dir_entries->size - index; + int chunk = remaining > MAX_MANIFEST_ENTRIES ? MAX_MANIFEST_ENTRIES : remaining; + if (!send_status(fd, STATUS_DIR_TIMES) || !send_int(fd, chunk)) + return false; + for (int i = 0; i < chunk; i++) { + File* file = (File*)dir_entries->items[index + i]; + if (!file || !file_wire_path(file)) + return false; + if (!send_wire_str(fd, file_wire_path(file)) || !metadata_send(fd, file->metadata)) + return false; + } + index += chunk; + } + return true; +} + +/* Transmit one symlink entry: a STATUS_SYMLINK frame carrying the destination + * path, the (sender-munged, if --munge-links) target string, and metadata when + * negotiated. The receiver unmunges the target and creates the symlink beneath + * its root. Symlinks never need an incremental check or data payload. */ +static bool send_symlink_entry(const Client* client, File* file, const Config* config) { + if (!file || !file_wire_path(file) || !file->symlink_target) + return false; + int fd = client->file_descriptor; + if (!send_status(fd, STATUS_SYMLINK) || !send_wire_str(fd, file_wire_path(file)) || + !send_wire_str(fd, file->symlink_target)) + return false; + return !config->use_metadata || metadata_send(fd, file->metadata); } // Send a single file directly via sendfile (non-incremental path). -static bool send_file_direct_sendfile(File* file, int fd, bool use_metadata) { +static bool send_file_direct_sendfile(File* file, int fd, bool use_metadata, const Config* config) { if (!send_status(fd, STATUS_NEXT)) return false; - return file_send_sendfile(file, fd, use_metadata, 0, true); + int skip_count = config->skip_compress_set ? config->skip_compress_count : -1; + return file_send_sendfile_with_skip(file, fd, use_metadata, 0, true, + config->skip_compress_suffixes, skip_count, + config->compression_threads, config->use_xattrs); } // Process one file in a chunk: either via incremental check or direct send. @@ -159,13 +1196,16 @@ static bool send_file_direct_sendfile(File* file, int fd, bool use_metadata) { static int send_single_file(Client* client, File* file, Config* config, bool use_incremental, bool use_sendfile) { int compression_level = config->use_compression ? config->compression_level : 0; + log_info_message(LOG_INFO_COPY, "Transferring %s", file->path); if (!use_incremental) { if (use_sendfile) { - return send_file_direct_sendfile(file, client->file_descriptor, config->use_metadata) ? 0 - : -1; + return send_file_direct_sendfile(file, client->file_descriptor, config->use_metadata, config) + ? 0 + : -1; } - return send_file_direct(file, client->file_descriptor, config->use_metadata, compression_level) + return send_file_direct(file, client->file_descriptor, config->use_metadata, compression_level, + config) ? 0 : -1; } @@ -173,8 +1213,10 @@ static int send_single_file(Client* client, File* file, Config* config, bool use // Incremental path: use sendfile for the actual data if enabled and no compression if (use_sendfile) { DeltaSignature* sig = NULL; - int rc = incremental_check(client, file, &sig); + unsigned long long resume_offset = 0; + int rc = incremental_check(client, file, config, &sig, &resume_offset); if (rc == 1) { + log_info_message(LOG_INFO_SKIP, "Skipping unchanged %s", file->path); delta_signature_destroy(sig); return 1; } @@ -182,6 +1224,16 @@ static int send_single_file(Client* client, File* file, Config* config, bool use delta_signature_destroy(sig); return -1; } + // rc == 3: append resume (tail-only) -- send_append uses the data path. + if (rc == 3) { + delta_signature_destroy(sig); + int arc = send_append(client, file, config, resume_offset); + if (arc == 1) { + log_info_message(LOG_INFO_COPY, "Append prefix mismatch; full transfer of %s", file->path); + return 0; + } + return arc == 0 ? 0 : -1; + } // rc == 0: unchanged file, skip // rc == 2: server sent delta signature but sendfile doesn't support delta delta_signature_destroy(sig); @@ -191,24 +1243,39 @@ static int send_single_file(Client* client, File* file, Config* config, bool use return -1; } // Fall through: send full file via sendfile (pass 0 for compression_level) - if (!file_send_sendfile(file, client->file_descriptor, config->use_metadata, 0, false)) + int skip_count = config->skip_compress_set ? config->skip_compress_count : -1; + if (!file_send_sendfile_with_skip(file, client->file_descriptor, config->use_metadata, 0, false, + config->skip_compress_suffixes, skip_count, + config->compression_threads, config->use_xattrs)) return -1; return 0; } // Incremental path with single_calls (supports compression and delta) - file_send_fn send_fn = (file_send_fn)file_send_single_calls; DeltaSignature* sig = NULL; - int rc = incremental_check(client, file, &sig); + unsigned long long resume_offset = 0; + int rc = incremental_check(client, file, config, &sig, &resume_offset); if (rc < 0) { delta_signature_destroy(sig); return -1; } if (rc == 1) { + log_info_message(LOG_INFO_SKIP, "Skipping unchanged %s", file->path); delta_signature_destroy(sig); return 1; } - if (rc == 2 && config->use_delta) { + if (rc == 3) { + /* --append / --append-verify tail resume. send_append reports 1 when the + verified prefix mismatched and a full transfer was sent instead. */ + delta_signature_destroy(sig); + int arc = send_append(client, file, config, resume_offset); + if (arc == 1) { + log_info_message(LOG_INFO_COPY, "Append prefix mismatch; full transfer of %s", file->path); + return 0; + } + return arc == 0 ? 0 : -1; + } + if (rc == 2 && config->use_delta && !config->whole_file) { int drc = send_delta(client, file, sig, config); delta_signature_destroy(sig); if (drc == 0) @@ -225,18 +1292,43 @@ static int send_single_file(Client* client, File* file, Config* config, bool use return -1; } } - if (!send_fn(file, client->file_descriptor, config->use_metadata, compression_level, false)) + int skip_count = config->skip_compress_set ? config->skip_compress_count : -1; + if (!file_send_single_calls_with_skip(file, client->file_descriptor, config->use_metadata, + compression_level, false, config->skip_compress_suffixes, + skip_count, config->compression_threads, + config->use_xattrs)) return -1; return 0; } -int send_chunk(Client* client, Chunk* chunk, Config* config) { +/* Sendfile calls a blocking open() on the source (file_send_sendfile_with_skip + * -> file_open_for_read), which never returns for a FIFO/device with no writer. + * Only a regular file may take the zero-copy sendfile path; a non-regular source + * (FIFO/device copied by --copy-devices) must use the buffered, size-bounded + * read path instead. `stat` follows symlinks, so a dereferenced symlink to a + * regular file keeps the sendfile fast path. */ +static bool source_is_regular_file(const File* file) { + if (!file || !file->path) + return false; + struct stat st; + return stat(file->path, &st) == 0 && S_ISREG(st.st_mode); +} + +static int send_chunk_with_removal(Client* client, Chunk* chunk, Config* config, + ArrayList* remove_sources) { if (config->use_chunk_serialization) { + if (remove_sources) { + for (int i = 0; i < chunk->element_count; i++) { + if (!remember_source_file(remove_sources, chunk->items[i])) + return -1; + } + } if (!send_status(client->file_descriptor, STATUS_CHUNK)) return -1; Data* data; if (config->use_compression) { - data = chunk_compress(chunk, config->compression_level, config->use_metadata); + data = chunk_compress_with_threads(chunk, config->compression_level, config->use_metadata, + config->compression_threads); } else { data = chunk_serialize(chunk, config->use_metadata); } @@ -247,6 +1339,14 @@ int send_chunk(Client* client, Chunk* chunk, Config* config) { return -1; } data_destroy(data); + for (int i = 0; i < chunk->element_count; i++) { + if (chunk->items[i] == NULL) + continue; + if (chunk->items[i]->is_dir) + change_emit_dir_sent(config, chunk->items[i]); + else + change_emit_file_sent(config, chunk->items[i]); + } return 0; } @@ -254,121 +1354,349 @@ int send_chunk(Client* client, Chunk* chunk, Config* config) { File* f = chunk->items[i]; if (f == NULL) continue; - bool stream = f->data->data == NULL && f->data->size > 0; - bool use_sendfile = (config->use_sendfile && !config->use_compression) || stream; - int rc = send_single_file(client, f, config, config->use_incremental, use_sendfile); - if (rc == 1) + if (f->is_dir) { + /* Explicit directory entry (--dirs): a MKDIR frame carrying the + destination path (and metadata when negotiated). Directories have no + source to remove and no incremental check. */ + if (!send_directory_entry(client, f, config)) + return -1; + change_emit_dir_sent(config, f); continue; - if (rc < 0) + } + /* --hard-links/-H sibling: a later member of a hard-link group that has no + data (its payload lives in the first member). Transmit a dedicated + STATUS_HARDLINK frame carrying the first member's destination-relative + wire path so the receiver links this entry to that installed file. */ + if (f->link_group != 0 && !f->link_first && f->hardlink_target != NULL) { + if (!send_status(client->file_descriptor, STATUS_HARDLINK) || + !send_wire_str(client->file_descriptor, file_wire_path(f)) || + !send_int(client->file_descriptor, f->link_group) || + !send_wire_str(client->file_descriptor, f->hardlink_target)) + return -1; + change_emit_file_sent(config, f); + continue; + } + /* Symlink entry (-l / -k keep-as-symlink): only the target rides the wire. */ + if (f->is_symlink) { + if (!send_symlink_entry(client, f, config)) + return -1; + change_emit_file_sent(config, f); + continue; + } + /* --devices/--specials: a device/special node is recreated on the receiver, + not transferred as content. Send the dedicated STATUS_SPECIAL frame. */ + if (f->is_special) { + if (!file_send_special(f, client->file_descriptor, config->use_metadata)) + return -1; + change_emit_file_sent(config, f); + continue; + } + bool stream = f->data->data == NULL && f->data->size > 0; + bool use_sendfile = ((config->use_sendfile && !config->use_compression) || + (stream && !config->use_compression)) && + source_is_regular_file(f); + SourceFile* source = remove_sources ? source_file_create(f) : NULL; + int rc = send_single_file(client, f, config, config->use_incremental, use_sendfile); + if (rc == 1) { + source_file_destroy(source); + continue; + } + if (rc < 0) { + source_file_destroy(source); return -1; + } + change_emit_file_sent(config, f); + if (source && !array_list_add(remove_sources, source)) { + source_file_destroy(source); + return -1; + } } return 0; } +int send_chunk(Client* client, Chunk* chunk, Config* config) { + return send_chunk_with_removal(client, chunk, config, NULL); +} + static int send_chunks_multithreaded(void* pipeline_context) { PipelineContextSender* context = (PipelineContextSender*)pipeline_context; - Client* client; - if (context->config->transport == TRANSPORT_SSH) { - if (context->config->use_sendfile) { - fprintf(stderr, "Error: -f/--sendfile is not supported with SSH transport\n"); - return 1; - } - client = client_connect_ssh(context->config->ssh_destination, context->config->ssh_port, - context->config->fastsync_server_path); - } else if (context->config->use_tls) { - client = client_create(); - if (!client || !client_connect_tls(client, context->config->server_host, - context->config->server_port, context->config->tls_cert, - context->config->tls_key, context->config->tls_ca)) { - if (client) - client_delete(client); - fprintf(stderr, "Error: could not connect to server via TLS\n"); - return thrd_error; - } - } else { - client = client_create(); - if (!client || - !client_connect(client, context->config->server_host, context->config->server_port)) { - if (client) - client_delete(client); - fprintf(stderr, "Error: could not connect to server\n"); - return thrd_error; - } - } - if (!config_send(client->file_descriptor, context->config)) { - client_disconnect(client); - client_delete(client); + Client* client = connect_transfer_client(context->config); + if (!client) { + if (context->config->transport == TRANSPORT_TCP) + log_message(LOG_LEVEL_ERROR, "could not connect to server%s", + context->config->use_tls ? " via TLS" : ""); + pipeline_cancel(context); + mark_sender_done(context); return thrd_error; } + ProtocolSession session; + protocol_session_init(&session, client->file_descriptor, client->file_descriptor); + protocol_session_set_ssl(&session, (SSL*)client->ssl); + protocol_session_bind(&session); + if (!config_send(client->file_descriptor, context->config)) { + pipeline_cancel(context); + disconnect_transfer_client(client); + mark_sender_done(context); + protocol_session_unbind(); + return thrd_error; + } + receive_daemon_motd(client, context->config); + if (context->early_delete) { + /* The keep-set manifest was prebuilt by a path-only pre-scan. Transmit it + and wait for the receiver to delete extras before streaming any data. */ + if (!send_delete_manifest_early(client, context->manifest, context->excluded_paths, + context->missing_args)) { + pipeline_cancel(context); + disconnect_transfer_client(client); + mark_sender_done(context); + protocol_session_unbind(); + return thrd_error; + } + } while (true) { + /* Phase 6: stop-elegantly at the next chunk boundary once the --stop-after + / --stop-at deadline has passed. Everything already sent is finalized by + the completion tail below; the run still returns success. */ + if (stop_condition_reached(&context->stop_condition)) { + log_info_message(LOG_INFO_MISC, + "Stop deadline reached; stopping transfer at the next chunk boundary"); + context->scan_stopped_early = true; + pipeline_cancel(context); + break; + } Chunk* current_chunk = queue_dequeue_multithreaded( context->queue_loader, &context->mutex_loader, &context->condition_not_empty_loader, &context->condition_not_full_loader, &context->loader_done); if (current_chunk == NULL) { - if (context->config->use_delete) { - if (send_delete_manifest(client->file_descriptor, context->manifest) != 0) - goto send_fail; + if (atomic_load(&context->cancelled)) { + pipeline_cancel(context); + disconnect_transfer_client(client); + mark_sender_done(context); + protocol_session_unbind(); + return thrd_error; } - if (!send_status(client->file_descriptor, STATUS_FINISHED)) - goto send_fail; - Status s; - int ok = receive_status(client->file_descriptor, &s) && s == STATUS_OK; - client_disconnect(client); - client_delete(client); - return ok ? thrd_success : thrd_error; - - send_fail: - client_disconnect(client); - client_delete(client); + break; + } + if (send_chunk_with_removal(client, current_chunk, context->config, + context->remove_source_files) != 0) { + log_message(LOG_LEVEL_ERROR, "unexpected error while sending chunk"); + chunk_destroy(current_chunk); + pipeline_cancel(context); + disconnect_transfer_client(client); + mark_sender_done(context); + protocol_session_unbind(); return thrd_error; } - if (send_chunk(client, current_chunk, context->config) != 0) { - fprintf(stderr, "Error: unexpected error while sending chunk\n"); - client_disconnect(client); - client_delete(client); - return thrd_error; + unsigned long long chunk_bytes = 0; + int chunk_files = 0; + for (int i = 0; i < current_chunk->element_count; i++) { + if (current_chunk->items[i] && current_chunk->items[i]->data) { + chunk_files++; + chunk_bytes += current_chunk->items[i]->data->size; + } } + mtx_lock(&context->mutex_progress); + context->total_files += chunk_files; + context->total_bytes += chunk_bytes; + context->progress_bytes = context->total_bytes; + mtx_unlock(&context->mutex_progress); chunk_destroy(current_chunk); } + + /* Completion tail: reached on natural exhaustion or an early stop deadline. + A deadline that cut the scan short leaves an incomplete keep-set manifest; + transmitting it would make the receiver --delete the unscanned source + mirrors (data loss), so it is deliberately suppressed. Suppressing it also + means the manifest (which the scanner thread may still be appending) is + never read here on the early-stop path, so no scanner synchronization is + required to enter the tail. */ + context->scan_stopped_early = + context->scan_stopped_early || stop_condition_reached(&context->stop_condition); + if (context->scan_stopped_early) { + if (context->config->use_delete || context->config->delete_missing_args) + log_message(LOG_LEVEL_WARNING, + "transfer stopped early (stop deadline); skipping --delete keep-set so " + "unscanned source mirrors are not deleted"); + else + log_message(LOG_LEVEL_WARNING, "transfer stopped early (stop deadline)"); + } else if (context->config->use_delete && !context->early_delete) { + /* Empty keep-set + scan I/O error must not delete the whole destination + (the source may not be genuinely empty -- see send_files). */ + bool empty_io; + mtx_lock(&context->mutex_scanner); + empty_io = context->scan_had_io_error && context->manifest && context->manifest->size == 0; + mtx_unlock(&context->mutex_scanner); + if (empty_io) { + log_message(LOG_LEVEL_ERROR, + "source scan hit an I/O error before finding any file; refusing to delete " + "with an empty keep-set (--delete)"); + goto send_fail; + } + if (send_delete_manifest(client->file_descriptor, context->manifest, context->excluded_paths, + context->missing_args) != 0) + goto send_fail; + } else if (context->config->delete_missing_args && !context->early_delete) { + /* --delete-missing-args without --delete: no keep-set is built, but the + exact-delete paths still ride the same manifest frame (commit once the + transfer succeeded). */ + if (send_delete_manifest(client->file_descriptor, NULL, NULL, context->missing_args) != 0) + goto send_fail; + } + /* P7 Wave D: transmit the captured directory times last. The scanner thread + (and all parallel workers) has been joined before scanner_done was set, so + the list is complete and race-free; on an early stop the list may be + incomplete and is deliberately not sent. */ + if (!context->scan_stopped_early && + !send_dir_times(client, context->config, context->dir_entries)) + goto send_fail; + bool ok = finalize_transfer(client, context->config, context->remove_source_files); + if (!ok && context->config->use_delete) + log_message(LOG_LEVEL_ERROR, + "server reported a deletion failure (--delete); see the server log for the " + "reason (a --max-delete limit that the run would exceed deletes nothing)"); + if (ok) + remove_transferred_sources(context->config, context->remove_source_files); + mtx_lock(&context->mutex_progress); + int total_files = context->total_files; + unsigned long long total_bytes = context->total_bytes; + mtx_unlock(&context->mutex_progress); + if (context->config->stats) + fprintf(stderr, "Stats: %d files, %.1f MB\n", total_files, total_bytes / 1048576.0); + log_info_message(LOG_INFO_STATS, "Transfer summary: %d files, %.1f MB", total_files, + total_bytes / 1048576.0); + disconnect_transfer_client(client); + mark_sender_done(context); + protocol_session_unbind(); + return ok ? thrd_success : thrd_error; + +send_fail: + pipeline_cancel(context); + disconnect_transfer_client(client); + mark_sender_done(context); + protocol_session_unbind(); + return thrd_error; } +/* Scan thread of the -m pipeline. --dirs disables recursive traversal (the + transfer is a small set of explicit directory/file entries), so it uses the + sequential scanner rather than spawning worker threads. */ static int scan_directory_multithreaded(void* pipeline_context) { PipelineContextSender* context = (PipelineContextSender*)pipeline_context; - ParallelScanner* scanner = parallel_scanner_create( - context->config->send_directory, context->config->use_metadata, context->config->chunk_size, - context->config->exclude_patterns, context->config->exclude_count, - context->config->include_patterns, context->config->include_count, context->config->max_size, - context->config->min_size, context->config->max_depth, 4, context->config->follow_symlinks, - context->config->copy_links, context->config->safe_links, context->config->copy_unsafe_links); - + protocol_session_bind(&context->allocation_session); + PreparedScanner prepared; + if (!prepare_scanner(context->config, 4, &prepared)) { + pipeline_cancel(context); + protocol_session_unbind(); + return thrd_error; + } + prepared.options.stop_condition = &context->stop_condition; + /* P7 Wave D: the recursive scan feeds the shared directory-time list; the + parallel workers append under the context's dedicated mutex. */ + prepared.options.dir_entries = context->dir_entries; + prepared.options.dir_entries_mutex = &context->dir_entries_mutex; + /* The keep-set manifest for the late modes is built from this data pass, so + the parallel scanner records the protected excluded prefixes here. The + early modes already transmitted the pre-scan keep-set and its protected + list, so the data pass must not append to it again. */ + if (!context->early_delete) + prepared.options.excluded_paths = context->excluded_paths; + bool dirs_mode = prepared.options.dirs; + /* -H also selects the sequential scanner (see the comment at the branch), + * so the loop below must choose the scanner by which object exists, not by + * --dirs alone. */ + bool use_dscanner = dirs_mode || prepared.options.hardlinks; + DirectoryScanner* dscanner = NULL; + ParallelScanner* scanner = NULL; + /* --hard-links/-H forces the sequential scanner even in -m mode: a hard-link + group's first member must be emitted before any of its siblings so the + receiver always links to an already-installed first member. The parallel + scanner hands different subdirectories to different worker threads, which + can reorder a group whose members span directories. */ + if (use_dscanner) { + dscanner = + directory_scanner_create_with_options(context->config->send_directory, &prepared.options); + } else { + scanner = parallel_scanner_create_with_options(context->config->send_directory, + &prepared.options, &context->allocation_session); + } + if (dscanner == NULL && scanner == NULL) { + log_message(LOG_LEVEL_ERROR, "Failed to create scanner"); + pipeline_cancel(context); + prepared_scanner_destroy(&prepared); + protocol_session_unbind(); + return thrd_error; + } + bool failed = false; Chunk* current_chunk; - while ((current_chunk = parallel_scanner_next(scanner)) != NULL) { - if (context->config->use_delete) { - mtx_lock(&context->mutex_scanner); - for (int i = 0; i < current_chunk->element_count; i++) { - const char* p = current_chunk->items[i]->path; - if (*p == '/') - p++; - array_list_add(context->manifest, str_dup(p)); - } - mtx_unlock(&context->mutex_scanner); + while (1) { + if (use_dscanner) + current_chunk = directory_scanner_next(dscanner); + else + current_chunk = parallel_scanner_next(scanner); + if (current_chunk == NULL) { + failed = use_dscanner ? directory_scanner_failed(dscanner) : parallel_scanner_failed(scanner); + break; } - queue_enqueue_multithreaded(context->queue_scanner, current_chunk, &context->mutex_scanner, - &context->condition_not_empty_scanner, - &context->condition_not_full_scanner); + if (context->config->use_delete && !context->early_delete) { + mtx_lock(&context->mutex_scanner); + bool manifest_ok = add_chunk_to_manifest(context->manifest, current_chunk); + mtx_unlock(&context->mutex_scanner); + if (!manifest_ok) { + failed = true; + chunk_destroy(current_chunk); + break; + } + } + if (!queue_enqueue_multithreaded_cancel( + context->queue_scanner, current_chunk, &context->mutex_scanner, + &context->condition_not_empty_scanner, &context->condition_not_full_scanner, + &context->cancelled)) { + chunk_destroy(current_chunk); + failed = true; + break; + } + } + /* Capture the scanner results BEFORE destroying the scanner objects (the + io_error flag lives on the scanner, so reading it after destroy would be a + use-after-free). */ + bool had_io = use_dscanner ? directory_scanner_had_io_error(dscanner) + : parallel_scanner_had_io_error(scanner); + if (use_dscanner) + directory_scanner_destroy(dscanner); + else + parallel_scanner_destroy(scanner); + if (failed) { + prepared_scanner_destroy(&prepared); + mtx_lock(&context->mutex_scanner); + context->scanner_done = true; + cnd_broadcast(&context->condition_not_empty_scanner); + cnd_broadcast(&context->condition_not_full_scanner); + mtx_unlock(&context->mutex_scanner); + pipeline_cancel(context); + protocol_session_unbind(); + return thrd_error; + } + /* --ignore-errors: an unreadable subdirectory was skipped (workers recorded + io_error, not failure); the deletion still runs but the run reports it. */ + if (had_io) { + mtx_lock(&context->mutex_scanner); + context->scan_had_io_error = true; + mtx_unlock(&context->mutex_scanner); } mtx_lock(&context->mutex_scanner); context->scanner_done = true; cnd_signal(&context->condition_not_empty_scanner); mtx_unlock(&context->mutex_scanner); - parallel_scanner_destroy(scanner); + prepared_scanner_destroy(&prepared); + protocol_session_unbind(); return thrd_success; } static int load_files_multithreaded(void* pipeline_context) { PipelineContextSender* context = (PipelineContextSender*)pipeline_context; + protocol_session_bind(&context->allocation_session); while (true) { Chunk* chunk = queue_dequeue_multithreaded( context->queue_scanner, &context->mutex_scanner, &context->condition_not_empty_scanner, @@ -378,144 +1706,483 @@ static int load_files_multithreaded(void* pipeline_context) { context->loader_done = true; cnd_signal(&context->condition_not_empty_loader); mtx_unlock(&context->mutex_loader); + protocol_session_unbind(); return thrd_success; } if (!context->config->use_sendfile) { for (int i = 0; i < chunk->element_count; i++) { File* f = chunk->items[i]; - if (f->data->size > STREAM_THRESHOLD) - continue; - if (!file_load_data(f)) { - log_message(LOG_LEVEL_ERROR, "Failed to load file data, skipping"); - file_destroy(f); - chunk->items[i] = NULL; - } - } - } - queue_enqueue_multithreaded(context->queue_loader, chunk, &context->mutex_loader, - &context->condition_not_empty_loader, - &context->condition_not_full_loader); - } -} - -int send_files(Config* config) { - if (config->dry_run) - return send_dry_run_manifest(config); - - Client* client; - if (config->transport == TRANSPORT_SSH) { - if (config->use_sendfile) { - fprintf(stderr, "Error: -f/--sendfile is not supported with SSH transport\n"); - return 1; - } - client = - client_connect_ssh(config->ssh_destination, config->ssh_port, config->fastsync_server_path); - if (!client) - return 1; - } else if (config->use_tls) { - client = client_create(); - if (!client || !client_connect_tls(client, config->server_host, config->server_port, - config->tls_cert, config->tls_key, config->tls_ca)) { - if (client) - client_delete(client); - fprintf(stderr, "Error: could not connect to server via TLS\n"); - return 1; - } - } else { - client = client_create(); - if (!client || !client_connect(client, config->server_host, config->server_port)) { - if (client) - client_delete(client); - fprintf(stderr, "Error: could not connect to server\n"); - return 1; - } - } - if (!config_send(client->file_descriptor, config)) { - client_disconnect(client); - client_delete(client); - return 1; - } - DirectoryScanner* scanner = directory_scanner_create( - config->send_directory, config->use_metadata, config->chunk_size, config->exclude_patterns, - config->exclude_count, config->include_patterns, config->include_count, config->max_size, - config->min_size, config->max_depth, config->follow_symlinks, config->copy_links, - config->safe_links, config->copy_unsafe_links); - Chunk* current_chunk; - unsigned long long total_bytes = 0; - time_t last_progress = 0; - time_t start = time(NULL); - ArrayList* manifest = config->use_delete ? array_list_create(free) : NULL; - while ((current_chunk = directory_scanner_next(scanner)) != NULL) { - unsigned long long chunk_bytes = 0; - for (int i = 0; i < current_chunk->element_count; i++) { - chunk_bytes += current_chunk->items[i]->data->size; - if (manifest) { - const char* p = current_chunk->items[i]->path; - if (*p == '/') - p++; - array_list_add(manifest, str_dup(p)); - } - } - if (!config->use_sendfile) { - for (int i = 0; i < current_chunk->element_count; i++) { - File* f = current_chunk->items[i]; - if (f->data->size > STREAM_THRESHOLD) + if (f->data->size > STREAM_THRESHOLD && !context->config->use_compression) continue; if (!file_load_data(f)) { log_message(LOG_LEVEL_ERROR, "Failed to load file data"); - continue; + chunk_destroy(chunk); + pipeline_cancel(context); + protocol_session_unbind(); + return thrd_error; } } } - if (send_chunk(client, current_chunk, config) != 0) { - log_message(LOG_LEVEL_ERROR, "Failed to send chunk"); - chunk_destroy(current_chunk); + if (!queue_enqueue_multithreaded_cancel(context->queue_loader, chunk, &context->mutex_loader, + &context->condition_not_empty_loader, + &context->condition_not_full_loader, + &context->cancelled)) { + chunk_destroy(chunk); + pipeline_cancel(context); + protocol_session_unbind(); + return thrd_error; + } + } +} + +/* Print a one-line transfer progress report to stderr. `suffix` ends the + line (e.g. "Done.\n") or is "" for in-place refresh. Shared by the + single-threaded loop and the multithreaded progress thread. */ +static void print_transfer_progress(unsigned long long total_bytes, time_t start, + const char* suffix, bool human_readable) { + double elapsed = difftime(time(NULL), start); + double rate = elapsed > 0.0 ? total_bytes / (1048576.0 * elapsed) : 0.0; + if (human_readable) { + char total_buffer[32]; + char rate_buffer[32]; + fprintf(stderr, "\rSent %s (%s/s) %s", + display_bytes(total_bytes, true, total_buffer, sizeof(total_buffer)), + display_bytes((unsigned long long)(rate * 1048576.0), true, rate_buffer, + sizeof(rate_buffer)), + suffix); + } else { + fprintf(stderr, "\rSent %.1f MB (%.1f MB/s) %s", total_bytes / 1048576.0, rate, suffix); + } + fflush(stderr); +} + +/* Progress-reporting thread for multithreaded send. Runs in parallel with + the scanner/loader/sender threads and prints periodic progress to stderr. */ +static int progress_thread_fn(void* arg) { + PipelineContextSender* context = (PipelineContextSender*)arg; + time_t last_progress = 0; + time_t start = time(NULL); + + while (true) { + mtx_lock(&context->mutex_progress); + bool done = context->sender_done; + unsigned long long total = context->progress_bytes; + mtx_unlock(&context->mutex_progress); + + if (done) { + print_transfer_progress(total, start, "Done.\n", context->config->human_readable); break; } - if (config->show_progress) { - total_bytes += chunk_bytes; + + time_t now = time(NULL); + if (now - last_progress >= 1) { + last_progress = now; + print_transfer_progress(total, start, "", context->config->human_readable); + } + + struct timespec ts = {0, 100 * 1000000L}; /* 100 ms */ + thrd_sleep(&ts, NULL); + } + return thrd_success; +} + +/* Phase 6 residual-batch (client-only). --write-batch=FILE / --only-write-batch + * emit a self-contained single-file batch of a whole source tree from a + * deterministic separate scan pass. Each chunk's file images are fully loaded + * into memory (so chunk_serialize sees complete content, matching the -s wire + * codec byte-for-byte) and written to FILE as a length-prefixed record. The + * batch never crosses the wire and needs no server. Returns 0 on success. */ +int write_batch_from_source(const Config* config, const char* batch_path) { + if (!config || !batch_path || !config->send_directory) + return 1; + PreparedScanner prepared; + memset(&prepared, 0, sizeof(prepared)); + if (!prepare_scanner(config, 0, &prepared)) + return 1; + DirectoryScanner* scanner = + directory_scanner_create_with_options(config->send_directory, &prepared.options); + if (!scanner) { + prepared_scanner_destroy(&prepared); + return 1; + } + int fd = open(batch_path, O_WRONLY | O_CREAT | O_TRUNC, 0644); + if (fd < 0) { + log_perror("could not create batch file"); + directory_scanner_destroy(scanner); + prepared_scanner_destroy(&prepared); + return 1; + } + bool ok = batch_write_header(fd, config); + Chunk* chunk; + while (ok && (chunk = directory_scanner_next(scanner)) != NULL) { + for (int i = 0; i < chunk->element_count && ok; i++) { + File* f = chunk->items[i]; + if (f == NULL || f->data == NULL) + continue; + if (f->data->size > 0 && f->data->data == NULL && !file_load_data(f)) { + log_message(LOG_LEVEL_ERROR, "batch: failed to load data for %s", + f->path ? f->path : ""); + ok = false; + break; + } + } + if (ok) + ok = batch_write_chunk(fd, chunk); + chunk_destroy(chunk); + } + if (ok && directory_scanner_failed(scanner)) + ok = false; + if (directory_scanner_had_io_error(scanner)) + log_message(LOG_LEVEL_WARNING, "batch: source scan hit an unreadable directory"); + close(fd); + directory_scanner_destroy(scanner); + prepared_scanner_destroy(&prepared); + if (!ok && batch_path[0] != '\0') + unlink(batch_path); /* never leave a partial batch behind */ + return ok ? 0 : 1; +} + +/* Apply a batch FILE to DEST_ROOT (client-only, no server). Returns 0 on + * success; a malformed/truncated/oversized record or an apply error fails the + * whole apply. */ +int apply_batch_to_dest(const Config* config, const char* batch_path, const char* dest_root) { + if (!batch_path || !dest_root) + return 1; + int fd = open(batch_path, O_RDONLY); + if (fd < 0) { + log_perror("could not open batch file"); + return 1; + } + int rc = batch_read_apply(fd, config, dest_root); + close(fd); + return rc; +} + +int send_files(Config* config) { + if (config->list_only) + return send_list_only(config); + if (config->dry_run) + return send_dry_run_manifest(config); + ArrayList* missing_args = NULL; + int skipped = 0; + if (config->delete_missing_args) { + missing_args = array_list_create(free); + if (!missing_args) + return 1; + } + if (!files_from_list_check(config, missing_args, &skipped)) { + if (missing_args) + array_list_delete(missing_args); + return 1; + } + if (config_has_basis(config) && !basis_oversize_preflight(config)) { + if (missing_args) + array_list_delete(missing_args); + return 1; + } + + Client* client = connect_transfer_client(config); + if (!client) { + if (config->transport == TRANSPORT_TCP) + log_message(LOG_LEVEL_ERROR, "could not connect to server%s", + config->use_tls ? " via TLS" : ""); + return 1; + } + ProtocolSession session; + protocol_session_init(&session, client->file_descriptor, client->file_descriptor); + protocol_session_set_ssl(&session, (SSL*)client->ssl); + protocol_session_bind(&session); + int ret = 1; + DirectoryScanner* scanner = NULL; + ArrayList* manifest = NULL; + ArrayList* remove_sources = NULL; + /* P7 Wave D: captured source directory times, transmitted in trailing + STATUS_DIR_TIMES frame(s) (only when metadata rides the wire). */ + ArrayList* dir_entries = NULL; + /* Protected excluded prefixes (delete-excluded default protection). */ + ArrayList* excluded = NULL; + bool delete_early = config->use_delete && config_delete_timing_early(config); + bool send_failed = false; + bool had_scan_io = false; + PreparedScanner prepared; + memset(&prepared, 0, sizeof(prepared)); + if (!config_send(client->file_descriptor, config)) + goto send_fail; + receive_daemon_motd(client, config); + if (!prepare_scanner(config, 0, &prepared)) + goto send_fail; + if (config->use_metadata) { + dir_entries = array_list_create(file_destroy); + if (!dir_entries) + goto send_fail; + } + if (config->remove_source_files) + remove_sources = array_list_create(source_file_destroy); + if (config->remove_source_files && !remove_sources) + goto send_fail; + /* Unless --delete-excluded opts out, collect the paths the source scan prunes + by user-selection rules so the receiver protects their destination mirrors + from --delete (rsync's default). Only scans that build the keep-set get the + sink attached (prescan for early timing, the streaming data pass otherwise). */ + if (config->use_delete && !config->delete_excluded) { + excluded = array_list_create(free); + if (!excluded) + goto send_fail; + prepared.options.excluded_paths = excluded; + } + /* The late-timing modes (plain --delete / --delete-after / --delete-delay) + build the manifest while streaming and send it after the last data frame. + The early modes (--delete-before/--delete-during) send it up front from a + dedicated path-only pre-scan, so no manifest is kept during the data pass. */ + if (delete_early) { + /* Pass 1: collect the complete keep-set (paths only, no data loaded) and + transmit it now, before any file data. The receiver removes extras and + acks; the transfer aborts here if the deletion could not commit. */ + ArrayList* early_manifest = array_list_create(free); + if (!early_manifest) + goto send_fail; + bool prescan_ok = scan_paths_only(config, &prepared.options, early_manifest, &had_scan_io); + bool early_ok = false; + if (prescan_ok) { + /* A scan that hit an I/O error and produced NO keep entries is ambiguous + (the source may not be genuinely empty -- part of it was unreadable), + and an empty keep-set would delete the whole destination. Refuse to + delete; the genuine-empty-source case has no io_error and still sends + its (empty) keep-set. */ + if (had_scan_io && early_manifest->size == 0) { + log_message(LOG_LEVEL_ERROR, + "source scan hit an I/O error before finding any file; refusing to delete " + "with an empty keep-set (--delete)"); + prescan_ok = false; + } else { + early_ok = send_delete_manifest_early(client, early_manifest, excluded, missing_args); + } + } + array_list_delete(early_manifest); + /* The keep-set (and its protected prefixes) are already on the wire; the + data pass must not append to the exclusion list again. */ + prepared.options.excluded_paths = NULL; + if (!prescan_ok || !early_ok) + goto send_fail; + } else if (config->use_delete) { + manifest = array_list_create(free); + if (!manifest) + goto send_fail; + } + /* Phase 6: compute the client-only stop deadline once at transfer start. The + early-delete pre-scan above deliberately ignores it so the keep-set (and + its committed deletion) is always complete and correct. */ + struct timespec now_mono; + if (clock_gettime(CLOCK_MONOTONIC, &now_mono) != 0) { + now_mono.tv_sec = 0; + now_mono.tv_nsec = 0; + } + StopCondition stop = stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, + config->stop_at_set, config->stop_at, now_mono); + prepared.options.stop_condition = &stop; + /* The early-delete pre-scan above already ran; only the data pass should feed + the directory-time list (otherwise every directory would be captured + twice). */ + prepared.options.dir_entries = dir_entries; + scanner = directory_scanner_create_with_options(config->send_directory, &prepared.options); + if (!scanner) + goto send_fail; + + Chunk* current_chunk; + unsigned long long total_bytes = 0; + int total_files = 0; + time_t last_progress = 0; + time_t start = time(NULL); + /* True when the stop deadline cut the scan short so the keep-set manifest is + only a prefix of the source. */ + bool scan_stopped_early = false; + while ((current_chunk = directory_scanner_next(scanner)) != NULL) { + /* Phase 6: stop-elegantly at the next chunk boundary once the deadline has + passed. The scanner may also have stopped early itself; either way the + completion tail below keeps everything already sent. */ + if (stop_condition_reached(&stop)) { + chunk_destroy(current_chunk); + log_info_message(LOG_INFO_MISC, + "Stop deadline reached; stopping transfer at the next chunk boundary"); + scan_stopped_early = true; + break; + } + unsigned long long chunk_bytes = 0; + for (int i = 0; i < current_chunk->element_count; i++) { + chunk_bytes += current_chunk->items[i]->data->size; + total_files++; + } + if (manifest && !add_chunk_to_manifest(manifest, current_chunk)) { + chunk_destroy(current_chunk); + goto send_fail; + } + if (!config->use_sendfile) { + bool load_ok = true; + for (int i = 0; i < current_chunk->element_count; i++) { + File* f = current_chunk->items[i]; + if (f->data->size > STREAM_THRESHOLD && !config->use_compression) + continue; + if (!file_load_data(f)) { + log_message(LOG_LEVEL_ERROR, "Failed to load file data"); + load_ok = false; + break; + } + } + if (!load_ok) { + chunk_destroy(current_chunk); + goto send_fail; + } + } + if (send_chunk_with_removal(client, current_chunk, config, remove_sources) != 0) { + log_message(LOG_LEVEL_ERROR, "Failed to send chunk"); + chunk_destroy(current_chunk); + send_failed = true; + break; + } + total_bytes += chunk_bytes; + if (config->show_progress && !config->quiet) { time_t now = time(NULL); if (now - last_progress >= 1) { last_progress = now; - double elapsed = difftime(now, start); - double rate = elapsed > 0 ? total_bytes / (1048576.0 * elapsed) : 0; - fprintf(stderr, "\rSent %.1f MB (%.1f MB/s) ", total_bytes / 1048576.0, rate); - fflush(stderr); + print_transfer_progress(total_bytes, start, "", config->human_readable); } } chunk_destroy(current_chunk); } - if (config->use_delete) { - if (send_delete_manifest(client->file_descriptor, manifest) != 0) { + if (send_failed) { + if (manifest) { array_list_delete(manifest); + manifest = NULL; + } + goto send_fail; + } + if (directory_scanner_failed(scanner)) + goto send_fail; + if (directory_scanner_had_io_error(scanner)) + had_scan_io = true; + /* Phase 6: the scanner may have stopped early (returning NULL without a + failure) as soon as the deadline passed, so reflect that here too. A + deadline that cut the scan short leaves an incomplete keep-set; transmitting + it would make the receiver --delete the unscanned source mirrors (data + loss), so the late delete manifest is suppressed below. */ + scan_stopped_early = scan_stopped_early || stop_condition_reached(&stop); + if (scan_stopped_early) { + if (config->use_delete || config->delete_missing_args) + log_message(LOG_LEVEL_WARNING, + "transfer stopped early (stop deadline); skipping --delete keep-set so " + "unscanned source mirrors are not deleted"); + else + log_message(LOG_LEVEL_WARNING, "transfer stopped early (stop deadline)"); + } else { + if (had_scan_io && manifest && manifest->size == 0) { + /* A scan that hit an I/O error and produced no keep entries is ambiguous; + an empty keep-set would delete the whole destination. Refuse to delete + (see the early-timing comment above). */ + log_message(LOG_LEVEL_ERROR, + "source scan hit an I/O error before finding any file; refusing to delete with " + "an empty keep-set (--delete)"); goto send_fail; } - array_list_delete(manifest); + if ((manifest || config->delete_missing_args) && !delete_early) { + /* Late (commit) ordering: all file data is out; transmit the manifest so + the receiver commits the extras walk (--delete) and/or the + --delete-missing-args exact-path deletions only after the transfer + succeeds. In the early modes (--delete-before/--delete-during) the + manifest already went out up front, so nothing is re-sent here. */ + if (send_delete_manifest(client->file_descriptor, manifest, excluded, missing_args) != 0) { + if (manifest) { + array_list_delete(manifest); + manifest = NULL; + } + goto send_fail; + } + if (manifest) { + array_list_delete(manifest); + manifest = NULL; + } + } } - if (!send_status(client->file_descriptor, STATUS_FINISHED)) + /* P7 Wave D: every directory has now been traversed (or the scan stopped + early), so transmit the captured directory times last. The receiver defers + applying them until after its own deletion/publication phase. */ + if (!send_dir_times(client, config, dir_entries)) goto send_fail; - Status s; - int ok = receive_status(client->file_descriptor, &s) && s == STATUS_OK; - if (config->show_progress) { - double elapsed = difftime(time(NULL), start); - double rate = elapsed > 0 ? total_bytes / (1048576.0 * elapsed) : 0; - fprintf(stderr, "\rSent %.1f MB (%.1f MB/s) Done.\n", total_bytes / 1048576.0, rate); + bool ok = finalize_transfer(client, config, remove_sources); + if (!ok && config->use_delete) + log_message(LOG_LEVEL_ERROR, + "server reported a deletion failure (--delete); see the server log for the " + "reason (a --max-delete limit that the run would exceed deletes nothing)"); + if (ok) + remove_transferred_sources(config, remove_sources); + if (config->show_progress && !config->quiet) + print_transfer_progress(total_bytes, start, "Done.\n", config->human_readable); + if (config->stats && !config->quiet) { + double elapsed_total = difftime(time(NULL), start); + double rate = elapsed_total > 0 ? total_bytes / (1048576.0 * elapsed_total) : 0; + if (config->human_readable) { + char total_buffer[32]; + char rate_buffer[32]; + fprintf(stderr, "Stats: %d files, %s, %s/s\n", total_files, + display_bytes(total_bytes, true, total_buffer, sizeof(total_buffer)), + display_bytes((unsigned long long)(rate * 1048576.0), true, rate_buffer, + sizeof(rate_buffer))); + } else { + fprintf(stderr, "Stats: %d files, %.1f MB, %.1f MB/s\n", total_files, total_bytes / 1048576.0, + rate); + } } - directory_scanner_destroy(scanner); - client_disconnect(client); - client_delete(client); - return ok ? 0 : 1; + log_info_message(LOG_INFO_STATS, "Transfer summary: %d files, %.1f MB", total_files, + total_bytes / 1048576.0); + /* --ignore-errors: an unreadable source directory was skipped but the run + still completed (and deleted); report the run as errored like rsync does. */ + ret = (ok && !had_scan_io) ? 0 : 1; send_fail: - directory_scanner_destroy(scanner); - client_disconnect(client); - client_delete(client); - return 1; + /* Single cleanup path for all exits. The manifest is intentionally deleted + here even on success without --delete, fixing a pre-existing leak. */ + if (manifest) + array_list_delete(manifest); + if (excluded) + array_list_delete(excluded); + if (missing_args) + array_list_delete(missing_args); + if (remove_sources) + array_list_delete(remove_sources); + if (dir_entries) + array_list_delete(dir_entries); + if (scanner) + directory_scanner_destroy(scanner); + prepared_scanner_destroy(&prepared); + disconnect_transfer_client(client); + protocol_session_unbind(); + return ret; } -int send_files_multithreaded(Config* config) { +int send_files_multithreaded(Config** config_ptr) { + if (!config_ptr || !*config_ptr) + return 1; + Config* config = *config_ptr; + if (config->list_only) + return send_list_only(config); if (config->dry_run) return send_dry_run_manifest(config); + ArrayList* missing_args = NULL; + int skipped = 0; + if (config->delete_missing_args) { + missing_args = array_list_create(free); + if (!missing_args) + return 1; + } + if (!files_from_list_check(config, missing_args, &skipped)) { + if (missing_args) + array_list_delete(missing_args); + return 1; + } + if (config_has_basis(config) && !basis_oversize_preflight(config)) { + if (missing_args) + array_list_delete(missing_args); + return 1; + } long pages = sysconf(_SC_AVPHYS_PAGES); long page_size = sysconf(_SC_PAGE_SIZE); @@ -542,25 +2209,128 @@ int send_files_multithreaded(Config* config) { if (!context) { queue_destroy(q1); queue_destroy(q2); + if (missing_args) + array_list_delete(missing_args); return 1; } - if (config->use_delete) + context->missing_args = missing_args; + missing_args = NULL; /* owned by the context from here on */ + *config_ptr = NULL; /* context now owns config through all remaining paths */ + struct timespec now_mono; + if (clock_gettime(CLOCK_MONOTONIC, &now_mono) != 0) { + now_mono.tv_sec = 0; + now_mono.tv_nsec = 0; + } + context->stop_condition = + stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, config->stop_at_set, + config->stop_at, now_mono); + bool collect_excluded = config->use_delete && !config->delete_excluded; + if (config->use_delete) { context->manifest = array_list_create(free); - - thrd_t scanner, loader, sender; - if (thrd_create(&scanner, scan_directory_multithreaded, context) != thrd_success || - thrd_create(&loader, load_files_multithreaded, context) != thrd_success || - thrd_create(&sender, send_chunks_multithreaded, context) != thrd_success) { - perror("Error creating threads.\n"); + if (!context->manifest) { + pipeline_context_sender_destroy(context); + return 1; + } + if (collect_excluded) { + context->excluded_paths = array_list_create(free); + if (!context->excluded_paths) { + pipeline_context_sender_destroy(context); + return 1; + } + } + if (config_delete_timing_early(config)) { + /* --delete-before/--delete-during: build the complete keep-set manifest + (paths only, nothing loaded or sent) up front so the sender thread can + transmit it before the first data byte. The path-only pre-scan also + fills the protected excluded prefixes. */ + PreparedScanner prepared; + memset(&prepared, 0, sizeof(prepared)); + bool prepared_ok = prepare_scanner(config, 4, &prepared); + if (prepared_ok && context->excluded_paths) + prepared.options.excluded_paths = context->excluded_paths; + bool prebuilt = prepared_ok && scan_paths_only(config, &prepared.options, context->manifest, + &context->scan_had_io_error); + prepared_scanner_destroy(&prepared); + if (prebuilt && context->scan_had_io_error && context->manifest->size == 0) { + /* Empty keep-set + scan I/O error: refusing an empty keep-set manifest + would have deleted the whole destination (see send_files). */ + log_message(LOG_LEVEL_ERROR, + "source scan hit an I/O error before finding any file; refusing to delete " + "with an empty keep-set (--delete)"); + prebuilt = false; + } + if (!prebuilt) { + pipeline_context_sender_destroy(context); + return 1; + } + context->early_delete = true; + } + } + if (config->remove_source_files) + context->remove_source_files = array_list_create(source_file_destroy); + if ((config->use_delete && !context->manifest) || + (config->remove_source_files && !context->remove_source_files)) { pipeline_context_sender_destroy(context); return 1; } + thrd_t scanner, loader, sender; + bool scanner_created = false; + bool loader_created = false; + bool sender_created = false; + + scanner_created = (thrd_create(&scanner, scan_directory_multithreaded, context) == thrd_success); + if (scanner_created) + loader_created = (thrd_create(&loader, load_files_multithreaded, context) == thrd_success); + if (scanner_created && loader_created) + sender_created = (thrd_create(&sender, send_chunks_multithreaded, context) == thrd_success); + + if (!scanner_created || !loader_created || !sender_created) { + log_perror("Error creating threads"); + pipeline_cancel(context); + mtx_lock(&context->mutex_progress); + context->sender_done = true; + mtx_unlock(&context->mutex_progress); + if (sender_created) + thrd_join(sender, NULL); + if (loader_created) + thrd_join(loader, NULL); + if (scanner_created) + thrd_join(scanner, NULL); + pipeline_context_sender_destroy(context); + return 1; + } + + thrd_t progress; + bool progress_created = false; + if (config->show_progress && !config->quiet) { + progress_created = (thrd_create(&progress, progress_thread_fn, context) == thrd_success); + if (!progress_created) { + log_perror("Error creating progress thread"); + /* Non-fatal; continue without progress reporting */ + } + } + int sender_result; thrd_join(scanner, NULL); thrd_join(loader, NULL); thrd_join(sender, &sender_result); + if (progress_created) { + /* Signal progress thread to exit if it hasn't already */ + mtx_lock(&context->mutex_progress); + context->sender_done = true; + mtx_unlock(&context->mutex_progress); + thrd_join(progress, NULL); + } + + bool scan_io; + mtx_lock(&context->mutex_scanner); + scan_io = context->scan_had_io_error; + mtx_unlock(&context->mutex_scanner); + bool sender_ok = sender_result == thrd_success; + /* --ignore-errors: the run completed (and deleted) past an unreadable source + directory; report it as errored like rsync does. */ pipeline_context_sender_destroy(context); - return sender_result == thrd_success ? 0 : 1; + return sender_ok && !scan_io ? 0 : 1; } diff --git a/src/client/client_send.h b/src/client/client_send.h index e3c484f..0b12973 100644 --- a/src/client/client_send.h +++ b/src/client/client_send.h @@ -7,6 +7,10 @@ int send_chunk(Client* client, Chunk* chunk, Config* config); int send_files(Config* config); -int send_files_multithreaded(Config* config); +/* Takes ownership only when *config is set to NULL on return. */ +int send_files_multithreaded(Config** config); +/* Phase 6 residual-batch (client-only). See client_send.c. */ +int write_batch_from_source(const Config* config, const char* batch_path); +int apply_batch_to_dest(const Config* config, const char* batch_path, const char* dest_root); #endif diff --git a/src/client/client_validation.c b/src/client/client_validation.c new file mode 100644 index 0000000..f6a8ad0 --- /dev/null +++ b/src/client/client_validation.c @@ -0,0 +1,194 @@ +#include "client_validation.h" +#include "charset.h" +#include "delay_updates.h" +#include "log.h" +#include "usage.h" +#include "utils.h" +#include +#include + +/* Validate config after parsing. Returns true if valid. */ +bool validate_config(const Config* config) { + /* Phase 6 residual-batch modes relax the normal source+destination pair: the + batch driver is local and needs only what it consumes. --only-write-batch + emits a batch from the source (no destination, no server); + --read-batch applies a batch to the destination (no source, no server); + --write-batch runs the live transfer AND emits a batch, so it keeps the + full pair. */ + bool write_batch = config->write_batch != NULL; + bool only_write_batch = config->only_write_batch != NULL; + bool read_batch = config->read_batch != NULL; + if ((write_batch && only_write_batch) || (write_batch && read_batch) || + (only_write_batch && read_batch)) { + log_message(LOG_LEVEL_ERROR, + "--write-batch, --only-write-batch, and --read-batch are mutually exclusive"); + return false; + } + if (read_batch) { + if (!config->receive_root_directory) { + log_message(LOG_LEVEL_ERROR, "--read-batch requires a destination directory"); + print_usage(); + return false; + } + } else if (only_write_batch) { + if (!config->send_directory) { + log_message(LOG_LEVEL_ERROR, "--only-write-batch requires a source directory"); + print_usage(); + return false; + } + } else if (!config->send_directory || !config->receive_root_directory) { + log_message(LOG_LEVEL_ERROR, "source and destination directories are required"); + print_usage(); + return false; + } + if (config_has_basis(config) && config->use_chunk_serialization) { + log_message(LOG_LEVEL_ERROR, + "--compare-dest/--copy-dest/--link-dest require per-file incremental checks and " + "cannot be combined with -s (chunk serialization)"); + return false; + } + if (config->use_sendfile && (config->use_chunk_serialization || config->use_compression)) { + log_message(LOG_LEVEL_ERROR, "-f/--sendfile cannot be combined with -c (compression) or -s " + "(chunk serialization)"); + return false; + } + if (config->compression_threads > 0 && !config->use_compression) { + log_message(LOG_LEVEL_ERROR, "--compress-threads requires compression (-c or -z)"); + return false; + } + if (config->transport == TRANSPORT_SSH && config->use_sendfile) { + log_message(LOG_LEVEL_ERROR, "-f/--sendfile is not supported with SSH transport"); + return false; + } + if (config->use_incremental && config->use_chunk_serialization) { + log_message(LOG_LEVEL_ERROR, "--incremental is not supported with -s (chunk serialization)"); + return false; + } + /* -4 and -6 are mutually exclusive: a socket address family cannot be both. */ + if (config->ipv4 && config->ipv6) { + log_message(LOG_LEVEL_ERROR, "-4/--ipv4 and -6/--ipv6 are mutually exclusive"); + return false; + } + if (config->skip_compress_set && config->use_chunk_serialization) { + log_message(LOG_LEVEL_ERROR, + "--skip-compress cannot be combined with -s (chunk serialization)"); + return false; + } + if (config->use_delta && !config->whole_file && !config->use_incremental) { + log_message(LOG_LEVEL_ERROR, "--delta requires --incremental"); + return false; + } + if (config->use_delta && !config->whole_file && config->use_chunk_serialization) { + log_message(LOG_LEVEL_ERROR, "--delta cannot be combined with -s (chunk serialization)"); + return false; + } + if (config->use_delta && !config->whole_file && config->use_sendfile) { + log_message(LOG_LEVEL_ERROR, "--delta cannot be combined with -f (sendfile)"); + return false; + } + /* --append / --append-verify resume a shorter existing destination by + transmitting only the tail. The resume needs the per-file STATUS_CHECK + handshake (so the dest length is learned), which chunk serialization -s + disables; and whole-file is the opposite intent (send everything), so the + two would silently make the resume pointless. Both are rejected up front + rather than silently degrading to a full transfer. */ + if ((config->append || config->append_verify) && config->use_chunk_serialization) { + log_message(LOG_LEVEL_ERROR, + "--append/--append-verify require the per-file incremental check and cannot be " + "combined with -s (chunk serialization)"); + return false; + } + if ((config->append || config->append_verify) && config->whole_file) { + log_message(LOG_LEVEL_ERROR, + "--append/--append-verify are incompatible with --whole-file (which forces a " + "full transfer)"); + return false; + } + /* --hard-links/-H transmits each later group member as a dedicated per-file + STATUS_HARDLINK frame, which chunk serialization -s does not support; and a + hard-links sibling carries no payload, so the tail-resume of --append is + meaningless for it. Both combinations are rejected up front rather than + silently degrading. */ + if (config->preserve_hard_links && config->use_chunk_serialization) { + log_message(LOG_LEVEL_ERROR, + "--hard-links/-H cannot be combined with -s (chunk serialization)"); + return false; + } + /* -X/-A ride the per-file metadata frame; the buffer-based chunk-serialization + wire format does not carry the xattr block, so the pair is rejected up front + (mirroring -H + -s) rather than silently dropping attributes. */ + if ((config->preserve_xattrs || config->preserve_acls) && config->use_chunk_serialization) { + log_message(LOG_LEVEL_ERROR, + "--xattrs/-X and --acls/-A cannot be combined with -s (chunk serialization)"); + return false; + } + if (config->preserve_hard_links && (config->append || config->append_verify)) { + log_message(LOG_LEVEL_ERROR, + "--hard-links/-H cannot be combined with --append/--append-verify"); + return false; + } + if (config->log_file_format && !config->log_file) { + log_message(LOG_LEVEL_ERROR, "--log-file-format requires --log-file"); + return false; + } + if (config->use_tls) { + if (!config->tls_cert || !config->tls_key || !config->tls_ca) { + log_message(LOG_LEVEL_ERROR, "--tls requires --cert, --key, and --ca"); + return false; + } + } + /* Daemon credentials (A7, protocol 2.19.0): a --password-file would send the + username in the clear and derive a SCRAM proof a network sniffer could + attack offline, so it is only allowed over TLS (which itself mandates a + verified --cert/--key/--ca set above) or to a loopback destination. A + remote plaintext daemon is refused here, before any network I/O. */ + if (config->password_file && !config->use_tls && !utils_host_is_loopback(config->server_host)) { + log_message(LOG_LEVEL_ERROR, "sending daemon credentials to a non-local server requires --tls"); + return false; + } + if (config->delay_updates && config->inplace) { + log_message(LOG_LEVEL_ERROR, "--delay-updates does not work with --inplace"); + return false; + } + if (config->delay_updates && delay_updates_staging_name_conflict(config->backup_dir)) { + log_message(LOG_LEVEL_ERROR, + "--backup-dir is reserved when --delay-updates is active (used for the internal " + "staging directory)"); + return false; + } + if (!config_has_valid_delete_timing(config)) { + log_message(LOG_LEVEL_ERROR, + "--delete-before/--delete-during/--delete-delay/--delete-after select the delete " + "timing; at most one may be given and each implies --delete"); + return false; + } + /* --iconv: reject a malformed CONVERT_SPEC or an unsupported charset name at + startup (a probe iconv_open is attempted), so a typo'd charset never fails + the run mid-transfer with per-file errors. */ + if (!charset_spec_valid(config->iconv_spec)) { + log_message(LOG_LEVEL_ERROR, + "--iconv requires LOCAL[,REMOTE] charset names supported by iconv"); + return false; + } + /* --protocol: FastSync has exactly one wire format, so the forced version + must equal the current PROTOCOL_VERSION exactly. Rejected here, before any + network I/O, rather than letting the server hit its own mismatch check. */ + if (strcmp(config->version, PROTOCOL_VERSION) != 0) { + log_message(LOG_LEVEL_ERROR, + "--protocol must be %s (FastSync supports only its current wire " + "protocol version and cannot speak an older or virtual one)", + PROTOCOL_VERSION); + return false; + } + /* --copy-as pushes the source ids through the metadata path (it implies + --preserve). A later --no-preserve would clear use_metadata, leaving the + transfer with nothing to chown while the receiver gate would still pass. + Refuse the combination up front rather than silently chowning nothing. */ + if (config->copy_as_set && !config->use_metadata) { + log_message(LOG_LEVEL_ERROR, + "--copy-as requires metadata preservation and cannot be combined with " + "--no-preserve"); + return false; + } + return true; +} diff --git a/src/client/client_validation.h b/src/client/client_validation.h new file mode 100644 index 0000000..e830ccb --- /dev/null +++ b/src/client/client_validation.h @@ -0,0 +1,9 @@ +#ifndef CLIENT_VALIDATION_H +#define CLIENT_VALIDATION_H + +#include "config.h" +#include + +bool validate_config(const Config* config); + +#endif diff --git a/src/client/scanner.c b/src/client/scanner.c index 0bcd851..4911ccf 100644 --- a/src/client/scanner.c +++ b/src/client/scanner.c @@ -1,3 +1,4 @@ +#include "log.h" #include "scanner.h" #include "array_list.h" #include "chunk.h" @@ -9,14 +10,72 @@ #include #include #include +#include #include #include +#include + +#include "xattr.h" typedef struct { char* path; int depth; + FilterNode* context; /* inherited per-directory filter context */ } DirEntry; +/* A chain node: `own` holds the .rsync-filter rules of one directory, `parent` + * the context that directory inherited (nearest ancestor with a filter file). + * The chain for a directory's contents runs from that directory's own node up + * to the root; the command-line base rules are evaluated after the whole + * chain. */ +struct FilterNode { + FilterNode* parent; + FilterRuleList* own; +}; + +static void filter_node_destroy(void* item) { + if (item) { + FilterNode* node = (FilterNode*)item; + if (node->own) + filter_rule_list_free(node->own); + free(node); + } +} + +static FilterNode* filter_node_alloc(FilterNode* parent, FilterRuleList* own) { + FilterNode* node = malloc(sizeof(FilterNode)); + if (!node) + return NULL; + node->parent = parent; + node->own = own; + return node; +} + +/* Evaluate a rule chain for an entry inside the directory whose content + * context is `node`. rsync precedence, highest first: the innermost (current) + * directory's .rsync-filter rules, then each ancestor's, then the root's, and + * finally the command-line base rules (--filter/-C). A deeper per-directory + * file therefore overrides a shallower one, and per-directory files override + * the base rules by default. Returns FILTER_ACTION_NONE when nothing matched. */ +static FilterAction chain_rules_apply(const FilterRuleList* base, const FilterNode* node, + const char* rel, const char* leaf, bool is_dir) { + if (node) { + FilterAction own_action = filter_rules_apply(node->own, rel, leaf, is_dir); + if (own_action != FILTER_ACTION_NONE) + return own_action; + return chain_rules_apply(base, node->parent, rel, leaf, is_dir); + } + return base ? filter_rules_apply(base, rel, leaf, is_dir) : FILTER_ACTION_NONE; +} + +static bool entry_allowed(const FilterRuleList* base, const FilterNode* node, const char* rel, + const char* leaf, bool is_dir, bool per_dir_filters) { + /* -F: per-directory .rsync-filter files are never transferred. */ + if (per_dir_filters && !is_dir && strcmp(leaf, ".rsync-filter") == 0) + return false; + return chain_rules_apply(base, node, rel, leaf, is_dir) != FILTER_ACTION_EXCLUDE; +} + static void dir_entry_destroy(void* item) { if (item) { DirEntry* de = (DirEntry*)item; @@ -25,44 +84,489 @@ static void dir_entry_destroy(void* item) { } } -static DirEntry* dir_entry_create(const char* path, int depth) { +static DirEntry* dir_entry_create(const char* path, int depth, FilterNode* context) { DirEntry* de = malloc(sizeof(DirEntry)); - if (de) { - de->path = str_dup(path); - de->depth = depth; + if (!de) + return NULL; + de->path = str_dup(path); + if (!de->path) { + free(de); + return NULL; } + de->depth = depth; + de->context = context; return de; } +static bool safe_relative_link(const char* source_root, const char* containing_dir, + const char* link_target) { + char root[PATH_MAX]; + if (!realpath(source_root, root)) + return false; + char* joined = path_cat(containing_dir, link_target); + char resolved[PATH_MAX]; + bool safe = joined && realpath(joined, resolved) && strncmp(root, resolved, strlen(root)) == 0 && + (resolved[strlen(root)] == '\0' || resolved[strlen(root)] == '/'); + free(joined); + return safe; +} + +typedef struct { + char* path; + struct stat stats; + bool is_directory; + /* True when the entry should be carried through as a SYMLINK (is_symlink) + rather than a dereferenced file/directory. When true, `link_target` holds + the owned target string to transmit (sender-munged under --munge-links); + ownership transfers to the File built from this entry. */ + bool is_symlink; + char* link_target; + /* True when the entry was pruned by a user selection rule (--filter/-C/per-dir + rules, the --exclude/--include layer, or --max-size/--min-size) rather than + skipped for another reason (unreadable, symlink policy, not applicable). */ + bool excluded; +} ScannerEntry; + +/* --one-file-system (-x) decision. Only directories can carry a different + * device than their parent (mount points), so this is checked when a child + * directory is about to be descended into. */ +bool scanner_same_filesystem(bool one_file_system, dev_t root_device, dev_t entry_device) { + return !one_file_system || entry_device == root_device; +} + +/* Relative path of an on-disk path below `root`. The transfer root may be + * given with a trailing slash; the returned rel path never has one and is "" + * for the root itself. A root of "/" is handled (its children start at "/"). + * Exposed so tests can exercise the mapping directly. */ +char* scanner_path_relative(const char* root, const char* fs_path) { + size_t root_len = strlen(root); + while (root_len > 1 && root[root_len - 1] == '/') + root_len--; + if (strncmp(root, fs_path, root_len) != 0) + return NULL; + if (root_len == 1 && root[0] == '/') { + if (fs_path[1] == '\0') + return str_dup(""); + return str_dup(fs_path + 1); + } + if (fs_path[root_len] == '\0') + return str_dup(""); + if (fs_path[root_len] != '/') + return NULL; + return str_dup(fs_path + root_len + 1); +} + +/* Relative path of a child entry below the current directory. */ +static char* child_rel_path(const char* parent_rel, const char* name) { + if (!parent_rel || parent_rel[0] == '\0') + return str_dup(name); + return path_cat(parent_rel, name); +} + +/* Apply the --files-from allow-set and the filter layer to one entry. */ +static bool entry_passes_selection(const FileListSet* file_list, const FilterRuleList* base, + const FilterNode* node, const char* rel, const char* leaf, + bool is_dir, bool per_dir_filters) { + if (file_list && !file_list_affects(file_list, rel)) + return false; + if (base || per_dir_filters) + return entry_allowed(base, node, rel, leaf, is_dir, per_dir_filters); + return true; +} + +/* Best-effort capture of the file's whitelisted xattrs (-X/-A). A failure to + * read xattrs is non-fatal: the file is transferred without them. */ +static void scanner_capture_xattrs(const DirectoryScanner* scanner, File* file) { + if (!scanner || !file || !(scanner->preserve_xattrs || scanner->preserve_acls)) + return; + file->xattrs = xattr_capture_path(file->path); +} + +/* Apply --hard-links (-H) detection to one regular File. On a sibling (a + * later member of an already-seen source inode) the File keeps the group id + * and the first member's wire path but carries NO data payload (size 0); the + * first member is left untouched (data present, link_first). Allocation + * failure is fatal: the scanner is marked failed. */ +static void scanner_assign_hardlink(DirectoryScanner* scanner, HardLinkTable* table, File* file, + const struct stat* stats) { + if (!table || !file || !stats) + return; + int gid; + bool is_first; + char* first_path = NULL; + if (!hardlink_table_assign(table, file_wire_path(file), stats->st_dev, stats->st_ino, &gid, + &is_first, &first_path)) { + if (scanner) + scanner->failed = true; + return; + } + file->link_group = gid; + file->link_first = is_first; + if (!is_first) { + file->hardlink_target = first_path; + file->data->size = 0; + } else { + free(first_path); + } +} + +/* Phase 4 special/devices: detect a device (char/block), FIFO or socket entry + and, when the matching --devices/--specials flag asks it be preserved, + convert the File into a node to recreate (is_special, empty payload) with its + device rdev captured from the source stat. When the entry is not preserved + (or --copy-devices instead copies its content as an ordinary regular file) + the File is left as a normal data file. Returns true when converted. */ +static bool scanner_prepare_special(bool preserve_devices, bool preserve_specials, File* file, + const struct stat* stats) { + if (!file || !stats) + return false; + bool is_device = S_ISCHR(stats->st_mode) || S_ISBLK(stats->st_mode); + bool is_fifo = S_ISFIFO(stats->st_mode); + bool is_socket = S_ISSOCK(stats->st_mode); + if (!is_device && !is_fifo && !is_socket) + return false; + bool preserve = is_device ? preserve_devices : preserve_specials; + if (!preserve) + return false; + file->is_special = true; + file->data->size = 0; + file->data->data = NULL; + if (is_device) { + file->rdev_major = (int32_t)major(stats->st_rdev); + file->rdev_minor = (int32_t)minor(stats->st_rdev); + } + return true; +} + +/* Append `rel` to the caller's exclusion sink, taking `mtx` when shared across + parallel worker threads. Returns false on allocation failure (list left + unchanged). */ +static bool excluded_sink_append(ArrayList* list, mtx_t* mtx, const char* rel) { + if (!list) + return true; + char* dup = str_dup(rel); + if (!dup) + return false; + if (mtx) + mtx_lock(mtx); + bool ok = array_list_add(list, dup); + if (mtx) + mtx_unlock(mtx); + if (!ok) + free(dup); + return ok; +} + +/* Record one pruned-by-user-selection filesystem path in the scanner's + exclusion sink (see ScannerOptions.excluded_paths). The stored form is the + entry's wire/destination-relative path (a single leading '/' removed, exactly + how manifest keep entries are stored), so the receiver's walker prefixes + match the destination layout. An allocation failure is a fatal scan error. */ +static void scanner_record_excluded(DirectoryScanner* scanner, const char* fs_path) { + if (!scanner->excluded_paths || !fs_path) + return; + const char* rel = *fs_path == '/' ? fs_path + 1 : fs_path; + if (!excluded_sink_append(scanner->excluded_paths, scanner->excluded_mutex, rel)) + scanner->failed = true; +} + +/* Merge the open directory's own .rsync-filter rules into the inherited + * context, returning the context used for this directory's entries. On a parse + * error the scanner is marked failed. Returns 0 on success, -1 on failure. */ +static int open_directory_filter_context(DirectoryScanner* scanner, const FilterNode* inherited) { + if (!scanner->per_dir_filters) { + scanner->current_node = (FilterNode*)inherited; + return 0; + } + char err[256]; + bool exists = false; + FilterRuleList* own = + filter_file_read(scanner->current_path, scanner->current_rel ? scanner->current_rel : "", + &exists, err, sizeof(err)); + if (!own) { + log_message(LOG_LEVEL_ERROR, "invalid .rsync-filter in %s: %s", scanner->current_path, err); + scanner->failed = true; + return -1; + } + if (exists && own->count > 0) { + FilterNode* node = filter_node_alloc((FilterNode*)inherited, own); + if (!node || !array_list_add(scanner->filter_nodes, node)) { + filter_node_destroy(node); + scanner->failed = true; + return -1; + } + scanner->current_node = node; + } else { + filter_rule_list_free(own); + scanner->current_node = (FilterNode*)inherited; + } + return 0; +} + +/* Inspect symlinks, resolve the entry type, and apply file filters once for both scanners. */ +static int scanner_inspect_entry(const ScannerOptions* options, const char* source_root, + const char* containing_dir, const char* name, + ScannerEntry* entry) { + entry->excluded = false; + entry->is_symlink = false; + entry->link_target = NULL; + entry->path = path_cat(containing_dir, name); + if (!entry->path) + return -1; + + struct stat link_stats; + if (lstat(entry->path, &link_stats) != 0) { + free(entry->path); + return 0; + } + bool is_symlink = S_ISLNK(link_stats.st_mode); + if (!is_symlink) + goto regular; + + /* Symlink: choose between dereferencing (---copy-links / --safe-links / + --copy-unsafe-links, plus -k for symlinks-to-directories) and carrying the + link through as a symlink (-l, and -k for symlinks-to-files). No link + option means the symlink is skipped entirely (pre-existing behavior). */ + const bool any_link_option = options->follow_symlinks || options->copy_links || + options->safe_links || options->copy_unsafe_links || + options->copy_dirlinks; + if (!any_link_option) + goto skip; + + char link_target[4096]; + ssize_t length = readlink(entry->path, link_target, sizeof(link_target) - 1); + if (length < 0) + goto skip; + link_target[length] = '\0'; + + if (options->safe_links) { + if (link_target[0] == '/' || !safe_relative_link(source_root, containing_dir, link_target)) + goto skip; + } + if (options->copy_unsafe_links && !options->copy_links) { + if (link_target[0] != '/') + goto skip; + } + + bool emit_symlink = false; + if (options->copy_links) { + emit_symlink = false; /* --copy-links dereferences every referent */ + } else if (options->safe_links || options->copy_unsafe_links) { + emit_symlink = false; /* preserve pre-existing dereference behavior */ + } else if (options->copy_dirlinks) { + struct stat ref; + if (stat(entry->path, &ref) == 0 && S_ISDIR(ref.st_mode)) + emit_symlink = false; /* -k: symlink to a directory recurses as a dir */ + else + emit_symlink = true; /* -k: symlink to a file stays a symlink */ + } else if (options->follow_symlinks) { + emit_symlink = true; /* -l: copy symlink as symlink */ + } + + if (!emit_symlink) { + if (stat(entry->path, &entry->stats) != 0) + goto skip; + entry->is_directory = S_ISDIR(entry->stats.st_mode); + if (entry->is_directory) + return 1; + goto apply_filters; + } + + /* Carry the link as a symlink. --munge-links containment: a target that + could escape the receive root (absolute or containing "..") is never + transmitted -- the entry is merely skipped ("contained"). */ + if (link_target[0] == '\0' || + (options->munge_links && !file_symlink_target_contained(link_target))) + goto skip; + entry->is_symlink = true; + entry->stats = link_stats; + entry->is_directory = false; + entry->link_target = + options->munge_links ? file_symlink_munge(link_target) : str_dup(link_target); + if (!entry->link_target) + goto skip; + goto apply_filters; + +regular: + if (stat(entry->path, &entry->stats) != 0) + goto skip; + entry->is_directory = S_ISDIR(entry->stats.st_mode); + if (entry->is_directory) + return 1; + +apply_filters: + for (int i = 0; i < options->exclude_count; i++) + if (glob_match(options->exclude_patterns[i], name)) { + entry->excluded = true; + goto skip; + } + if (options->include_count > 0) { + bool included = false; + for (int i = 0; i < options->include_count; i++) + if (glob_match(options->include_patterns[i], name)) + included = true; + if (!included) { + entry->excluded = true; + goto skip; + } + } + if ((options->max_size > 0 && (unsigned long long)entry->stats.st_size > options->max_size) || + (options->min_size > 0 && (unsigned long long)entry->stats.st_size < options->min_size)) { + entry->excluded = true; + goto skip; + } + return 1; + +skip: + free(entry->path); + entry->path = NULL; + free(entry->link_target); + entry->link_target = NULL; + return 0; +} + +DirectoryScanner* directory_scanner_create_with_options(const char* root_directory, + const ScannerOptions* options) { + if (!root_directory || !options) + return NULL; + DirectoryScanner* scanner = calloc(1, sizeof(DirectoryScanner)); + if (scanner == NULL) + return NULL; + scanner->directories = queue_create(100, dir_entry_destroy); + if (!scanner->directories) { + free(scanner); + return NULL; + } + scanner->current_dir = NULL; + scanner->current_path = NULL; + scanner->use_metadata = options->use_metadata; + scanner->preserve_atimes = options->preserve_atimes; + scanner->preserve_crtimes = options->preserve_crtimes; + scanner->preserve_xattrs = options->preserve_xattrs; + scanner->preserve_acls = options->preserve_acls; + scanner->chunk_size = options->chunk_size > 0 ? options->chunk_size : DESIRED_CHUNK_SIZE; + scanner->exclude_patterns = options->exclude_patterns; + scanner->exclude_count = options->exclude_count; + scanner->include_patterns = options->include_patterns; + scanner->include_count = options->include_count; + scanner->max_size = options->max_size; + scanner->min_size = options->min_size; + scanner->max_depth = options->max_depth; + scanner->current_depth = 0; + scanner->follow_symlinks = options->follow_symlinks; + scanner->copy_links = options->copy_links; + scanner->safe_links = options->safe_links; + scanner->copy_unsafe_links = options->copy_unsafe_links; + scanner->copy_dirlinks = options->copy_dirlinks; + scanner->munge_links = options->munge_links; + scanner->checksum = options->checksum; + scanner->one_file_system = options->one_file_system; + scanner->preserve_devices = options->preserve_devices; + scanner->preserve_specials = options->preserve_specials; + scanner->copy_devices = options->copy_devices; + scanner->failed = false; + scanner->root_path = str_dup(root_directory); + if (!scanner->root_path) { + queue_destroy(scanner->directories); + free(scanner); + return NULL; + } + scanner->current_rel = NULL; + scanner->at_seed_dir = true; + scanner->seed_node = NULL; + scanner->current_node = NULL; + scanner->file_list = options->file_list; + scanner->base_filters = options->base_filters; + scanner->per_dir_filters = options->per_dir_filters; + scanner->excluded_paths = options->excluded_paths; + scanner->excluded_mutex = options->excluded_mutex; + scanner->ignore_io_errors = options->ignore_io_errors; + scanner->ignore_missing_args = options->ignore_missing_args; + scanner->io_error = false; + scanner->dirs_mode = options->dirs; + scanner->relative_mode = options->relative && options->file_list != NULL; + scanner->hardlinks = options->hardlinks; + scanner->prune_empty_dirs = options->prune_empty_dirs; + scanner->stop_condition = options->stop_condition; + scanner->capture_dir_times = options->capture_dir_times; + scanner->dir_entries = options->dir_entries; + scanner->dir_entries_mutex = options->dir_entries_mutex; + scanner->dirs_root_emitted = false; + scanner->list_index = 0; + scanner->dirs_batch = NULL; + scanner->dirs_batch_size = 0; + scanner->filter_nodes = NULL; + if (scanner->base_filters || scanner->per_dir_filters) { + scanner->filter_nodes = array_list_create(filter_node_destroy); + if (!scanner->filter_nodes) { + free(scanner->root_path); + queue_destroy(scanner->directories); + free(scanner); + return NULL; + } + } + if (scanner->one_file_system) { + struct stat root_stats; + if (stat(root_directory, &root_stats) != 0) { + log_perror("Could not stat source directory"); + free(scanner->root_path); + queue_destroy(scanner->directories); + array_list_delete(scanner->filter_nodes); + free(scanner); + return NULL; + } + scanner->root_dev = root_stats.st_dev; + } + DirEntry* root = dir_entry_create(root_directory, 0, NULL); + if (!root) { + free(scanner->root_path); + queue_destroy(scanner->directories); + array_list_delete(scanner->filter_nodes); + free(scanner); + return NULL; + } + if (!queue_enqueue(scanner->directories, root)) { + dir_entry_destroy(root); + free(scanner->root_path); + queue_destroy(scanner->directories); + array_list_delete(scanner->filter_nodes); + free(scanner); + return NULL; + } + return scanner; +} + DirectoryScanner* directory_scanner_create(const char* root_directory, bool use_metadata, unsigned long long chunk_size, char** exclude_patterns, int exclude_count, char** include_patterns, int include_count, unsigned long long max_size, unsigned long long min_size, int max_depth, bool follow_symlinks, bool copy_links, bool safe_links, - bool copy_unsafe_links) { - DirectoryScanner* scanner = malloc(sizeof(DirectoryScanner)); - if (scanner == NULL) - return NULL; - scanner->directories = queue_create(100, dir_entry_destroy); - scanner->current_dir = NULL; - scanner->current_path = NULL; - scanner->use_metadata = use_metadata; - scanner->chunk_size = chunk_size > 0 ? chunk_size : DESIRED_CHUNK_SIZE; - scanner->exclude_patterns = exclude_patterns; - scanner->exclude_count = exclude_count; - scanner->include_patterns = include_patterns; - scanner->include_count = include_count; - scanner->max_size = max_size; - scanner->min_size = min_size; - scanner->max_depth = max_depth; - scanner->current_depth = 0; - scanner->follow_symlinks = follow_symlinks; - scanner->copy_links = copy_links; - scanner->safe_links = safe_links; - scanner->copy_unsafe_links = copy_unsafe_links; - queue_enqueue(scanner->directories, dir_entry_create(root_directory, 0)); - return scanner; + bool copy_unsafe_links, bool checksum) { + ScannerOptions options = { + .use_metadata = use_metadata, + .chunk_size = chunk_size, + .exclude_patterns = exclude_patterns, + .exclude_count = exclude_count, + .include_patterns = include_patterns, + .include_count = include_count, + .max_size = max_size, + .min_size = min_size, + .max_depth = max_depth, + .num_threads = 0, + .follow_symlinks = follow_symlinks, + .copy_links = copy_links, + .safe_links = safe_links, + .copy_unsafe_links = copy_unsafe_links, + .checksum = checksum, + .one_file_system = false, + .file_list = NULL, + .base_filters = NULL, + .per_dir_filters = false, + .dirs = false, + .relative = false, + }; + return directory_scanner_create_with_options(root_directory, &options); } void directory_scanner_destroy(DirectoryScanner* scanner) { @@ -73,57 +577,406 @@ void directory_scanner_destroy(DirectoryScanner* scanner) { scanner->current_dir = NULL; } free(scanner->current_path); + free(scanner->current_rel); + free(scanner->root_path); + array_list_delete(scanner->filter_nodes); + array_list_delete(scanner->dirs_batch); queue_destroy(scanner->directories); free(scanner); } static Chunk* chunk_data_to_chunk(ArrayList* chunk_data) { void** chunk_items = array_list_to_array(chunk_data); + if (!chunk_items) + return NULL; Chunk* chunk = chunk_create((File**)chunk_items, chunk_data->size); free(chunk_items); + if (!chunk) + return NULL; chunk_data->item_destroyer = NULL; array_list_delete(chunk_data); return chunk; } +/* P7 Wave D: append one traversed source directory's captured metadata to the + * shared pending-directory-time list. The File carries no payload; only the + * wire path (absolute fs path normally, the bare relative path under + * -R + --files-from) and its metadata are used, and the sender transmits them + * in trailing STATUS_DIR_TIMES frame(s). `mutex` (optional) serializes the + * append for the parallel scanner's shared workers. An unstattable or + * non-directory path is silently skipped (the transfer is unaffected); an + * allocation failure is fatal and reported to the caller. */ +static bool scanner_capture_dir_time(ArrayList* dir_entries, mtx_t* mutex, const char* root_path, + const char* fs_path, bool relative_mode, bool preserve_atimes, + bool preserve_crtimes) { + if (!dir_entries || !root_path || !fs_path) + return true; + struct stat st; + if (stat(fs_path, &st) != 0 || !S_ISDIR(st.st_mode)) + return true; + char* rel = scanner_path_relative(root_path, fs_path); + if (!rel) + return true; + if (relative_mode && rel[0] == '\0') { + /* -R + --files-from: the transfer root itself has no bare relative wire + path (matches the -R scan, which never emits the root). */ + free(rel); + return true; + } + File* file = file_create(fs_path); + if (!file) { + free(rel); + return false; + } + file->is_dir = true; + file->metadata = file_metadata_create(fs_path, &st, preserve_atimes, preserve_crtimes); + if (!file->metadata) { + free(rel); + file_destroy(file); + return false; + } + if (relative_mode) { + file->send_path = rel; + rel = NULL; + } + free(rel); + bool added; + if (mutex) { + mtx_lock(mutex); + added = array_list_add(dir_entries, file); + mtx_unlock(mutex); + } else { + added = array_list_add(dir_entries, file); + } + if (!added) { + file_destroy(file); + return false; + } + return true; +} + +/* Open the next queued directory and set up its filter context. Returns 1 when + a directory is open, 0 when the queue is exhausted, and -1 on a fatal error. + A directory that cannot be opened is an I/O error: it is recorded on the + scanner and, when --ignore-errors is active, skipped so the rest of the tree + is still scanned (the caller decides whether to treat the recorded error as + fatal). */ static int open_next_directory(DirectoryScanner* scanner) { if (scanner->current_dir) { closedir(scanner->current_dir); scanner->current_dir = NULL; } free(scanner->current_path); + scanner->current_path = NULL; - if (queue_is_empty(scanner->directories)) - return 0; + while (!queue_is_empty(scanner->directories)) { + DirEntry* de = (DirEntry*)queue_dequeue(scanner->directories); + scanner->current_path = de->path; + scanner->current_depth = de->depth; + /* The seed directory inherits the scanner's configured context (the root + * .rsync-filter context in parallel mode); other dirs inherit the context of + * the directory that enqueued them. */ + const FilterNode* inherited = scanner->at_seed_dir ? scanner->seed_node : de->context; + scanner->at_seed_dir = false; + free(de); - DirEntry* de = (DirEntry*)queue_dequeue(scanner->directories); - scanner->current_path = de->path; - scanner->current_depth = de->depth; - free(de); - scanner->current_dir = opendir(scanner->current_path); - if (scanner->current_dir == NULL) { - perror("Could not open directory"); - free(scanner->current_path); - scanner->current_path = NULL; - return -1; + free(scanner->current_rel); + scanner->current_rel = scanner_path_relative(scanner->root_path, scanner->current_path); + if (!scanner->current_rel) { + log_message(LOG_LEVEL_ERROR, "Could not compute relative path under %s", scanner->root_path); + scanner->failed = true; + free(scanner->current_path); + scanner->current_path = NULL; + return -1; + } + + scanner->current_dir = opendir(scanner->current_path); + if (scanner->current_dir == NULL) { + scanner->io_error = true; + log_perror("Could not open directory"); + /* The transfer ROOT (a sequential scanner's seed directory) must be + readable even under --ignore-errors: an unreadable root would produce + an empty scan whose keep-set would delete the whole destination. Only + subdirectories discovered during an otherwise-successful root scan are + skippable. (The parallel scanner never reaches this for the root: its + root open failure aborts scanner creation; worker seeds are assigned + subdirectories with a non-empty relative path and stay skippable.) */ + bool is_root_seed = scanner->current_rel != NULL && scanner->current_rel[0] == '\0' && + scanner->current_depth == 0; + free(scanner->current_rel); + scanner->current_rel = NULL; + free(scanner->current_path); + scanner->current_path = NULL; + if (!scanner->ignore_io_errors || is_root_seed) { + scanner->failed = true; + return -1; + } + /* --ignore-errors: record the I/O error and keep scanning the rest. */ + continue; + } + if (open_directory_filter_context(scanner, inherited) != 0) { + closedir(scanner->current_dir); + scanner->current_dir = NULL; + free(scanner->current_path); + scanner->current_path = NULL; + return -1; + } + if (scanner->capture_dir_times && + !scanner_capture_dir_time(scanner->dir_entries, scanner->dir_entries_mutex, + scanner->root_path, scanner->current_path, scanner->relative_mode, + scanner->preserve_atimes, scanner->preserve_crtimes)) { + closedir(scanner->current_dir); + scanner->current_dir = NULL; + free(scanner->current_path); + scanner->current_path = NULL; + scanner->failed = true; + return -1; + } + return 1; } - return 1; + return 0; +} + +/* ---- --dirs mode ---- + With -d the scanner transfers directory entries and never recurses into + contents. A plain `-d ` sends only the source-root directory mirror + (created empty at the destination). With -d + --files-from exactly the + listed items are sent: listed directories become empty directory entries and + listed regular files are transferred as files; nothing else is scanned, so + no descent into a listed directory can happen. */ + +/* Directory entries carry no payload, so the dirs generator also bounds every + chunk by element count; chunk_deserialize refuses more than this many files + per chunk (see MAX_FILES_PER_CHUNK in chunk.c). */ +#define DIRS_CHUNK_MAX_FILES 65536U + +/* Build the File for the transfer root directory itself (the `-d ` and + * "." cases). */ +static File* dirs_root_dir_file(DirectoryScanner* scanner) { + struct stat st; + if (stat(scanner->root_path, &st) != 0 || !S_ISDIR(st.st_mode)) { + log_perror("Could not stat source directory"); + scanner->failed = true; + return NULL; + } + File* file = file_create(scanner->root_path); + if (!file) { + scanner->failed = true; + return NULL; + } + file->is_dir = true; + if (scanner->use_metadata) { + file->metadata = file_metadata_create(scanner->root_path, &st, scanner->preserve_atimes, + scanner->preserve_crtimes); + if (!file->metadata) { + file_destroy(file); + scanner->failed = true; + return NULL; + } + } + return file; +} + +/* Map one normalized --files-from entry to a File (a directory entry or a + * regular file to transfer), or NULL to skip the entry. */ +static File* dirs_file_for_entry(DirectoryScanner* scanner, const char* entry) { + if (entry[0] == '\0') { + /* "." (whole tree): under -R the bare receive root is the destination and + there is nothing to create for the root itself; otherwise mirror the + source-root directory (empty). */ + if (scanner->relative_mode) + return NULL; + return dirs_root_dir_file(scanner); + } + char* abs_path = path_cat(scanner->root_path, entry); + if (!abs_path) { + scanner->failed = true; + return NULL; + } + struct stat link_stats; + if (lstat(abs_path, &link_stats) != 0) { + /* --ignore-missing-args (implied by --delete-missing-args): an explicitly + listed entry that does not exist under the source is a preflight-detected + missing argument and is skipped here, exactly as the recursive scan skips + nothing (missing entries never appear there). Without the flags it stays + a hard pre-transfer error. */ + if (scanner->ignore_missing_args) { + log_info_message(LOG_INFO_MISC, "skipping missing --files-from entry '%s'", entry); + free(abs_path); + return NULL; + } + log_message(LOG_LEVEL_ERROR, "--dirs listed entry is not present under the source: %s", entry); + free(abs_path); + scanner->failed = true; + return NULL; + } + struct stat effective = link_stats; + if (S_ISLNK(link_stats.st_mode)) { + /* A symlink is transferred (following its referent) only when a link + resolution option is active, mirroring the regular scanner. */ + bool resolve = scanner->follow_symlinks || scanner->copy_links || scanner->safe_links || + scanner->copy_unsafe_links; + if (!resolve || stat(abs_path, &effective) != 0) { + free(abs_path); + return NULL; + } + } + bool is_dir = S_ISDIR(effective.st_mode); + bool is_file = S_ISREG(effective.st_mode); + if (!is_dir && !is_file) { + free(abs_path); + return NULL; + } + File* file = file_create(abs_path); + free(abs_path); + if (!file) { + scanner->failed = true; + return NULL; + } + file->is_dir = is_dir; + file->data->size = is_file ? (unsigned long long)effective.st_size : 0; + if (scanner->relative_mode) { + file->send_path = str_dup(entry); + if (!file->send_path) { + file_destroy(file); + scanner->failed = true; + return NULL; + } + } + if (scanner->use_metadata) { + file->metadata = file_metadata_create(file->path, &effective, scanner->preserve_atimes, + scanner->preserve_crtimes); + if (!file->metadata) { + file_destroy(file); + scanner->failed = true; + return NULL; + } + } + return file; +} + +/* True when the directory contains no entries at all (ignoring "." and ".."). + An unreadable directory is reported as non-empty so the regular (erroring) + root-entry path runs instead of silently transferring nothing. */ +static bool dirs_source_dir_is_empty(const char* path) { + DIR* dir = opendir(path); + if (!dir) + return false; + bool empty = true; + const struct dirent* entry; + while ((entry = readdir(dir)) != NULL) { + if (strcmp(entry->d_name, ".") != 0 && strcmp(entry->d_name, "..") != 0) { + empty = false; + break; + } + } + closedir(dir); + return empty; +} + +/* The next File from the --dirs generator, or NULL when exhausted. */ +static File* dirs_next_file(DirectoryScanner* scanner) { + if (!scanner->file_list) { + if (scanner->dirs_root_emitted) + return NULL; + scanner->dirs_root_emitted = true; + /* --prune-empty-dirs: a physically empty source directory's explicit entry + would only create an empty destination directory, so it is omitted. */ + if (scanner->prune_empty_dirs && dirs_source_dir_is_empty(scanner->root_path)) + return NULL; + return dirs_root_dir_file(scanner); + } + while (scanner->list_index < scanner->file_list->count) { + const char* entry = scanner->file_list->entries[scanner->list_index++]; + File* file = dirs_file_for_entry(scanner, entry); + if (scanner->failed) + return NULL; + if (file) + return file; + } + return NULL; +} + +static Chunk* dirs_flush_batch(DirectoryScanner* scanner) { + if (!scanner->dirs_batch || scanner->dirs_batch->size == 0) { + array_list_delete(scanner->dirs_batch); + scanner->dirs_batch = NULL; + scanner->dirs_batch_size = 0; + return NULL; + } + ArrayList* batch = scanner->dirs_batch; + scanner->dirs_batch = NULL; + scanner->dirs_batch_size = 0; + Chunk* chunk = chunk_data_to_chunk(batch); + if (!chunk) + scanner->failed = true; + return chunk; +} + +static Chunk* directory_scanner_next_dirs(DirectoryScanner* scanner) { + while (scanner->dirs_batch == NULL || scanner->dirs_batch_size <= scanner->chunk_size) { + if (scanner->stop_condition && stop_condition_reached(scanner->stop_condition)) { + Chunk* leftover = dirs_flush_batch(scanner); + if (leftover) + chunk_destroy(leftover); + return NULL; + } + if (!scanner->dirs_batch) { + scanner->dirs_batch = array_list_create(file_destroy); + if (!scanner->dirs_batch) { + scanner->failed = true; + return NULL; + } + scanner->dirs_batch_size = 0; + } + File* file = dirs_next_file(scanner); + if (scanner->failed) { + dirs_flush_batch(scanner); + return NULL; + } + if (!file) { + return dirs_flush_batch(scanner); + } + if (!array_list_add(scanner->dirs_batch, file)) { + file_destroy(file); + scanner->failed = true; + dirs_flush_batch(scanner); + return NULL; + } + scanner->dirs_batch_size += file->data ? file->data->size : 0; + /* Empty directory entries carry no bytes, so a large --dirs --files-from + list must also be bounded by element count (the chunk deserializer caps + the number of files per chunk). */ + if (scanner->dirs_batch->size >= (int)DIRS_CHUNK_MAX_FILES) + return dirs_flush_batch(scanner); + } + return dirs_flush_batch(scanner); } Chunk* directory_scanner_next(DirectoryScanner* scanner) { + if (scanner && scanner->dirs_mode) + return directory_scanner_next_dirs(scanner); ArrayList* chunk_data = array_list_create(file_destroy); + if (!chunk_data) { + scanner->failed = true; + return NULL; + } unsigned long long chunk_data_size = 0; while (1) { + if (scanner->stop_condition && stop_condition_reached(scanner->stop_condition)) { + array_list_delete(chunk_data); + return NULL; + } if (scanner->current_dir == NULL) { int ret = open_next_directory(scanner); if (ret == 0) break; if (ret < 0) - continue; + break; } - struct dirent* entry = readdir(scanner->current_dir); + const struct dirent* entry = readdir(scanner->current_dir); if (entry == NULL) { closedir(scanner->current_dir); scanner->current_dir = NULL; @@ -135,357 +988,691 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) continue; - char* cur_path = path_cat(scanner->current_path, entry->d_name); - struct stat stats; - struct stat lstats; - bool is_symlink = false; - if (lstat(cur_path, &lstats) != 0) { - free(cur_path); + ScannerOptions options = { + .use_metadata = scanner->use_metadata, + .chunk_size = scanner->chunk_size, + .exclude_patterns = scanner->exclude_patterns, + .exclude_count = scanner->exclude_count, + .include_patterns = scanner->include_patterns, + .include_count = scanner->include_count, + .max_size = scanner->max_size, + .min_size = scanner->min_size, + .max_depth = scanner->max_depth, + .num_threads = 0, + .follow_symlinks = scanner->follow_symlinks, + .copy_links = scanner->copy_links, + .safe_links = scanner->safe_links, + .copy_unsafe_links = scanner->copy_unsafe_links, + .copy_dirlinks = scanner->copy_dirlinks, + .munge_links = scanner->munge_links, + .checksum = scanner->checksum, + .one_file_system = scanner->one_file_system, + .file_list = scanner->file_list, + .base_filters = scanner->base_filters, + .per_dir_filters = scanner->per_dir_filters, + .dirs = false, + .relative = false, + }; + ScannerEntry inspected; + int inspection = scanner_inspect_entry(&options, scanner->current_path, scanner->current_path, + entry->d_name, &inspected); + if (inspection < 0) { + scanner->failed = true; + break; + } + if (inspection == 0) { + /* The entry was pruned by a user selection rule (exclude/include/size) or + skipped for another reason; only the user-selection prunes protect the + corresponding destination mirror from --delete. */ + if (inspected.excluded) { + char* abs_path = path_cat(scanner->current_path, entry->d_name); + if (!abs_path) { + scanner->failed = true; + break; + } + scanner_record_excluded(scanner, abs_path); + free(abs_path); + } continue; } - is_symlink = S_ISLNK(lstats.st_mode); + char* cur_path = inspected.path; + struct stat stats = inspected.stats; - if (is_symlink && !scanner->follow_symlinks && !scanner->copy_links && !scanner->safe_links && - !scanner->copy_unsafe_links) { + /* --files-from allow-set and the filter layer apply to files and to + * directories (an excluded directory is not descended into). */ + bool is_dir = inspected.is_directory; + char* rel = child_rel_path(scanner->current_rel, entry->d_name); + if (!rel) { + free(cur_path); + scanner->failed = true; + break; + } + bool passes_selection = + entry_passes_selection(scanner->file_list, scanner->base_filters, scanner->current_node, + rel, entry->d_name, is_dir, scanner->per_dir_filters); + if (!passes_selection) { + /* --files-from subset pruning is not a filter exclusion: its delete + semantics stay keep-set-only (an unlisted source path is treated as + absent, so its destination mirror is a deletable extra). A rule-based + exclusion is recorded as a protected prefix. -R + --files-from bare + wire paths are never recorded (see ScannerOptions.excluded_paths). */ + bool files_from_prune = scanner->file_list && !file_list_affects(scanner->file_list, rel); + if (!files_from_prune && !scanner->relative_mode) + scanner_record_excluded(scanner, cur_path); + } + /* With -R + --files-from the wire/destination path is the entry's bare + relative path; keep `rel` alive to attach it to a transferred file. */ + char* rel_copy = scanner->relative_mode ? str_dup(rel) : NULL; + free(rel); + if (rel_copy == NULL && scanner->relative_mode) { + free(cur_path); + scanner->failed = true; + break; + } + if (!passes_selection) { + free(rel_copy); free(cur_path); continue; } - if (is_symlink && scanner->safe_links) { - char link_target[4096]; - ssize_t len = readlink(cur_path, link_target, sizeof(link_target) - 1); - if (len < 0) { + if (is_dir) { + free(rel_copy); + if (!scanner_same_filesystem(scanner->one_file_system, scanner->root_dev, stats.st_dev)) { free(cur_path); continue; } - link_target[len] = '\0'; - if (link_target[0] == '/') { - free(cur_path); - continue; - } - } - - if (is_symlink && scanner->copy_unsafe_links && !scanner->copy_links) { - char link_target[4096]; - ssize_t len = readlink(cur_path, link_target, sizeof(link_target) - 1); - if (len < 0) { - free(cur_path); - continue; - } - link_target[len] = '\0'; - bool unsafe = (link_target[0] == '/'); - if (!unsafe) { - free(cur_path); - continue; - } - } - - bool use_lstat = is_symlink && scanner->follow_symlinks && !scanner->copy_links; - if (use_lstat) { - stats = lstats; - } else { - if (stat(cur_path, &stats) != 0) { - free(cur_path); - continue; - } - } - - if (S_ISDIR(stats.st_mode)) { int next_depth = scanner->current_depth + 1; if (scanner->max_depth <= 0 || next_depth < scanner->max_depth) { - DirEntry* de = dir_entry_create(cur_path, next_depth); - if (!queue_enqueue(scanner->directories, de)) + DirEntry* de = dir_entry_create(cur_path, next_depth, scanner->current_node); + if (!de || !queue_enqueue(scanner->directories, de)) { dir_entry_destroy(de); + scanner->failed = true; + } } free(cur_path); } else { if (scanner->max_depth > 0 && scanner->current_depth + 1 > scanner->max_depth) { + free(rel_copy); free(cur_path); continue; } - bool excluded = false; - for (int i = 0; i < scanner->exclude_count; i++) { - if (glob_match(scanner->exclude_patterns[i], entry->d_name)) { - excluded = true; - break; - } - } - if (excluded) { - free(cur_path); - continue; - } - - if (scanner->include_count > 0) { - bool included = false; - for (int i = 0; i < scanner->include_count; i++) { - if (glob_match(scanner->include_patterns[i], entry->d_name)) { - included = true; - break; - } - } - if (!included) { - free(cur_path); - continue; - } - } - - if ((scanner->max_size > 0 && (unsigned long long)stats.st_size > scanner->max_size) || - (scanner->min_size > 0 && (unsigned long long)stats.st_size < scanner->min_size)) { - free(cur_path); - continue; - } - File* file = file_create(cur_path); + free(cur_path); if (file == NULL) { - free(cur_path); + free(rel_copy); + free(inspected.link_target); + inspected.link_target = NULL; + scanner->failed = true; continue; } - file->data->size = stats.st_size; + if (inspected.is_symlink) { + file->is_symlink = true; + file->symlink_target = inspected.link_target; + inspected.link_target = NULL; + } else { + file->data->size = stats.st_size; + } + if (scanner->relative_mode) { + file->send_path = rel_copy; + rel_copy = NULL; + } + /* --devices/--specials: a device/FIFO/socket entry marked for preservation + becomes a node to recreate (is_special, no data, rdev captured). */ + scanner_prepare_special(scanner->preserve_devices, scanner->preserve_specials, file, &stats); + if (scanner->hardlinks && S_ISREG(stats.st_mode)) + scanner_assign_hardlink(scanner, scanner->hardlinks, file, &stats); if (scanner->use_metadata) - file->metadata = file_metadata_create(&stats); - array_list_add(chunk_data, file); + file->metadata = file_metadata_create(file->path, &stats, scanner->preserve_atimes, + scanner->preserve_crtimes); + if (scanner->use_metadata && !file->metadata) { + free(rel_copy); + file_destroy(file); + scanner->failed = true; + break; + } + if (!(file->link_group != 0 && !file->link_first)) + scanner_capture_xattrs(scanner, file); + if (!array_list_add(chunk_data, file)) { + free(rel_copy); + file_destroy(file); + scanner->failed = true; + break; + } chunk_data_size += file->data->size; if (chunk_data_size > scanner->chunk_size) { - free(cur_path); - return chunk_data_to_chunk(chunk_data); + free(rel_copy); + Chunk* result = chunk_data_to_chunk(chunk_data); + if (!result) + scanner->failed = true; + return result; } - free(cur_path); + free(rel_copy); } } - if (chunk_data->size > 0) - return chunk_data_to_chunk(chunk_data); + if (chunk_data->size > 0) { + Chunk* result = chunk_data_to_chunk(chunk_data); + if (!result) + scanner->failed = true; + return result; + } array_list_delete(chunk_data); return NULL; } +bool directory_scanner_failed(const DirectoryScanner* scanner) { + return scanner == NULL || scanner->failed; +} + +bool directory_scanner_had_io_error(const DirectoryScanner* scanner) { + return scanner != NULL && scanner->io_error; +} + typedef struct { ParallelScanner* ps; char** dirs; int dir_count; - bool use_metadata; - unsigned long long chunk_size; - char** exclude_patterns; - int exclude_count; - char** include_patterns; - int include_count; - unsigned long long max_size; - unsigned long long min_size; - int max_depth; - bool follow_symlinks; - bool copy_links; - bool safe_links; - bool copy_unsafe_links; + char* root_dir; /* the transfer root, for relative-path computation */ + ScannerOptions options; + ProtocolSession* allocation_session; } ParallelWorkerArg; static int parallel_worker_thread(void* arg) { ParallelWorkerArg* wa = (ParallelWorkerArg*)arg; + ProtocolSession* allocation_session = wa->allocation_session; + if (allocation_session) + protocol_session_bind(allocation_session); for (int i = 0; i < wa->dir_count; i++) { - DirectoryScanner* ds = directory_scanner_create( - wa->dirs[i], wa->use_metadata, wa->chunk_size, wa->exclude_patterns, wa->exclude_count, - wa->include_patterns, wa->include_count, wa->max_size, wa->min_size, wa->max_depth, - wa->follow_symlinks, wa->copy_links, wa->safe_links, wa->copy_unsafe_links); + DirectoryScanner* ds = directory_scanner_create_with_options(wa->dirs[i], &wa->options); + if (!ds) { + mtx_lock(&wa->ps->result_mutex); + wa->ps->failed = true; + atomic_store(&wa->ps->cancelled, true); + cnd_broadcast(&wa->ps->result_not_empty); + cnd_broadcast(&wa->ps->result_not_full); + mtx_unlock(&wa->ps->result_mutex); + for (int j = i; j < wa->dir_count; j++) + free(wa->dirs[j]); + break; + } + /* Root .rsync-filter rules (parsed by the parallel scanner) apply to the + * contents of every assigned subdirectory. Relative paths (used by the + * allow-set and per-directory rules) are computed against the transfer + * root, not the subdirectory the worker is seeded with. Exclusion + * recording shares one caller-owned list across the workers. */ + free(ds->root_path); + ds->root_path = str_dup(wa->root_dir); + ds->seed_node = wa->ps->root_filter_node; + ds->excluded_mutex = &wa->ps->result_mutex; Chunk* chunk; while ((chunk = directory_scanner_next(ds)) != NULL) { - queue_enqueue_multithreaded(wa->ps->result_queue, chunk, &wa->ps->result_mutex, - &wa->ps->result_not_empty, &wa->ps->result_not_full); + if (!queue_enqueue_multithreaded_cancel(wa->ps->result_queue, chunk, &wa->ps->result_mutex, + &wa->ps->result_not_empty, &wa->ps->result_not_full, + &wa->ps->cancelled)) { + chunk_destroy(chunk); + break; + } + } + if (directory_scanner_failed(ds)) { + mtx_lock(&wa->ps->result_mutex); + wa->ps->failed = true; + atomic_store(&wa->ps->cancelled, true); + cnd_broadcast(&wa->ps->result_not_empty); + cnd_broadcast(&wa->ps->result_not_full); + mtx_unlock(&wa->ps->result_mutex); + } else if (directory_scanner_had_io_error(ds)) { + /* --ignore-errors path: an unreadable directory was skipped, not fatal. */ + mtx_lock(&wa->ps->result_mutex); + wa->ps->io_error = true; + mtx_unlock(&wa->ps->result_mutex); } directory_scanner_destroy(ds); free(wa->dirs[i]); } ParallelScanner* ps = wa->ps; + free(wa->root_dir); free(wa->dirs); free(wa); mtx_lock(&ps->result_mutex); ps->completed++; - if (ps->completed >= ps->num_threads) { + if (ps->completed >= ps->expected_threads) { ps->done = true; cnd_signal(&ps->result_not_empty); } mtx_unlock(&ps->result_mutex); + if (allocation_session) + protocol_session_unbind(); return thrd_success; } -ParallelScanner* parallel_scanner_create(char* root_directory, bool use_metadata, - unsigned long long chunk_size, char** exclude_patterns, - int exclude_count, char** include_patterns, - int include_count, unsigned long long max_size, - unsigned long long min_size, int max_depth, - int num_threads, bool follow_symlinks, bool copy_links, - bool safe_links, bool copy_unsafe_links) { +static void parallel_scanner_creation_failed(ParallelScanner* ps) { + mtx_lock(&ps->result_mutex); + ps->failed = true; + atomic_store(&ps->cancelled, true); + ps->expected_threads = ps->created_threads; + if (ps->completed >= ps->expected_threads) + ps->done = true; + cnd_broadcast(&ps->result_not_empty); + cnd_broadcast(&ps->result_not_full); + mtx_unlock(&ps->result_mutex); +} + +/* Initialize result queue and synchronization primitives. Returns true on success. */ +static bool parallel_scanner_init(ParallelScanner* ps) { + ps->result_queue = queue_create(100, chunk_destroy); + if (!ps->result_queue) + return false; + atomic_init(&ps->cancelled, false); + int init = 0; + bool ok = true; + if (mtx_init(&ps->result_mutex, mtx_plain) != thrd_success) + ok = false; + if (ok) { + init++; + if (cnd_init(&ps->result_not_empty) != thrd_success) + ok = false; + } + if (ok) { + // cppcheck-suppress unreadVariable + init++; + if (cnd_init(&ps->result_not_full) != thrd_success) + ok = false; + } + if (!ok) { + if (init >= 3) + cnd_destroy(&ps->result_not_full); + if (init >= 2) + cnd_destroy(&ps->result_not_empty); + if (init >= 1) + mtx_destroy(&ps->result_mutex); + queue_destroy(ps->result_queue); + ps->result_queue = NULL; + return false; + } + return true; +} + +/* Split files into chunks of roughly chunk_size bytes. Returns the first chunk (also stored + * chunks beyond the first are enqueued on `queue`). Nulls out consumed entries in `files`. + * Sets *failed on allocation/enqueue errors. */ +static Chunk* batch_files(ArrayList* files, unsigned long long chunk_size, Queue* queue, + bool* failed) { + Chunk* first = NULL; + if (files->size <= 0) + return NULL; + ArrayList* batch = array_list_create(NULL); + if (!batch) { + *failed = true; + return NULL; + } + unsigned long long batch_size = 0; + for (int i = 0; i < files->size; i++) { + File* f = (File*)files->items[i]; + if (!array_list_add(batch, f)) { + *failed = true; + break; + } + batch_size += f->data->size; + if (batch_size >= chunk_size || i == files->size - 1) { + void** items = array_list_to_array(batch); + if (!items) { + *failed = true; + array_list_delete(batch); + batch = NULL; + break; + } + Chunk* c = chunk_create((File**)items, batch->size); + free(items); + if (!c) { + *failed = true; + array_list_delete(batch); + batch = NULL; + break; + } + int batch_start = i - batch->size + 1; + for (int j = batch_start; j <= i; j++) + files->items[j] = NULL; + batch->item_destroyer = NULL; + array_list_delete(batch); + batch = NULL; + if (!first) { + first = c; + } else { + if (!queue_enqueue(queue, c)) { + chunk_destroy(c); + *failed = true; + } + } + if (i < files->size - 1) { + batch = array_list_create(NULL); + if (!batch) { + *failed = true; + break; + } + batch_size = 0; + } + } + } + if (batch) { + batch->item_destroyer = NULL; + array_list_delete(batch); + } + return first; +} + +/* Scan one root-directory entry into either the subdirs or files list. */ +static void scan_root_entry(const ScannerOptions* options, const FilterNode* root_node, + const char* root_directory, const struct dirent* entry, + ArrayList* root_files, ArrayList* subdirs, dev_t root_dev, + ParallelScanner* ps) { + ScannerEntry inspected; + int inspection = + scanner_inspect_entry(options, root_directory, root_directory, entry->d_name, &inspected); + if (inspection < 0) { + ps->failed = true; + return; + } + if (inspection == 0) { + if (inspected.excluded && options->excluded_paths) { + /* A root-level user-selection prune protects the destination mirror of + the same-named wire path (at the root the bare name is the wire path in + every layout). */ + char* abs_path = path_cat(root_directory, entry->d_name); + if (!abs_path) { + ps->failed = true; + } else { + const char* rel = *abs_path == '/' ? abs_path + 1 : abs_path; + if (!excluded_sink_append(options->excluded_paths, options->excluded_mutex, rel)) + ps->failed = true; + free(abs_path); + } + } + return; + } + char* cur_path = inspected.path; + struct stat st = inspected.stats; + bool is_dir = inspected.is_directory; + char* rel = str_dup(entry->d_name); + if (!rel) { + free(cur_path); + ps->failed = true; + return; + } + bool passes = entry_passes_selection(options->file_list, options->base_filters, root_node, rel, + entry->d_name, is_dir, options->per_dir_filters); + /* -R + --files-from: root-level files keep their bare relative send path. */ + bool use_rel = options->relative && options->file_list != NULL; + if (!passes) { + /* --files-from subset pruning is not a filter exclusion; -R bare-wire-path + exclusions are never recorded (see ScannerOptions.excluded_paths). */ + bool files_from_prune = options->file_list && !file_list_affects(options->file_list, rel); + if (!files_from_prune && !use_rel && options->excluded_paths) { + const char* rel_path = *cur_path == '/' ? cur_path + 1 : cur_path; + if (!excluded_sink_append(options->excluded_paths, options->excluded_mutex, rel_path)) + ps->failed = true; + } + free(rel); + free(cur_path); + return; + } + if (is_dir) { + free(rel); + if (!scanner_same_filesystem(options->one_file_system, root_dev, st.st_dev)) { + free(cur_path); + return; + } + if (!array_list_add(subdirs, cur_path)) { + free(cur_path); + ps->failed = true; + } + return; + } + File* file = file_create(cur_path); + free(cur_path); + if (!file) { + free(rel); + free(inspected.link_target); + inspected.link_target = NULL; + ps->failed = true; + return; + } + if (inspected.is_symlink) { + file->is_symlink = true; + file->symlink_target = inspected.link_target; + inspected.link_target = NULL; + } else { + file->data->size = st.st_size; + } + if (use_rel) { + file->send_path = rel; + rel = NULL; + } + scanner_prepare_special(options->preserve_devices, options->preserve_specials, file, &st); + if (options->hardlinks && S_ISREG(st.st_mode)) { + int gid; + bool is_first; + char* first_path = NULL; + if (!hardlink_table_assign((HardLinkTable*)options->hardlinks, file_wire_path(file), st.st_dev, + st.st_ino, &gid, &is_first, &first_path)) { + ps->failed = true; + } else { + file->link_group = gid; + file->link_first = is_first; + if (!is_first) { + file->hardlink_target = first_path; + file->data->size = 0; + } else { + free(first_path); + } + } + } + if (options->use_metadata) + file->metadata = + file_metadata_create(file->path, &st, options->preserve_atimes, options->preserve_crtimes); + if (options->use_metadata && !file->metadata) { + free(rel); + file_destroy(file); + ps->failed = true; + return; + } + if ((options->preserve_xattrs || options->preserve_acls) && + !(file->link_group != 0 && !file->link_first)) + file->xattrs = xattr_capture_path(file->path); + if (!array_list_add(root_files, file)) { + free(rel); + file_destroy(file); + ps->failed = true; + return; + } + free(rel); +} + +/* Scan the root directory itself, collecting root files and subdirectories. + * Returns false if the root directory could not be opened. */ +static bool scan_root_directory(ParallelScanner* ps, const char* root_directory, + const ScannerOptions* options, const FilterNode* root_node, + dev_t root_dev, ArrayList* root_files, ArrayList* subdirs) { + DIR* dir = opendir(root_directory); + if (!dir) { + log_perror("Could not open root directory for parallel scan"); + return false; + } + const struct dirent* entry; + while ((entry = readdir(dir)) != NULL) { + if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) + continue; + scan_root_entry(options, root_node, root_directory, entry, root_files, subdirs, root_dev, ps); + } + closedir(dir); + return true; +} + +/* Spawn worker threads, one per group of subdirectories. */ +static void spawn_parallel_workers(ParallelScanner* ps, ArrayList* subdirs, + const ScannerOptions* options, const char* root_directory, + unsigned long long cs) { + if (subdirs->size <= 0) + return; + int n = options->num_threads > 0 ? options->num_threads : 4; + if (n > subdirs->size) + n = subdirs->size; + + ps->num_threads = n; + ps->expected_threads = n; + ps->threads = calloc(n, sizeof(thrd_t)); + if (!ps->threads) { + ps->num_threads = 0; + ps->expected_threads = 0; + ps->failed = true; + return; + } + int dirs_per_thread = subdirs->size / n; + int remainder = subdirs->size % n; + int start = 0; + ps->num_threads = 0; + for (int t = 0; t < n; t++) { + int count = dirs_per_thread + (t < remainder ? 1 : 0); + if (count == 0) + break; + ParallelWorkerArg* wa = calloc(1, sizeof(ParallelWorkerArg)); + if (!wa) { + parallel_scanner_creation_failed(ps); + break; + } + wa->ps = ps; + wa->dirs = calloc(count, sizeof(char*)); + wa->root_dir = str_dup(root_directory); + if (!wa->dirs || !wa->root_dir) { + free(wa->root_dir); + free(wa->dirs); + free(wa); + parallel_scanner_creation_failed(ps); + break; + } + bool dup_ok = true; + for (int j = 0; j < count; j++) { + wa->dirs[j] = str_dup((char*)subdirs->items[start + j]); + if (!wa->dirs[j]) + dup_ok = false; + } + if (!dup_ok) { + for (int j = 0; j < count; j++) + free(wa->dirs[j]); + free(wa->root_dir); + free(wa->dirs); + free(wa); + parallel_scanner_creation_failed(ps); + break; + } + wa->dir_count = count; + wa->options = *options; + wa->options.chunk_size = cs; + wa->allocation_session = ps->allocation_session; + start += count; + if (thrd_create(&ps->threads[t], parallel_worker_thread, wa) != thrd_success) { + for (int j = 0; j < count; j++) + free(wa->dirs[j]); + free(wa->root_dir); + free(wa->dirs); + free(wa); + parallel_scanner_creation_failed(ps); + break; + } + ps->num_threads++; + ps->created_threads++; + } +} + +ParallelScanner* parallel_scanner_create_with_options(const char* root_directory, + const ScannerOptions* options, + ProtocolSession* allocation_session) { + if (!root_directory || !options) + return NULL; ParallelScanner* ps = calloc(1, sizeof(ParallelScanner)); if (!ps) return NULL; - ps->result_queue = queue_create(100, chunk_destroy); - if (!ps->result_queue) { - free(ps); - return NULL; - } - if (mtx_init(&ps->result_mutex, mtx_plain) != thrd_success || - cnd_init(&ps->result_not_empty) != thrd_success || - cnd_init(&ps->result_not_full) != thrd_success) { - queue_destroy(ps->result_queue); + if (!parallel_scanner_init(ps)) { free(ps); return NULL; } + ps->allocation_session = allocation_session; - DIR* dir = opendir(root_directory); - if (!dir) { - perror("Could not open root directory for parallel scan"); + ArrayList* root_files = array_list_create(file_destroy); + ArrayList* subdirs = array_list_create(free); + if (!root_files || !subdirs) { + array_list_delete(root_files); + array_list_delete(subdirs); parallel_scanner_destroy(ps); return NULL; } - ArrayList* root_files = array_list_create(file_destroy); - ArrayList* subdirs = array_list_create(free); - struct dirent* entry; - while ((entry = readdir(dir)) != NULL) { - if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) - continue; - char* cur_path = path_cat(root_directory, entry->d_name); - if (!cur_path) - continue; - struct stat st; - if (stat(cur_path, &st) != 0) { - free(cur_path); - continue; - } - if (S_ISDIR(st.st_mode)) { - array_list_add(subdirs, cur_path); - } else { - bool excluded = false; - for (int i = 0; i < exclude_count; i++) { - if (glob_match(exclude_patterns[i], entry->d_name)) { - excluded = true; - break; - } - } - if (excluded) { - free(cur_path); - continue; - } - if (include_count > 0) { - bool included = false; - for (int i = 0; i < include_count; i++) { - if (glob_match(include_patterns[i], entry->d_name)) { - included = true; - break; - } - } - if (!included) { - free(cur_path); - continue; - } - } - if ((max_size > 0 && (unsigned long long)st.st_size > max_size) || - (min_size > 0 && (unsigned long long)st.st_size < min_size)) { - free(cur_path); - continue; - } - File* file = file_create(cur_path); - free(cur_path); - if (!file) - continue; - file->data->size = st.st_size; - if (use_metadata) - file->metadata = file_metadata_create(&st); - array_list_add(root_files, file); - } - } - closedir(dir); - - unsigned long long cs = chunk_size > 0 ? chunk_size : DESIRED_CHUNK_SIZE; - if (root_files->size > 0) { - ArrayList* batch = array_list_create(NULL); - unsigned long long batch_size = 0; - Chunk* first = NULL; - for (int i = 0; i < root_files->size; i++) { - File* f = (File*)root_files->items[i]; - array_list_add(batch, f); - batch_size += f->data->size; - if (batch_size >= cs || i == root_files->size - 1) { - void** items = array_list_to_array(batch); - Chunk* c = chunk_create((File**)items, batch->size); - free(items); - batch->item_destroyer = NULL; - array_list_delete(batch); - batch = NULL; - if (!first) { - first = c; - } else { - queue_enqueue_multithreaded(ps->result_queue, c, &ps->result_mutex, &ps->result_not_empty, - &ps->result_not_full); - } - if (i < root_files->size - 1) { - batch = array_list_create(NULL); - batch_size = 0; - } - } - } - if (batch) { - batch->item_destroyer = NULL; - array_list_delete(batch); - } - ps->initial_chunk = first; - root_files->item_destroyer = NULL; - } - array_list_delete(root_files); - - int n = num_threads > 0 ? num_threads : 4; - if (n > subdirs->size) - n = subdirs->size > 0 ? subdirs->size : 1; - - if (subdirs->size > 0) { - ps->num_threads = n; - ps->threads = calloc(n, sizeof(thrd_t)); - if (!ps->threads) { + dev_t root_dev = 0; + if (options->one_file_system) { + struct stat root_stats; + if (stat(root_directory, &root_stats) != 0) { + log_perror("Could not stat source directory"); + array_list_delete(root_files); array_list_delete(subdirs); parallel_scanner_destroy(ps); return NULL; } - int dirs_per_thread = subdirs->size / n; - int remainder = subdirs->size % n; - int start = 0; - for (int t = 0; t < n; t++) { - int count = dirs_per_thread + (t < remainder ? 1 : 0); - if (count == 0) - break; - ParallelWorkerArg* wa = calloc(1, sizeof(ParallelWorkerArg)); - if (!wa) - break; - wa->ps = ps; - wa->dirs = calloc(count, sizeof(char*)); - if (!wa->dirs) { - free(wa); - break; - } - for (int j = 0; j < count; j++) - wa->dirs[j] = str_dup((char*)subdirs->items[start + j]); - wa->dir_count = count; - wa->use_metadata = use_metadata; - wa->chunk_size = cs; - wa->exclude_patterns = exclude_patterns; - wa->exclude_count = exclude_count; - wa->include_patterns = include_patterns; - wa->include_count = include_count; - wa->max_size = max_size; - wa->min_size = min_size; - wa->max_depth = max_depth; - wa->follow_symlinks = follow_symlinks; - wa->copy_links = copy_links; - wa->safe_links = safe_links; - wa->copy_unsafe_links = copy_unsafe_links; - start += count; - if (thrd_create(&ps->threads[t], parallel_worker_thread, wa) != thrd_success) { - for (int j = 0; j < count; j++) - free(wa->dirs[j]); - free(wa->dirs); - free(wa); - ps->num_threads = t; - break; + root_dev = root_stats.st_dev; + } + + /* Build the root directory's .rsync-filter context once; workers seed their + * scanners with it so per-dir rules behave identically to the sequential + * scanner. */ + FilterNode* root_node = NULL; + if (options->per_dir_filters) { + char err[256]; + bool exists = false; + FilterRuleList* own = filter_file_read(root_directory, "", &exists, err, sizeof(err)); + if (!own) { + log_message(LOG_LEVEL_ERROR, "invalid .rsync-filter in %s: %s", root_directory, err); + array_list_delete(root_files); + array_list_delete(subdirs); + parallel_scanner_destroy(ps); + return NULL; + } + if (exists && own->count > 0) { + root_node = filter_node_alloc(NULL, own); + if (!root_node) { + filter_rule_list_free(own); + array_list_delete(root_files); + array_list_delete(subdirs); + parallel_scanner_destroy(ps); + return NULL; } + } else { + filter_rule_list_free(own); } } + ps->root_filter_node = root_node; + + if (!scan_root_directory(ps, root_directory, options, root_node, root_dev, root_files, subdirs)) { + array_list_delete(root_files); + array_list_delete(subdirs); + parallel_scanner_destroy(ps); + return NULL; + } + /* P7 Wave D: the parallel scanner never runs a DirectoryScanner over the + transfer root itself (it hands the root's immediate subdirectories to + workers), so capture the root's directory time here. */ + if (options->capture_dir_times && + !scanner_capture_dir_time(options->dir_entries, options->dir_entries_mutex, root_directory, + root_directory, options->relative && options->file_list != NULL, + options->preserve_atimes, options->preserve_crtimes)) { + array_list_delete(root_files); + array_list_delete(subdirs); + parallel_scanner_destroy(ps); + return NULL; + } + + unsigned long long cs = options->chunk_size > 0 ? options->chunk_size : DESIRED_CHUNK_SIZE; + ps->initial_chunk = batch_files(root_files, cs, ps->result_queue, &ps->failed); + array_list_delete(root_files); + + spawn_parallel_workers(ps, subdirs, options, root_directory, cs); array_list_delete(subdirs); return ps; } @@ -497,7 +1684,14 @@ Chunk* parallel_scanner_next(ParallelScanner* ps) { return c; } if (ps->num_threads == 0) { + mtx_lock(&ps->result_mutex); + if (!queue_is_empty(ps->result_queue)) { + Chunk* chunk = queue_dequeue(ps->result_queue); + mtx_unlock(&ps->result_mutex); + return chunk; + } ps->done = true; + mtx_unlock(&ps->result_mutex); return NULL; } Chunk* chunk = queue_dequeue_multithreaded( @@ -505,14 +1699,28 @@ Chunk* parallel_scanner_next(ParallelScanner* ps) { return chunk; } +bool parallel_scanner_failed(const ParallelScanner* ps) { + return ps == NULL || ps->failed; +} + +bool parallel_scanner_had_io_error(const ParallelScanner* ps) { + return ps != NULL && ps->io_error; +} + void parallel_scanner_destroy(ParallelScanner* ps) { if (!ps) return; + mtx_lock(&ps->result_mutex); ps->done = true; - cnd_signal(&ps->result_not_empty); + atomic_store(&ps->cancelled, true); + cnd_broadcast(&ps->result_not_empty); + cnd_broadcast(&ps->result_not_full); + mtx_unlock(&ps->result_mutex); for (int i = 0; i < ps->num_threads; i++) thrd_join(ps->threads[i], NULL); free(ps->threads); + if (ps->root_filter_node) + filter_node_destroy(ps->root_filter_node); if (ps->initial_chunk) chunk_destroy(ps->initial_chunk); queue_destroy(ps->result_queue); diff --git a/src/client/scanner.h b/src/client/scanner.h index f3e0f0c..950a14e 100644 --- a/src/client/scanner.h +++ b/src/client/scanner.h @@ -2,16 +2,129 @@ #define SCANNER_H #include "chunk.h" +#include "file_list.h" +#include "filter.h" +#include "hardlink.h" +#include "protocol.h" #include "queue.h" +#include "stop_condition.h" #include #include +#include +#include #include +typedef struct { + bool use_metadata; + /* Phase 4 metadata capture: -U/--atimes and -N/--crtimes tell the scanner to + * capture the source access / birth time into each entry's FileMetadata. */ + bool preserve_atimes; + bool preserve_crtimes; + /* Phase 4 xattrs: when preserve_xattrs || preserve_acls is set the scanner + * captures each regular file's whitelisted xattr set onto the File. */ + bool preserve_xattrs; + bool preserve_acls; + unsigned long long chunk_size; + char** exclude_patterns; + int exclude_count; + char** include_patterns; + int include_count; + unsigned long long max_size; + unsigned long long min_size; + int max_depth; + int num_threads; + bool follow_symlinks; + bool copy_links; + bool safe_links; + bool copy_unsafe_links; + /* Phase 4 symlink-trust sender options: -k/--copy-dirlinks (dereference a + * symlink to a directory as a directory, keeping symlinks-to-files as + * symlinks) and --munge-links (rewrite each transmitted symlink target with a + * marker; escaping targets are never transmitted). Both are client/sender + * side only and never serialized to the wire (keep_dirlinks is the + * receiver-side counterpart). */ + bool copy_dirlinks; + bool munge_links; + bool checksum; + bool one_file_system; + /* Phase 4 special/devices: whether device nodes (--devices) and special files + * (--specials) are preserved via recreation, and whether --copy-devices + * copies a device's content as an ordinary regular file. */ + bool preserve_devices; + bool preserve_specials; + bool copy_devices; + /* Phase 2 (files-from / filter layer). All pointers are shared read-only + * across scanner instances and worker threads; ownership stays with the + * caller (client_send). */ + const FileListSet* file_list; /* --files-from allow-set, or NULL */ + const FilterRuleList* base_filters; /* command-line + -C rules, or NULL */ + bool per_dir_filters; /* -F: read .rsync-filter per directory */ + bool dirs; /* -d/--dirs: transfer dir entries, no recursion */ + bool relative; /* -R/--relative (dest rel paths, with --files-from) */ + /* --prune-empty-dirs (long only): in --dirs mode an empty source directory's + explicit entry is omitted from the transfer file list (so nothing is + created at the destination and it can be pruned by --delete); explicitly + --files-from-listed directories always pass through. Recursive transfers + never emit empty directories, so the flag has no additional effect there. */ + bool prune_empty_dirs; + /* Delete-excluded protection sink (optional): when non-NULL the scanner + * appends the destination-relative path of every entry it prunes because a + * USER SELECTION rule excluded it (--filter/-C/per-dir rules, the legacy + * --exclude/--include layer, and --max-size/--min-size). The sender turns + * this list into the manifest's protected prefixes so `--delete` leaves the + * destination mirror of excluded source paths alone (rsync's default), and + * empties it when --delete-excluded opts back into deleting them. NOT + * recorded for --files-from subset pruning (whose delete semantics stay + * keep-set-only) or for -R/--files-from relative wire paths. When + * `excluded_mutex` is non-NULL it is taken around every append (the parallel + * scanner shares one list across its worker threads). */ + ArrayList* excluded_paths; + mtx_t* excluded_mutex; + /* --ignore-errors: an unreadable directory during the scan is recorded as an + * I/O error and skipped instead of aborting the scan. Client-only. */ + bool ignore_io_errors; + /* --ignore-missing-args (implied by --delete-missing-args): an explicitly + * --files-from-listed entry that does not exist under the source is skipped + * instead of failing (the --dirs generator is the only scanner path that + * observes a listed-but-missing entry). */ + bool ignore_missing_args; + /* --hard-links (-H): shared, mutable (mutex-guarded) link-group detection + * table, NULL when -H is off. Owned by the caller (client_send), shared + * read-only here; the parallel scanner passes it unchanged to every worker so + * one table detects every group across all subdirectories. */ + HardLinkTable* hardlinks; + /* Phase 6: optional sender stop deadline. When non-NULL the scanner checks + * it at natural loop boundaries and stops emitting chunks once reached + * (without marking the scan as failed), so a busy scan itself stops early. + * Client-only, never serialized to the wire. */ + const StopCondition* stop_condition; + /* P7 Wave D (protocol 2.17.0): directory-time capture sink. When + * `capture_dir_times` is true the recursive scan appends one is_dir File + * (with metadata, no payload) per source directory it traverses to + * `dir_entries`, so the sender can transmit trailing STATUS_DIR_TIMES + * frame(s) and the receiver can apply directory mtimes AFTER all children + * are written. `dir_entries_mutex` (optional) guards the list + * for the parallel scanner's shared worker threads; the caller owns both. + * The --dirs generator does not use this (its directory entries carry their + * metadata inline through STATUS_MKDIR). */ + bool capture_dir_times; + ArrayList* dir_entries; + mtx_t* dir_entries_mutex; +} ScannerOptions; + +/* Internal per-scanner filter state. FilterNode chains represent the ordered + * per-directory .rsync-filter rules that apply below a directory. */ +typedef struct FilterNode FilterNode; + typedef struct { Queue* directories; DIR* current_dir; char* current_path; bool use_metadata; + bool preserve_atimes; + bool preserve_crtimes; + bool preserve_xattrs; + bool preserve_acls; unsigned long long chunk_size; char** exclude_patterns; int exclude_count; @@ -25,6 +138,57 @@ typedef struct { bool copy_links; bool safe_links; bool copy_unsafe_links; + bool copy_dirlinks; + bool munge_links; + bool checksum; + bool one_file_system; + dev_t root_dev; + bool failed; + /* Phase 4 special/devices (see ScannerOptions). */ + bool preserve_devices; + bool preserve_specials; + bool copy_devices; + /* Phase 2 (files-from / filter layer). */ + char* root_path; /* transfer root (fs path) for rel computation */ + char* current_rel; /* rel path of the open directory ("" == root) */ + bool at_seed_dir; /* next open is the seed directory */ + FilterNode* seed_node; /* inherited context of the seed dir, or NULL */ + FilterNode* current_node; /* filter context of the open directory */ + ArrayList* filter_nodes; /* owned FilterNode arena (may be NULL) */ + const FileListSet* file_list; + const FilterRuleList* base_filters; + bool per_dir_filters; + /* --dirs / -R state for the directory-entry generator (dirs_mode replaces + the recursive scan). */ + bool dirs_mode; + bool relative_mode; /* file_list && relative: send bare relative wire paths */ + bool prune_empty_dirs; + bool dirs_root_emitted; + int list_index; + ArrayList* dirs_batch; /* owned when non-NULL */ + unsigned long long dirs_batch_size; + /* Excluded-path sink (see ScannerOptions). `excluded_mutex` is shared across + parallel worker threads. */ + ArrayList* excluded_paths; + mtx_t* excluded_mutex; + /* --ignore-errors: continue past unreadable directories (records io_error). */ + bool ignore_io_errors; + /* --ignore-missing-args: --dirs listed-but-missing entries are skipped, not + fatal (see ScannerOptions.ignore_missing_args). */ + bool ignore_missing_args; + /* A directory could not be opened (I/O error, e.g. EACCES). With + --ignore-errors the scan continues past it and the caller decides what to + do; `failed` is reserved for fatal errors that always abort the scan. */ + bool io_error; + /* --hard-links (-H): shared link-group detection table (see ScannerOptions). + NULL when -H is off. */ + HardLinkTable* hardlinks; + /* Phase 6: sender stop deadline (from ScannerOptions). */ + const StopCondition* stop_condition; + /* P7 Wave D directory-time capture (see ScannerOptions). */ + bool capture_dir_times; + ArrayList* dir_entries; + mtx_t* dir_entries_mutex; } DirectoryScanner; typedef struct { @@ -33,10 +197,18 @@ typedef struct { cnd_t result_not_empty; cnd_t result_not_full; int num_threads; + int expected_threads; + int created_threads; thrd_t* threads; bool done; + bool failed; + /* A worker skipped an unreadable directory under --ignore-errors (non-fatal). */ + bool io_error; + atomic_bool cancelled; int completed; Chunk* initial_chunk; + ProtocolSession* allocation_session; + FilterNode* root_filter_node; /* root .rsync-filter context (owned by ps) */ } ParallelScanner; DirectoryScanner* directory_scanner_create(const char* root_directory, bool use_metadata, @@ -45,18 +217,33 @@ DirectoryScanner* directory_scanner_create(const char* root_directory, bool use_ int include_count, unsigned long long max_size, unsigned long long min_size, int max_depth, bool follow_symlinks, bool copy_links, bool safe_links, - bool copy_unsafe_links); + bool copy_unsafe_links, bool checksum); +DirectoryScanner* directory_scanner_create_with_options(const char* root_directory, + const ScannerOptions* options); Chunk* directory_scanner_next(DirectoryScanner* scanner); +bool directory_scanner_failed(const DirectoryScanner* scanner); void directory_scanner_destroy(DirectoryScanner* scanner); -ParallelScanner* parallel_scanner_create(char* root_directory, bool use_metadata, - unsigned long long chunk_size, char** exclude_patterns, - int exclude_count, char** include_patterns, - int include_count, unsigned long long max_size, - unsigned long long min_size, int max_depth, - int num_threads, bool follow_symlinks, bool copy_links, - bool safe_links, bool copy_unsafe_links); +/* --one-file-system (-x) decision: a directory entry may be descended into + * only when the option is disabled or the entry lives on the same device as + * the transfer root. Exposed so tests can exercise the rule directly. */ +bool scanner_same_filesystem(bool one_file_system, dev_t root_device, dev_t entry_device); + +/* Relative path of an on-disk path below `root` ("" == the root itself, NULL + * when `fs_path` is not under `root`). Handles trailing slashes and a root of + * "/". Exposed so tests can exercise the mapping directly. */ +char* scanner_path_relative(const char* root, const char* fs_path); + +ParallelScanner* parallel_scanner_create_with_options(const char* root_directory, + const ScannerOptions* options, + ProtocolSession* allocation_session); Chunk* parallel_scanner_next(ParallelScanner* scanner); +bool parallel_scanner_failed(const ParallelScanner* scanner); +bool parallel_scanner_had_io_error(const ParallelScanner* scanner); void parallel_scanner_destroy(ParallelScanner* scanner); +/* True when a directory could not be opened during the scan (an I/O error, + recorded even when --ignore-errors keeps the scan going past it). */ +bool directory_scanner_had_io_error(const DirectoryScanner* scanner); + #endif diff --git a/src/client/usage.c b/src/client/usage.c new file mode 100644 index 0000000..f63a5c3 --- /dev/null +++ b/src/client/usage.c @@ -0,0 +1,298 @@ +#include "usage.h" +#include +#include +#include + +void print_usage(void) { + printf("Usage:\n"); + printf(" fastsync [options] \n"); + printf(" fastsync [options] --source-dir --dest-dir \n"); + printf("\n"); + printf("Destination formats:\n"); + printf(" user@host:/path SSH transport (rsync-style)\n"); + printf(" host:/path SSH transport (current user)\n"); + printf(" host::module/path Daemon TCP transport (fastsync-server --daemon);\n"); + printf(" module names a server-side module, path is relative\n"); + printf(" within it (connect with --server-port)\n"); + printf(" /local/path TCP transport (requires server on localhost:8080)\n"); + printf("\n"); + printf("Options:\n"); + printf(" -c, --checksum Verify content by checksum instead of size+mtime\n"); + printf(" -z, --compress [level] Enable compression (level 1-22, default 5)\n"); + printf(" -a, --archive rsync archive mode (-rlptgoD): links, metadata,\n"); + printf(" devices and specials (not compression/multithreading)\n"); + printf(" -n, --dry-run Show what would be transferred\n"); + printf(" --remove-source-files Remove regular source files after successful transfer\n"); + printf(" -p, --perms Preserve permission bits (part of the metadata bundle)\n"); + printf(" --ssh-port SSH port (default: 22)\n"); + printf(" -e, --rsh Remote shell to launch on the client for the SSH\n"); + printf(" transport (default: ssh). The command may include\n"); + printf(" arguments, e.g. -e \"ssh -p 2222\"\n"); + printf(" --rsync-path Alias for --fastsync-server-path (path to the\n"); + printf(" fastsync server binary on the remote side)\n"); + printf(" --blocking-io Leave the SSH transport socket without read/write\n"); + printf(" timeouts so it blocks naturally\n"); + printf(" --outbuf=MODE stdout/stderr buffering: N (none/unbuffered),\n"); + printf(" L (line-buffered), or B (block-buffered, default)\n"); + printf(" --progress Show transfer progress\n"); + printf(" -P Partial mode with progress (retention incomplete)\n"); + printf(" -8, --8-bit-output Leave high-bit characters unescaped in output\n"); + printf(" --iconv=LOCAL[,REMOTE] Convert file-NAME charsets at the wire boundary:\n"); + printf(" LOCAL is the charset of our file names, REMOTE is the\n"); + printf(" remote side's charset (defaults to LOCAL). Names are\n"); + printf(" converted before transmission and back on receipt; a\n"); + printf(" name that cannot be represented in the target charset\n"); + printf(" fails that transfer cleanly (rsync-compatible)\n"); + printf(" --protocol=NUM Force the wire protocol version (must equal the current\n"); + printf(" PROTOCOL_VERSION; FastSync cannot speak older/virtual\n"); + printf(" wire formats)\n"); + printf(" --write-batch=FILE Run the normal live transfer AND also emit a\n"); + printf(" self-contained batch file of the whole source tree\n"); + printf(" (implies the single-threaded transfer path)\n"); + printf(" --only-write-batch=FILE\n"); + printf(" Emit the batch file only (no destination, no server)\n"); + printf(" --read-batch=FILE Apply the batch file to the destination (no source, no\n"); + printf(" server); takes only the destination as an argument\n"); + printf(" --delete Delete files on receiver not in source\n"); + printf(" (default timing: delete only after the whole\n"); + printf(" transfer has succeeded)\n"); + printf(" --delete-before Delete extras before the transfer starts\n"); + printf(" (implies --delete)\n"); + printf(" --delete-during Delete extras once the keep-set manifest is known,\n"); + printf(" before the data is applied (implies --delete)\n"); + printf(" --del Alias for --delete-during\n"); + printf(" --delete-delay Delete extras only after a successful transfer\n"); + printf(" (implies --delete)\n"); + printf(" --delete-after Delete only after the whole transfer succeeded\n"); + printf(" (the default --delete timing; implies --delete)\n"); + printf(" --delete-excluded Also delete destination files that were excluded on\n"); + printf(" the source (default protects them, matching rsync)\n"); + printf(" --max-delete=NUM Never delete more than NUM destination entries per run;\n"); + printf(" if the extras would exceed NUM, nothing is deleted and\n"); + printf(" the run fails with a clear error (implies --delete only\n"); + printf(" when used with it)\n"); + printf(" --ignore-errors Continue (and still delete) when a source directory is\n"); + printf(" unreadable during the scan, instead of aborting with no\n"); + printf(" deletion\n"); + printf(" --force A file may replace a destination directory by removing\n"); + printf(" that (non-empty) directory first\n"); + printf(" --ignore-missing-args A --files-from entry that does not exist under the\n"); + printf(" source is silently skipped instead of failing the run\n"); + printf(" --delete-missing-args Implies --ignore-missing-args; also deletes each missing\n"); + printf(" entry's destination mirror receiver-side. Independent of\n"); + printf(" --delete (it does not imply --delete; a non-empty directory\n"); + printf(" mirror is removed only with --force or --delete)\n"); + printf(" -m, --prune-empty-dirs Do not transfer empty directory entries (--dirs mode);\n"); + printf(" recursive transfers never send empty dirs\n"); + printf(" Note: each timing flag implies --delete. Combining a timing flag with\n"); + printf(" --no-delete (in either order) is rejected as a config error.\n"); + printf(" --ignore-existing Skip files that already exist on receiver\n"); + printf(" --delay-updates Put updated files into place only at the end of transfer\n"); + printf(" --dirs, -d, --old-dirs, --old-d Transfer the named directory entries without\n"); + printf(" recursing into their contents (-d mirrors the source\n"); + printf(" directory empty; with --files-from listed dirs are created\n"); + printf(" empty and listed files are transferred)\n"); + printf(" -R, --relative With --files-from, preserve each listed entry's relative path\n"); + printf(" below the destination root instead of mirroring the full\n"); + printf(" source path (no effect without --files-from)\n"); + printf(" --no-implied-dirs With -R --files-from, refuse to place a listed file whose\n"); + printf(" parent directory is not itself listed\n"); + printf(" --mkpath Create the destination root directory on the server when it\n"); + printf(" does not exist yet\n"); + printf(" --exclude Exclude files matching pattern\n"); + printf(" --include Only include files matching pattern\n"); + printf(" --exclude-from Read exclude patterns from file\n"); + printf(" --include-from Read include patterns from file\n"); + printf(" --files-from Read the source file list from FILE (paths relative to the " + "source root)\n"); + printf(" -0, --from0 Entries in --files-from are NUL-delimited\n"); + printf(" -f, --filter=RULE rsync-style filter rule (+/- include/exclude; repeatable;\n"); + printf(" both --filter=RULE and the -f RULE / -f=RULE short forms work)\n"); + printf(" -C, --cvs-exclude Auto-ignore common CVS/SCM files (.git/, .svn/, *.o, *~, ...)\n"); + printf(" -F Apply per-directory .rsync-filter files during the scan\n"); + printf(" --max-size Skip files larger than n bytes\n"); + printf(" --min-size Skip files smaller than n bytes\n"); + printf(" --max-alloc Maximum single allocation (default: 1G)\n"); + printf(" --incremental Skip files unchanged since last transfer\n"); + printf(" --size-only Skip incremental files matching in size, ignoring mtime\n"); + printf(" -I, --ignore-times Transfer files even when size and mtime match\n"); + printf(" -@, --modify-window Modification time tolerance\n"); + printf(" -u, --update Skip files newer than the source on receiver\n"); + printf(" --existing Skip files not already present at destination\n"); + printf(" --compare-dest Treat DIR (relative to destination root) as an extra\n"); + printf(" comparison basis: unchanged files are not transferred\n"); + printf(" (requires --incremental, which is implied)\n"); + printf(" --copy-dest Like --compare-dest, but copies the unchanged file from DIR\n"); + printf(" into the destination instead of transferring its data\n"); + printf(" --link-dest Like --copy-dest, but hard-links the unchanged file from DIR\n"); + printf(" into the destination (repeatable; earlier DIRs win)\n"); + printf(" --checksum-choice, --cc Whole-file checksum algorithm for --incremental/\n"); + printf(" --checksum compares (xxh64/xxhash or md5; default xxh64 with\n"); + printf(" seed 0). The seed comes from --checksum-seed\n"); + printf(" --checksum-seed Seed for the whole-file xxHash64 digest (and the delta\n"); + printf(" block strong hash, low 32 bits); md5 ignores the seed. The\n"); + printf(" digest algorithm and seed must match on sender and receiver\n"); + printf(" --delta Delta transfer for changed files (requires --incremental)\n"); + printf(" -W, --whole-file Transfer changed files without delta processing\n"); + printf(" -y, --fuzzy Use a similar-named file already in the destination\n"); + printf(" directory as the delta basis when the destination has no\n"); + printf(" usable file at the exact path (saves bandwidth; implies\n"); + printf(" --incremental and --delta; inert with --whole-file,\n"); + printf(" --no-delta, or --no-incremental)\n"); + printf(" --no-fuzzy Disable --fuzzy\n"); + printf(" --delta-block , --block-size \n"); + printf(" Delta block size in bytes (default: %d)\n", DELTA_BLOCK_SIZE_DEFAULT); + printf(" --delta-max Max file size for delta transfer (default: %llu)\n", + DELTA_MAX_FILE_SIZE); + printf(" -j, --threads Enable multithreading\n"); + printf(" --chunk-serialization Enable chunk serialization (long form only)\n"); + printf(" -s, --secluded-args Protect-args compatibility option (no effect; remote\n"); + printf(" SSH argv is already built injection-safe)\n"); + printf(" --sendfile Enable sendfile zero-copy (TCP only; long form only)\n"); + printf(" --compress-choice Compression algorithm (default: zstd)\n"); + printf(" --zc Alias for --compress-choice\n"); + printf(" -v, --verbose Enable debug logging\n"); + printf(" -q, --quiet Suppress non-error output\n"); + printf(" --debug=FLAGS Fine-grained debug logging (use --debug=help for flags)\n"); + printf(" --info=FLAGS Fine-grained info: copy,misc,skip,stats,all,none\n"); + printf(" none suppresses info even with --verbose\n"); + printf(" --preserve Preserve file metadata (long form only)\n"); + printf(" -E, --executability Preserve executable permission bits\n"); + printf(" -X, --xattrs Preserve user extended attributes (user.* only;\n"); + printf(" privileged security.*/trusted.* namespaces are\n"); + printf(" never captured or applied)\n"); + printf(" -A, --acls Preserve POSIX ACLs (the system.posix_acl_* xattrs;\n"); + printf(" setting an ACL the receiver is not permitted to\n"); + printf(" set is warned and skipped, never fatal)\n"); + printf(" --fake-super Store the source uid/gid/mode/mtime in a reserved\n"); + printf(" user.fastsync.stat xattr on each written file and\n"); + printf(" re-apply it (fd-relative) on a privileged run; the\n"); + printf(" recording format diverges from rsync's user.rsync.%%stat%%\n"); + printf(" --super Permit the receiver to attempt super-user activities\n"); + printf(" (char/block device-node creation, --write-devices)\n"); + printf(" within the confined receive root. Never elevates\n"); + printf(" privileges and never bypasses confinement; ownership\n"); + printf(" is still applied only with an explicit identity flag\n"); + printf(" (--numeric-ids/--chown/--usermap/--groupmap/--copy-as)\n"); + printf(" --no-super Forbid those super-user activities even when the\n"); + printf(" receiver is running as root\n"); + printf(" --chmod Modify transferred permissions (rsync syntax)\n"); + printf(" --numeric-ids Do not map uid/gid by name: use the source numeric\n"); + printf(" ids directly when applying ownership\n"); + printf(" --usermap=MAP Map usernames when applying ownership: comma-separated\n"); + printf(" FROM:TO rules, first match wins. FROM/TO are names\n"); + printf(" (resolved on the source machine), * (match any /\n"); + printf(" current user), or @N numeric ids. e.g. *:nobody\n"); + printf(" --groupmap=MAP Map group names when applying ownership (same syntax)\n"); + printf(" --chown=USER:GROUP Override the ownership of transferred files. Forms:\n"); + printf(" USER:GROUP, USER (owner only), :GROUP (group only); a\n"); + printf(" value of * means the current/root user as appropriate.\n"); + printf(" Names resolve on the source machine; @N for numerics.\n"); + printf(" (Metadata is enabled with --preserve; -M now means\n"); + printf(" rsync's --remote-option.)\n"); + printf(" --copy-as=USER[:GROUP] Force every written entry (files, dirs, symlinks\n"); + printf(" and special nodes) to USER[:GROUP], resolved on the\n"); + printf(" source machine like --chown. Requires a privileged\n"); + printf(" (root) receiver and implies --preserve; an\n"); + printf(" unprivileged receiver refuses the transfer. Never\n"); + printf(" switches process credentials (safe-subset; see\n"); + printf(" RSYNC_COMPAT.md). A daemon refuses it.\n"); + printf(" --chunk-size Chunk size in bytes (default: %d)\n", DEFAULT_CHUNK_SIZE); + printf(" --source-dir Source directory\n"); + printf(" --dest-dir Destination directory\n"); + printf(" --save-to-disk Write received files to disk\n"); + printf(" --server-host Server IP address (default: 127.0.0.1)\n"); + printf(" --server-port Server port (default: 8080)\n"); + printf(" --password-file Authenticate a host::module/path daemon destination.\n"); + printf(" The file's first user:password line supplies the\n"); + printf(" username and password (only a SHA-256 digest of the\n"); + printf(" password is sent; keep the file mode 0600)\n"); + printf(" --no-motd Suppress display of the daemon's MOTD (the server\n"); + printf(" still sends it; the client just does not show it)\n"); + printf(" --bwlimit Bandwidth limit in kilobytes per second\n"); + printf(" --tls Enable TLS encryption\n"); + printf(" --cert TLS certificate file (PEM)\n"); + printf(" --key TLS private key file (PEM)\n"); + printf(" --ca TLS CA certificate file (PEM)\n"); + printf(" --timeout I/O timeout in seconds (default: 30; long form only)\n"); + printf(" --contimeout Connection timeout in seconds (default: 10)\n"); + printf(" --stop-after=MINS Stop the transfer after MINS minutes (a positive\n"); + printf(" integer); whatever was already transferred is kept\n"); + printf(" --stop-at=TIME Stop at an absolute time: HH:MM, HH:MM:SS, or\n"); + printf(" now+N[smhd] (a time already in the past stops the\n"); + printf(" transfer immediately; client-only). An early stop\n"); + printf(" skips the late --delete keep-set so it cannot delete\n"); + printf(" source mirrors that were not yet scanned\n"); + printf(" --address Bind the outgoing client socket to this source address\n"); + printf(" -4, --ipv4 Force IPv4 for destination resolution\n"); + printf(" -6, --ipv6 Force IPv6 for destination resolution\n"); + printf(" --sockopts=OPTS Comma-separated OPT=VAL socket options applied before connect:\n"); + printf(" TCP_NODELAY, SO_KEEPALIVE, SO_RCVBUF, SO_SNDBUF, SO_REUSEADDR\n"); + printf(" --backup Backup existing files before overwriting\n"); + printf(" --backup-dir Directory for backups (requires --backup)\n"); + printf(" --suffix Backup suffix (default: ~)\n"); + printf(" --stats Print transfer statistics at end\n"); + printf(" -i, --itemize-changes Print an rsync-style per-file change line\n"); + printf(" --out-format=FORMAT Output format for changed files (%%f %%n %%l %%b %%M %%%%)\n"); + printf(" --list-only List source files instead of transferring\n"); + printf(" --log-file-format=FORMAT Per-file log line format (needs --log-file)\n"); + printf(" -h, --human-readable Print byte sizes in human-readable form\n"); + printf(" --max-depth Maximum directory depth (0=unlimited)\n"); + printf(" -x, --one-file-system Do not cross filesystem boundaries\n"); + printf(" --log-file Write log messages to file\n"); + printf(" --stderr=MODE Route logging to stderr: errors or all\n"); + printf(" --partial Keep partial files on interrupted transfer\n"); + printf(" --partial-dir Directory for partial files\n"); + printf(" -T, --temp-dir Scratch dir for temp files before atomic install\n"); + printf(" --fastsync-server-path \n"); + printf(" Path to fastsync-server on remote (default: fastsync-server)\n"); + printf(" --old-args Accepted for rsync CLI compatibility; no effect (the\n"); + printf(" remote server path is always safely quoted now)\n"); + printf(" -M, --remote-option=OPT Append OPT to the REMOTE server invocation over SSH\n"); + printf(" (repeatable; each value is single-quote-escaped on the remote\n"); + printf(" command line; empty values and values with control characters\n"); + printf(" are rejected; -M OPT, -M=OPT and --remote-option=OPT work)\n"); + printf(" --trust-sender Trust the remote sender's file list: the receiver skips its\n"); + printf(" own up-front path-traversal/containment re-validation of the\n"); + printf(" incoming file list (fewer checks, faster, potentially unsafe).\n"); + printf(" Local receiver policy: never sent to the peer, off by default\n"); + printf(" -l, --links Copy symlinks as symlinks\n"); + printf(" --copy-links Transform symlinks into referent files\n"); + printf(" --safe-links Skip symlinks that point outside transfer tree\n"); + printf(" --copy-unsafe-links Only transform unsafe symlinks into referent files\n"); + printf(" -k, --copy-dirlinks Transform symlinks to directories into real dirs\n"); + printf(" -K, --keep-dirlinks Keep an existing symlink-to-dir as that dir\n"); + printf(" --munge-links Munge symlink targets on the wire (sender)\n"); + printf(" -H, --hard-links Preserve hard-link relationships across the transfer\n"); + printf(" -S, --sparse Handle sparse files efficiently\n"); + printf( + " -D Preserve device and special files (implies --devices --specials)\n"); + printf( + " --devices Recreate device nodes on the destination (privileged; skipped when\n"); + printf(" the receiver lacks CAP_MKNOD)\n"); + printf(" --specials Recreate special files (FIFOs) on the destination (sockets " + "skipped)\n"); + printf(" --copy-devices Copy a source device's content as a regular file instead\n"); + printf(" --write-devices Write received data into an existing destination device node\n"); + printf(" --inplace Update files in-place (no temp+rename)\n"); + printf( + " --preallocate Allocate destination file space up front (fail-fast on full disk)\n"); + printf(" --append Resume a shorter destination by appending only its tail\n"); + printf(" (prefix is not verified; requires --incremental)\n"); + printf(" --append-verify Like --append, but verifies the retained prefix checksum\n"); + printf(" before appending (falls back to a full transfer on mismatch)\n"); + printf(" --fsync Fsync every written file before publication\n"); + printf(" --compress-level Compression level (default: 5)\n"); + printf(" --zl Alias for --compress-level\n"); + printf(" --skip-compress=LIST Skip compression for comma-separated suffixes\n"); + printf(" --compress-threads Compression worker threads (requires zstd threaded support)\n"); + printf(" --no-OPTION Disable a supported boolean option\n"); + printf(" --help Show this help\n"); + printf(" -V, --version Show version\n"); +} + +void print_debug_usage(void) { + printf("Supported debug flags: IO,PROTO,PACK,UTIL,ALL,NONE\n"); + printf("Flags may be comma-separated, for example: --debug=io,proto\n"); + printf("Other rsync debug flags are unsupported and rejected.\n"); +} diff --git a/src/client/usage.h b/src/client/usage.h new file mode 100644 index 0000000..ca8d65b --- /dev/null +++ b/src/client/usage.h @@ -0,0 +1,7 @@ +#ifndef USAGE_H +#define USAGE_H + +void print_usage(void); +void print_debug_usage(void); + +#endif diff --git a/src/server/receiver.c b/src/server/receiver.c new file mode 100644 index 0000000..936fc1f --- /dev/null +++ b/src/server/receiver.c @@ -0,0 +1,402 @@ +#include "receiver.h" + +#include "charset.h" +#include "chunk.h" +#include "config.h" +#include "delay_updates.h" +#include "file.h" +#include "file_receive.h" +#include "log.h" +#include "metadata.h" +#include "protocol.h" +#include "utils.h" +#include +#include + +bool receiver_outcomes_append(ReceiverOutcomes* outcomes, unsigned char code) { + if (!outcomes) + return false; + if (outcomes->count == outcomes->capacity) { + size_t new_capacity = outcomes->capacity == 0 ? 64 : outcomes->capacity * 2; + if (new_capacity < outcomes->capacity) + return false; + unsigned char* grown = realloc(outcomes->entries, new_capacity); + if (!grown) + return false; + outcomes->entries = grown; + outcomes->capacity = new_capacity; + } + outcomes->entries[outcomes->count++] = code; + return true; +} + +void receiver_outcomes_destroy(ReceiverOutcomes* outcomes) { + if (!outcomes) + return; + free(outcomes->entries); + outcomes->entries = NULL; + outcomes->count = 0; + outcomes->capacity = 0; +} + +/* End-of-transfer success frame. When --remove-source-files was negotiated + each processed data file is acknowledged first (STATUS_NEXT = written, + STATUS_OK = skipped) so the sender never removes a source the receiver did + not actually store. The frame always ends with a plain STATUS_OK. */ +bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes) { + if (!config->remove_source_files) + return send_status(fd, STATUS_OK); + size_t count = outcomes ? outcomes->count : 0; + for (size_t i = 0; i < count; i++) { + Status per_file = outcomes->entries[i] == FILE_SAVE_WRITTEN ? STATUS_NEXT : STATUS_OK; + if (!send_status(fd, per_file)) + return false; + } + return send_status(fd, STATUS_OK); +} + +static bool receiver_process_chunk(Chunk* chunk, const ReceiverSink* sink) { + if (!chunk || !sink || !sink->store_file) + return false; + for (int i = 0; i < chunk->element_count; i++) { + File* file = chunk->items[i]; + if (!file) { + chunk_destroy(chunk); + return false; + } + chunk->items[i] = NULL; + if (!sink->store_file(file, sink->context)) { + chunk_destroy(chunk); + return false; + } + } + chunk_destroy(chunk); + return true; +} + +/* P7 Wave D: read one STATUS_DIR_TIMES frame (a count followed by that many + * (path, metadata) directory entries) and route every entry through the regular + * store_file sink. A dir-time entry is RECORD-ONLY (file->dir_time_only): the + * sink accumulates its metadata for end-of-transfer application but creates + * nothing, so an empty/pruned source directory is never resurrected. A large + * tree arrives as repeated frames, each bounded by MAX_MANIFEST_ENTRIES; a + * malformed count or entry is a hard error. */ +static bool receiver_process_dir_times(int fd, const Config* config, const ReceiverSink* sink) { + int count; + if (!receive_int(fd, &count) || count < 0 || count > MAX_MANIFEST_ENTRIES) + return false; + for (int i = 0; i < count; i++) { + File* dir = file_receive_dir_time(fd, config); + if (!dir || !sink->store_file(dir, sink->context)) + return false; + } + return true; +} + +static bool receiver_process_batch(Config* config, int file_descriptor) { + int count; + if (config->checksum || !receive_int(file_descriptor, &count) || count < 0 || + count > MAX_MANIFEST_ENTRIES) + return false; + for (int i = 0; i < count; i++) { + char* check_path = receive_wire_str(file_descriptor); + if (!check_path) + return false; + unsigned long long check_size; + long long check_mtime; + long long check_mtime_nsec; + if (!receive_n_data(file_descriptor, &check_size, sizeof(check_size)) || + !receive_n_data(file_descriptor, &check_mtime, sizeof(check_mtime)) || + !receive_n_data(file_descriptor, &check_mtime_nsec, sizeof(check_mtime_nsec)) || + check_mtime_nsec < 0 || check_mtime_nsec >= 1000000000LL) { + free(check_path); + send_status(file_descriptor, STATUS_ERROR); + return false; + } + /* --trust-sender: accept a ``..``/absolute check path (a trusted sender's + odd-but-legit entry) and defer containment to the secure stat below; + an empty path is still always rejected. */ + if (check_path[0] == '\0' || + (!file_get_trust_sender() && !utils_valid_batch_path(check_path))) { + free(check_path); + send_status(file_descriptor, STATUS_ERROR); + return false; + } + if (check_size > MAX_RECEIVE_WHOLE_FILE_SIZE) { + free(check_path); + send_status(file_descriptor, STATUS_ERROR); + return false; + } + char* full_path = path_cat(config->receive_root_directory, check_path); + if (!full_path) { + free(check_path); + send_status(file_descriptor, STATUS_ERROR); + return false; + } + struct stat st; + bool has_old = file_stat_secure(full_path, &st); + long long old_mtime_nsec = 0; + if (has_old) { +#ifdef __linux__ + old_mtime_nsec = st.st_mtim.tv_nsec; +#endif + } + bool match = !config->ignore_times && has_old && (unsigned long long)st.st_size == check_size && + metadata_mtime_matches(st.st_mtime, old_mtime_nsec, (time_t)check_mtime, + (long)check_mtime_nsec, config->modify_window); + bool sent = send_status(file_descriptor, match ? STATUS_OK : STATUS_NEXT); + free(full_path); + free(check_path); + if (!sent) + return false; + } + return true; +} + +int receiver_process(Config* config, int file_descriptor, const ReceiverSink* sink) { + return receiver_process_pending(config, file_descriptor, sink, NULL); +} + +/* Runs the whole receive loop. The delete manifest may legitimately arrive + either FIRST (--delete-before / --delete-during: the sender transmits the + validated keep-set before any file data) or LAST (plain --delete / + --delete-after / --delete-delay: the manifest closes the data stream). In + the early modes the receiver deletes as soon as the manifest has been read + and acknowledges with STATUS_OK so the sender only starts streaming once the + deletion has committed (or failed); in the late modes the manifest is held + and the deletion is committed only after the terminal STATUS_FINISHED proves + the whole transfer succeeded. See receiver_process_pending() for how the -m + receiver defers that commit until its disk writer has drained. */ +int receiver_process_pending(Config* config, int file_descriptor, const ReceiverSink* sink, + DeleteManifest** pending_manifest) { + Status status; + if (!receive_status(file_descriptor, &status)) + return -1; + bool early_delete = config_delete_timing_early(config); + /* Parked keep-set for the late/commit timing. Every exit path below frees it + exactly once; the only exception is the successful FINISHED handoff, which + transfers ownership to *pending_manifest (used by the -m receiver). */ + DeleteManifest* deferred_manifest = NULL; + while (status == STATUS_NEXT || status == STATUS_CHUNK || status == STATUS_CHECK || + status == STATUS_KEEPALIVE || status == STATUS_ABORT || status == STATUS_CHECK_BATCH || + status == STATUS_MKDIR || status == STATUS_MANIFEST || status == STATUS_HARDLINK || + status == STATUS_SYMLINK || status == STATUS_SPECIAL || status == STATUS_DIR_TIMES) { + if (status == STATUS_KEEPALIVE) { + if (!send_status(file_descriptor, STATUS_KEEPALIVE)) + goto fail; + goto next_status; + } + if (status == STATUS_ABORT) { + log_message(LOG_LEVEL_INFO, "Received abort from client, cleaning up"); + goto fail; + } + if (status == STATUS_CHECK) { + bool skipped; + File* file = receive_incremental_check(file_descriptor, config, &skipped); + if (!skipped && (!file || !sink->store_file(file, sink->context))) + goto receive_error; + } else if (status == STATUS_CHUNK) { + Chunk* chunk = receive_chunk_data(file_descriptor, config); + if (!chunk || !receiver_process_chunk(chunk, sink)) + goto receive_error; + } else if (status == STATUS_CHECK_BATCH) { + if (!receiver_process_batch(config, file_descriptor)) + goto fail; + goto next_status; + } else if (status == STATUS_MKDIR) { + File* dir = file_receive_directory(file_descriptor, config); + if (!dir || !sink->store_file(dir, sink->context)) + goto receive_error; + } else if (status == STATUS_DIR_TIMES) { + if (!receiver_process_dir_times(file_descriptor, config, sink)) + goto receive_error; + } else if (status == STATUS_HARDLINK) { + File* file = file_receive_hardlink(file_descriptor); + if (!file || !sink->store_file(file, sink->context)) + goto receive_error; + } else if (status == STATUS_SYMLINK) { + File* sym = file_receive_symlink(file_descriptor, config); + if (!sym || !sink->store_file(sym, sink->context)) + goto receive_error; + } else if (status == STATUS_SPECIAL) { + File* file = file_receive_special(file_descriptor); + if (!file || !sink->store_file(file, sink->context)) + goto receive_error; + } else if (status == STATUS_MANIFEST) { + DeleteManifest* manifest = receive_manifest_entries(file_descriptor); + if (!manifest) + goto fail; /* receive_manifest_entries already sent STATUS_ERROR */ + if (early_delete) { + /* --delete-before / --delete-during: the manifest is authoritative the + moment it arrives, before any file data. Delete now and acknowledge + so the sender only starts streaming once the deletion committed (or + failed). This is the rsync delete-before/delete-during window: a + later transfer failure does not restore these deletions. */ + bool deletion_ok = (config->use_delete || config->delete_missing_args) + ? manifest_delete_all(config, manifest) + : true; + delete_manifest_free(manifest); + if (!deletion_ok) { + send_status(file_descriptor, STATUS_ERROR); + goto fail; + } + if (!send_status(file_descriptor, STATUS_OK)) + goto fail; + } else if (config->use_delete || config->delete_missing_args) { + /* Plain --delete / --delete-after / --delete-delay and the + --delete-missing-args exact-path deletions: hold the manifest and + commit it only after STATUS_FINISHED. */ + if (deferred_manifest) { + log_message(LOG_LEVEL_ERROR, "Received a second delete manifest"); + delete_manifest_free(deferred_manifest); + deferred_manifest = NULL; + delete_manifest_free(manifest); + send_status(file_descriptor, STATUS_ERROR); + goto fail; + } + deferred_manifest = manifest; + } else { + delete_manifest_free(manifest); + } + goto next_status; + } else { + File* file = file_receive(config, file_descriptor); + if (!file) { + log_message(LOG_LEVEL_ERROR, "Failed to receive file"); + goto receive_error; + } + if (!sink->store_file(file, sink->context)) + goto receive_error; + } + next_status: + if (!receive_status(file_descriptor, &status)) + goto receive_error; + } + if (status != STATUS_FINISHED) { + log_message(LOG_LEVEL_ERROR, "Did not receive FINISHED Status"); + goto receive_error; + } + /* Commit-style (late) deletion: every data frame has been received and the + sender proved the whole tree with STATUS_FINISHED. The single-threaded + receiver stores files synchronously, so everything is on disk here and the + deletion can be committed before the --delay-updates publication in + send_success (the walker skips the staging dir, so staged files are never + treated as extras). The -m receiver passes `pending_manifest` because its + disk writer may still be draining; the caller commits after the writer has + joined so no extra file is removed unless the transfer is known to have + succeeded. */ + if (deferred_manifest) { + if (pending_manifest) { + *pending_manifest = deferred_manifest; + deferred_manifest = NULL; + } else { + bool deletion_ok = manifest_delete_all(config, deferred_manifest); + delete_manifest_free(deferred_manifest); + deferred_manifest = NULL; + if (!deletion_ok) { + send_status(file_descriptor, STATUS_ERROR); + goto fail; + } + } + } + if (sink->send_success) { + if (sink->send_success_frame) { + if (!sink->send_success_frame(file_descriptor, sink->context)) + goto fail; + } else if (!send_status(file_descriptor, STATUS_OK)) { + goto fail; + } + } + return 0; + +fail: + /* Failure exits that must not (or already did) report a STATUS_ERROR. The + parked keep-set is dropped: never commit a deletion for a failed stream. */ + if (deferred_manifest) { + delete_manifest_free(deferred_manifest); + deferred_manifest = NULL; + } + return -1; + +receive_error: + if (deferred_manifest) { + delete_manifest_free(deferred_manifest); + deferred_manifest = NULL; + } + if (sink->send_error) + send_status(file_descriptor, STATUS_ERROR); + return -1; +} + +/* ---- Single-threaded sink (used by receiver_receive_files) ---- */ + +typedef struct { + Config* config; + ReceiverOutcomes outcomes; + /* P7 Wave D: directory metadata accumulated during the stream, applied only + after the whole transfer (and its delete/publication phases) has run so a + child write never clobbers a directory mtime. */ + DirTimeList dir_times; +} ReceiverSaveContext; + +static bool receiver_save_file(File* file, void* context_pointer) { + ReceiverSaveContext* context = context_pointer; + FileSaveResult result = FILE_SAVE_ERROR; + if (!context->config->save_to_disk) { + /* Nothing is stored; report the file as not-written so a + --remove-source-files sender keeps its source. */ + result = FILE_SAVE_SKIPPED; + } else { + result = file_save_to_disk_full(context->config->receive_root_directory, file, context->config); + } + /* A directory's times are deferred, never applied inline: collect the + metadata now and apply it at the end. -O/--omit-dir-times is honored by + dir_time_list_apply's caller (see receiver_send_success_frame). */ + if (result != FILE_SAVE_ERROR && file->is_dir && file->metadata && + context->config->use_metadata && !context->config->omit_dir_times && + !dir_time_list_add(&context->dir_times, file->path, file->metadata)) { + file_destroy(file); + return false; + } + if (result != FILE_SAVE_ERROR && context->config->remove_source_files && !file->is_dir && + !file->is_special && !file->skip && + !receiver_outcomes_append(&context->outcomes, (unsigned char)result)) { + file_destroy(file); + return false; + } + file_destroy(file); + return result != FILE_SAVE_ERROR; +} + +static bool receiver_send_success_frame(int fd, void* context_pointer) { + ReceiverSaveContext* context = context_pointer; + /* --delay-updates: the whole protocol stream (including manifest/delete + handling, which ran inside receiver_process) has succeeded and every + staged file was fully written. Publish them atomically now, before the + success/outcome frame tells a --remove-source-files sender it may delete + its sources. */ + if (context->config->delay_updates && context->config->delay_context) { + if (!delay_updates_publish(context->config->delay_context, context->config)) { + send_status(fd, STATUS_ERROR); + return false; + } + } + /* P7 Wave D: every child is now written and the delete / --delay-updates + phases have committed, so it is finally safe to stamp directory times. + This runs after the deferred deletion because receiver_process commits it + before calling this success frame. */ + dir_time_list_apply(&context->dir_times, context->config->receive_root_directory); + return receiver_send_final_success(fd, context->config, &context->outcomes); +} + +int receiver_receive_files(Config* config, int file_descriptor) { + ReceiverSaveContext context = {.config = config, .outcomes = {0}}; + dir_time_list_init(&context.dir_times); + ReceiverSink sink = {receiver_save_file, &context, true, true, receiver_send_success_frame}; + int ret = receiver_process(config, file_descriptor, &sink); + if (ret != 0 && config->delay_updates && config->delay_context) + delay_updates_cleanup(config->delay_context); + receiver_outcomes_destroy(&context.outcomes); + dir_time_list_free(&context.dir_times); + return ret; +} diff --git a/src/server/receiver.h b/src/server/receiver.h new file mode 100644 index 0000000..1b07e13 --- /dev/null +++ b/src/server/receiver.h @@ -0,0 +1,48 @@ +#ifndef RECEIVER_H +#define RECEIVER_H + +#include "config.h" +#include "file.h" +#include "file_receive.h" + +typedef bool (*ReceiverFileSink)(File* file, void* context); + +/* Ordered per-file save outcomes for one connection. One entry is appended + for every data-bearing file the receiver processes (in the order the files + were sent) so the sender of a --remove-source-files transfer can be told + which sources were actually written versus skipped on the receiver. */ +typedef struct { + unsigned char* entries; /* FILE_SAVE_WRITTEN or FILE_SAVE_SKIPPED */ + size_t count; + size_t capacity; +} ReceiverOutcomes; + +typedef bool (*ReceiverSuccessFrame)(int fd, void* context); + +typedef struct { + ReceiverFileSink store_file; + void* context; + bool send_error; + bool send_success; + /* Emits the end-of-transfer success frame. When the sender requested + --remove-source-files this includes one per-file status per processed + data file followed by the final STATUS_OK; otherwise just STATUS_OK. */ + ReceiverSuccessFrame send_success_frame; +} ReceiverSink; + +bool receiver_outcomes_append(ReceiverOutcomes* outcomes, unsigned char code); +void receiver_outcomes_destroy(ReceiverOutcomes* outcomes); +bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes); + +int receiver_process(Config* config, int file_descriptor, const ReceiverSink* sink); +/* receiver_process with an escape hatch for the commit-style (late) deletion: + when `pending_manifest` is non-NULL the receiver does NOT delete at + STATUS_FINISHED itself; instead it stores the owned keep-set manifest there + (leaving *pending_manifest untouched on early modes/errors) so the caller can + commit the deletion only after its disk writer has fully drained. Pass NULL + to keep the default behaviour (delete before the success frame). */ +int receiver_process_pending(Config* config, int file_descriptor, const ReceiverSink* sink, + DeleteManifest** pending_manifest); +int receiver_receive_files(Config* config, int file_descriptor); + +#endif diff --git a/src/server/server.c b/src/server/server.c index f8fdb54..dce8fee 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -1,129 +1,625 @@ -#include "array_list.h" -#include "chunk.h" #include "config.h" -#include "data.h" +#include "charset.h" +#include "credentials.h" +#include "daemon_conf.h" +#include "delay_updates.h" #include "file.h" +#include "identity.h" #include "log.h" +#include "motd.h" #include "multiprocessing.h" #include "protocol.h" #include "queue.h" +#include "receiver.h" +#include "server_cli.h" #include "transport_tcp.h" #include "transport_tls.h" -#include "unistd.h" #include "utils.h" +#include +#include #include #include #include #include +#include +#include +#include +#include -int receive_files(Config* config, int fd) { - Status status; - if (!receive_status(fd, &status)) - return -1; +static char* authorized_root; +static int authorized_root_fd = -1; +static bool allow_delete; +static bool trust_sender; +static bool allow_unauthenticated; +/* --no-super operator veto: forces SUPER_MODE_OFF for every connection (even + * root), so no super-user activity is attempted and any client --copy-as is + * refused. Set once in main before the accept loop / stdio handler. */ +static bool server_no_super; +static const char* required_client_cn; +/* --iconv CONVERT_SPEC the server was itself started with (borrowed argv + * pointer). Its LOCAL half may override the local charset the client assumed; + * see charset_wire_init_receiver. */ +static const char* server_iconv_spec; - while (status == STATUS_NEXT || status == STATUS_CHUNK || status == STATUS_CHECK || - status == STATUS_KEEPALIVE || status == STATUS_ABORT || status == STATUS_CHECK_BATCH) { - if (status == STATUS_KEEPALIVE) { - send_status(fd, STATUS_KEEPALIVE); - goto next; - } - if (status == STATUS_ABORT) { - log_message(LOG_LEVEL_INFO, "Received abort from client, cleaning up"); - return -1; - } - if (status == STATUS_CHECK) { - bool skipped; - File* file = receive_incremental_check(fd, config, &skipped); - if (skipped) - goto next; - if (file == NULL && !skipped) - return -1; - if (config->save_to_disk) - file_save_to_disk(config->receive_root_directory, file, NULL); - file_destroy(file); - } else if (status == STATUS_CHUNK) { - Chunk* chunk = receive_chunk_data(fd, config); - if (chunk == NULL) { - send_status(fd, STATUS_ERROR); - return -1; - } - for (int i = 0; i < chunk->element_count; i++) { - if (config->save_to_disk) - file_save_to_disk(config->receive_root_directory, chunk->items[i], NULL); - } - chunk_destroy(chunk); - } else if (status == STATUS_CHECK_BATCH) { - int count; - if (!receive_int(fd, &count)) - return -1; - for (int i = 0; i < count; i++) { - char* check_path = receive_str(fd); - if (!check_path) - return -1; - unsigned long long check_size; - long long check_mtime; - if (!receive_n_data(fd, &check_size, sizeof(check_size)) || - !receive_n_data(fd, &check_mtime, sizeof(check_mtime))) { - free(check_path); - return -1; - } - char* full_path = path_cat(config->receive_root_directory, check_path); - struct stat st; - bool has_old = full_path && lstat(full_path, &st) == 0; - bool match = has_old && (unsigned long long)st.st_size == check_size && - (long long)st.st_mtime == check_mtime; - if (match) - send_status(fd, STATUS_OK); - else - send_status(fd, STATUS_NEXT); - free(full_path); - free(check_path); - } - goto next; - } else { - File* file = file_receive(config, fd); - if (file == NULL) { - log_message(LOG_LEVEL_ERROR, "Failed to receive file"); - send_status(fd, STATUS_ERROR); - return -1; - } - if (config->save_to_disk) - file_save_to_disk(config->receive_root_directory, file, NULL); - file_destroy(file); - } - next: - if (!receive_status(fd, &status)) { - send_status(fd, STATUS_ERROR); - return -1; - } - } +/* Non-NULL exactly when the listener runs in --daemon mode. Loaded once in + * main before any accept-loop fork, then shared read-only by every forked + * connection child (and their threads). */ +static DaemonConf* g_daemon_conf = NULL; - if (status == STATUS_MANIFEST) { - if (receive_manifest(fd, config, &status) != 0) - return -1; +/* Daemon credential store (Wave B), loaded once in main from --password-file / + * --early-input and shared read-only by every forked connection child. When a + * module declares `auth users` but no store was configured, the daemon refuses + * to start (fail closed); the store is never NULL after a successful start when + * such a module exists. */ +static CredentialStore* g_credentials = NULL; + +/* Opaque context threaded through to the config-frame gate: the connection's + * SSL object (NULL over plaintext) so the gate can warn when a credential + * exchange is not encrypted, plus the super-mode override the gate decides on. + * The gate never mutates the received (const) Config; it records a forced + * SUPER_MODE_OFF here and the handler applies it exactly once after acceptance. */ +typedef struct ModuleGateContext { + SSL* ssl; + /* The connection descriptor, so the gate can drive the SCRAM auth handshake + * while it still owns the config-frame exchange (before the STATUS_OK ack). */ + int fd; + /* SUPER_MODE_OFF when this connection must not attempt any super-user + activity (operator --no-super, or a daemon module without the + `client owner = yes` opt-in); -1 when the config's own mode stands. */ + int super_mode_override; +} ModuleGateContext; + +/* Server half of the SCRAM challenge/response (A7 remediation, protocol + * 2.19.0). Sends STATUS_AUTH_CHALLENGE (iteration count, base64 salt, base64 + * server nonce), expects STATUS_AUTH_RESPONSE (base64 client nonce, base64 + * ClientProof), verifies the proof constant-time and answers STATUS_AUTH_OK + * with the base64 ServerSignature. On any failure BEFORE the success response + * it sends exactly one generic STATUS_AUTH_FAILED and returns false; a failure + * while writing the success signature cannot send a status and just drops an + * already-broken connection. The verifier for an unknown/off-list + * user is a dummy (deterministic per-username salt, store-wide iterations, dummy + * keys, found=false) so the same math runs and no user-enumeration/timing oracle + * is exposed. */ +static bool server_auth_handshake(int fd, const Config* config, const DaemonModule* module) { + bool result = false; + CredentialVerifier verifier; + memset(&verifier, 0, sizeof(verifier)); + uint8_t snonce[CREDENTIAL_NONCE_LEN] = {0}; + char salt_b64[25] = {0}; + char snonce_b64[45] = {0}; + char* cnonce_b64 = NULL; + char* proof_b64 = NULL; + uint8_t cnonce[CREDENTIAL_NONCE_LEN] = {0}; + uint8_t proof[CREDENTIAL_KEY_LEN] = {0}; + uint8_t server_sig[CREDENTIAL_KEY_LEN] = {0}; + char sig_b64[45] = {0}; + size_t cnonce_len = 0; + size_t proof_len = 0; + + if (!config->auth_user) + goto fail; /* no username: generic failure, no challenge */ + if (!credentials_get_verifier(g_credentials, config->auth_user, + (const char* const*)module->auth_users, module->auth_user_count, + &verifier)) + goto fail; /* a crypto failure still owes the gate a terminal frame */ + if (!(credentials_random_bytes(snonce, sizeof(snonce)) && + credentials_b64_encode(verifier.salt, CREDENTIAL_SALT_LEN, salt_b64, sizeof(salt_b64)) && + credentials_b64_encode(snonce, sizeof(snonce), snonce_b64, sizeof(snonce_b64)))) + goto fail; + if (!(send_status(fd, STATUS_AUTH_CHALLENGE) && send_int(fd, (int)verifier.iters) && + send_str(fd, salt_b64) && send_str(fd, snonce_b64))) + goto fail; + + Status status = STATUS_ERROR; + if (!(receive_status(fd, &status) && status == STATUS_AUTH_RESPONSE)) + goto fail; + cnonce_b64 = receive_str_redacted(fd); + proof_b64 = receive_str_redacted(fd); + if (!(cnonce_b64 && proof_b64 && + credentials_b64_decode(cnonce_b64, cnonce, sizeof(cnonce), &cnonce_len) && + cnonce_len == CREDENTIAL_NONCE_LEN && + credentials_b64_decode(proof_b64, proof, sizeof(proof), &proof_len) && + proof_len == CREDENTIAL_KEY_LEN)) + goto fail; + if (!credentials_verify_response(&verifier, config->auth_user, snonce, cnonce, proof, server_sig)) + goto fail; + + /* Success writes exactly one terminal frame (STATUS_AUTH_OK). A broken pipe + * while sending the signature just drops the connection; it must never emit a + * second terminal status. */ + result = credentials_b64_encode(server_sig, sizeof(server_sig), sig_b64, sizeof(sig_b64)) && + send_status(fd, STATUS_AUTH_OK) && send_str_redacted(fd, sig_b64); + goto cleanup; + +fail: + /* Every failure path writes exactly one generic terminal status, satisfying + * the gate's CONFIG_VALIDATE_ALREADY_TERMINATED contract. */ + send_status(fd, STATUS_AUTH_FAILED); + +cleanup: + credentials_burn(cnonce_b64, cnonce_b64 ? strlen(cnonce_b64) : 0); + credentials_burn(proof_b64, proof_b64 ? strlen(proof_b64) : 0); + free(cnonce_b64); + free(proof_b64); + credentials_burn((char*)snonce, sizeof(snonce)); + credentials_burn(salt_b64, sizeof(salt_b64)); + credentials_burn(snonce_b64, sizeof(snonce_b64)); + credentials_burn((char*)cnonce, sizeof(cnonce)); + credentials_burn((char*)proof, sizeof(proof)); + credentials_burn((char*)server_sig, sizeof(server_sig)); + credentials_burn(sig_b64, sizeof(sig_b64)); + credentials_burn((char*)verifier.salt, sizeof(verifier.salt)); + credentials_burn((char*)verifier.stored_key, sizeof(verifier.stored_key)); + credentials_burn((char*)verifier.server_key, sizeof(verifier.server_key)); + return result; +} + +/* Aggregate payload bytes the multithreaded receiver may buffer ahead of the + slow disk writer. Receiving one more chunk adds up to ~2 * MAX_CHUNK_SIZE + of transient wire/decompression buffers on top of the queued payloads, so + this ceiling keeps total per-connection receive memory (decompressed and + per-file copied chunk buffers included) within MAX_CONNECTION_MEMORY. */ +#define RECEIVER_QUEUE_MAX_BYTES (MAX_CONNECTION_MEMORY - 2 * MAX_CHUNK_SIZE) + +static bool tls_client_identity_allowed(SSL* ssl) { + if (!ssl || !required_client_cn) + return false; + X509* certificate = SSL_get1_peer_certificate(ssl); + if (!certificate) + return false; + char common_name[256]; + int length = X509_NAME_get_text_by_NID(X509_get_subject_name(certificate), NID_commonName, + common_name, sizeof(common_name)); + size_t required_length = strlen(required_client_cn); + bool allowed = length >= 0 && (size_t)length == required_length && + required_length < sizeof(common_name) && + credentials_secure_equal(common_name, required_client_cn, required_length); + X509_free(certificate); + return allowed; +} + +static void release_authorization(void) { + file_set_authorized_root(-1, NULL); + utils_set_authorized_root_fd(-1); + if (authorized_root_fd >= 0) + close(authorized_root_fd); + authorized_root_fd = -1; + free(authorized_root); + authorized_root = NULL; +} + +static bool path_is_within(const char* root, const char* path) { + size_t n = strlen(root); + return strncmp(root, path, n) == 0 && (path[n] == '\0' || path[n] == '/'); +} + +/* --mkpath contract: when the client's destination root directory does not + exist yet on the server side, --mkpath tells the server to create it (and + any missing leading components) below the authorized root at connection + start. Without --mkpath the destination root must already exist: a missing + root is rejected up front instead of being silently invented by a later + write. Both paths are confined to the authorized root by the secure file + helpers. */ +static bool ensure_receive_root(const Config* config) { + if (!config || !config->receive_root_directory) + return false; + if (config->mkpath) + return file_ensure_directory_secure(config->receive_root_directory); + return file_directory_exists_secure(config->receive_root_directory); +} + +static bool configure_authorization(const char* root) { + char resolved[PATH_MAX]; + if (!root) { + file_set_authorized_root(-1, NULL); + utils_set_authorized_root(-1, NULL); + return false; } - if (status != STATUS_FINISHED) { - log_message(LOG_LEVEL_ERROR, "Did not receive FINISHED Status"); - send_status(fd, STATUS_ERROR); - return -1; + int root_fd = open(root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + if (root_fd < 0) { + file_set_authorized_root(-1, NULL); + utils_set_authorized_root(-1, NULL); + return false; } - send_status(fd, STATUS_OK); - return 0; + char fd_path[64]; + int fd_path_length = snprintf(fd_path, sizeof(fd_path), "/proc/self/fd/%d", root_fd); + if (fd_path_length < 0 || (size_t)fd_path_length >= sizeof(fd_path) || + !realpath(fd_path, resolved)) { + close(root_fd); + file_set_authorized_root(-1, NULL); + utils_set_authorized_root(-1, NULL); + return false; + } + authorized_root = str_dup(resolved); + if (!authorized_root) { + close(root_fd); + file_set_authorized_root(-1, NULL); + utils_set_authorized_root(-1, NULL); + return false; + } + authorized_root_fd = root_fd; + if (!file_set_authorized_root(authorized_root_fd, authorized_root) || + !utils_set_authorized_root(authorized_root_fd, authorized_root)) { + file_set_authorized_root(-1, NULL); + utils_set_authorized_root(-1, NULL); + close(authorized_root_fd); + authorized_root_fd = -1; + free(authorized_root); + authorized_root = NULL; + return false; + } + return true; +} + +/* Config-frame gate (runs inside config_receive_with_validate, BEFORE the + * STATUS_OK ack, so a rejected connection is refused at the config handshake + * and no file data is ever exchanged). + * + * Plain mode: a connection that carries a daemon module name is refused (the + * standalone server simply does not offer modules; honouring one would silently + * change what the destination means). Empty module -> accept. + * + * Daemon mode: the client MUST select a module (host::module/path). The + * requested module is looked up in the daemon config and its configured `path` + * becomes the authorized root via configure_authorization -- exactly the same + * root confinement the standalone server applies to its single + * --destination-root, but per-module and NEVER client-chosen. The module is + * refused (with a clear log) when it is unknown, when it is `read only` (every + * FastSync network transfer writes; there is no read-only wire operation yet), + * when it requests client-chosen ownership without the module's + * `client owner = yes` opt-in (P7 Wave E hardening), or when the presented + * daemon credentials fail for a module that declares `auth users`. Wave A + * refused every auth-required module (auth was not yet implemented); Wave B + * authenticates the client instead (see below). */ +static const char* server_module_gate(const Config* config, void* context) { + ModuleGateContext* gate_ctx = (ModuleGateContext*)context; + if (!config) + return "missing config frame"; + /* Operator veto: --no-super forces SUPER_MODE_OFF for this connection before + the copy-as gate is evaluated. The received config is const, so the gates + below evaluate a shallow effective copy (only super_mode differs); the + handler applies the recorded override to the accepted config exactly once. */ + Config effective = *config; + if (server_no_super) { + effective.super_mode = SUPER_MODE_OFF; + if (gate_ctx) + gate_ctx->super_mode_override = SUPER_MODE_OFF; + } + /* --copy-as (P7 Wave E, protocol 2.18.0): FastSync's safe subset forces the + ownership of every written entry to the requested ids, which needs a + privileged (root) receiver. An unprivileged receiver REFUSES the whole + transfer here, at the config handshake and BEFORE the STATUS_OK ack, so no + file data is exchanged and there is never a silent wrong-ownership result. + The daemon's per-module client-chosen-ownership refusal is enforced after + the module lookup below (it needs the module's opt-in) and covers --copy-as + like every other ownership flag. */ + if (identity_copy_as_refused(&effective)) { + if (geteuid() != 0) + log_message(LOG_LEVEL_ERROR, "--copy-as requires a privileged receiver (root); refusing"); + else + log_message(LOG_LEVEL_ERROR, + "--copy-as refused: super-user activities are disabled by the server " + "(--no-super); refusing"); + return "cannot perform --copy-as on this receiver"; + } + /* --iconv (protocol 2.16.0): the receiver's exact conversion direction (the + client spec's wire charset into this server's local charset, including a + server-side --iconv override) must be usable BEFORE the STATUS_OK ack, so + an impossible conversion is refused at the handshake instead of failing + the first file mid-transfer. The client spec itself was already sanity + checked by validate_received_config. */ + if (config->iconv_spec && + !charset_wire_receiver_spec_valid(config->iconv_spec, server_iconv_spec)) + return "client --iconv conversion cannot be honored by this server"; + bool is_daemon = g_daemon_conf != NULL; + bool has_module = config->module != NULL && config->module[0] != '\0'; + + if (!is_daemon) { + if (has_module) + return "client requested a daemon module but this server is not running " + "with --daemon"; + return NULL; + } + if (!has_module) + return "daemon connection did not select a module (expected a " + "host::module/path destination)"; + + const DaemonModule* module = daemon_conf_find_module(g_daemon_conf, config->module); + if (module == NULL) { + char* escaped_module = output_escape(config->module, config->eight_bit_output); + log_message(LOG_LEVEL_ERROR, "unknown daemon module '%s' requested", + escaped_module ? escaped_module : ""); + free(escaped_module); + return "requested daemon module does not exist"; + } + if (module->read_only) { + log_message(LOG_LEVEL_ERROR, "daemon module '%s' is read only; refusing write transfer", + config->module); + return "requested daemon module is read only"; + } + /* Client-chosen ownership / super-user policy (P7 Wave E hardening): a daemon + module refuses EVERY ownership-affecting request (--numeric-ids, --chown, + --usermap/--groupmap, --fake-super, --copy-as, explicit --super) unless the + operator opted THIS module in with `client owner = yes`. Otherwise any + client could force arbitrary ownership inside the module root. The + standalone/SSH server has a single operator-authorized root and keeps + honoring these. */ + if (!module->client_owner) { + /* Ownership: refuse the whole transfer up front (a clear failure). + Evaluated against the ORIGINAL config so an explicit --super is refused + even when an operator --no-super veto already forced the effective copy + to OFF (the veto must not silently convert a refusal into an accept). */ + if (identity_ownership_requested(config)) { + log_message(LOG_LEVEL_ERROR, + "daemon module '%s' refuses client-chosen ownership/super-user activities " + "(no `client owner = yes` opt-in); refusing", + config->module); + return "client-chosen ownership is not permitted by this daemon module"; + } + /* Super-user DEVICE activities (char/block mknod and --write-devices) are + permitted under the default AUTO mode, so without this override a root + daemon would still let a non-opted module create arbitrary device nodes + and write raw devices. Force them off for this connection: those entries + are skipped (never mknod'ed) while an ordinary `-a` push still succeeds + without device nodes, matching the operator's least-privilege choice. + The operator-level --no-super veto is already folded into this. */ + if (gate_ctx) + gate_ctx->super_mode_override = SUPER_MODE_OFF; + } + if (module->auth_user_count > 0) { + /* Auth-required module (A7, protocol 2.19.0): run the SCRAM challenge/ + * response BEFORE the module root is installed and before any data moves. + * Fail closed: no store -> refuse (server misconfiguration, STATUS_ERROR); + * a handshake that fails before the success response writes exactly one + * STATUS_AUTH_FAILED before signalling ALREADY_TERMINATED (a failure while + * writing the success signature instead just drops the broken connection). + * The username may be logged (never the password or any derived proof). */ + if (g_credentials == NULL) { + log_message(LOG_LEVEL_ERROR, + "daemon module '%s' requires authentication but no credential store is " + "configured (--password-file/--early-input); refusing", + config->module); + return "requested daemon module requires authentication and no credential " + "store is configured"; + } + /* Transport policy (A7-3/S1): an auth-required module only accepts + * credentials over (a) an encrypted, verified TLS connection whose client + * certificate matches --client-cn, or (b) an actual PLAINTEXT connection + * from a loopback peer that the operator explicitly opted into with + * --allow-unauthenticated. A remote plaintext peer, an un-flagged loopback + * plaintext peer, and a loopback TLS peer whose certificate does not match + * --client-cn are all refused HERE, before the challenge is sent, so an + * unverified client never receives a nonce: the loopback allowance requires + * !gate_ctx->ssl, so --tls + --allow-unauthenticated can never be used to + * bypass the client-CN check. The operator flag never permits REMOTE + * plaintext auth: remote peers still require verified TLS regardless. */ + bool tls_ok = gate_ctx && gate_ctx->ssl && SSL_get_verify_result(gate_ctx->ssl) == X509_V_OK && + tls_client_identity_allowed(gate_ctx->ssl); + bool local_ok = allow_unauthenticated && gate_ctx && !gate_ctx->ssl && gate_ctx->fd >= 0 && + utils_fd_peer_is_local(gate_ctx->fd); + if (!tls_ok && !local_ok) { + log_message(LOG_LEVEL_ERROR, + "daemon module '%s' requires authentication over an encrypted, verified TLS " + "connection (or an opted-in loopback plaintext transport); refusing", + config->module); + return "daemon module requires authentication over an encrypted, verified TLS " + "connection"; + } + /* Belt-and-braces: the transport policy above already guarantees a context + * with a usable socket (verified TLS implies a live SSL object and loopback + * allowance requires gate_ctx->fd >= 0), so this is unreachable today; keep + * the guard so the handshake can never be driven over an invalid fd. */ + if (!gate_ctx || gate_ctx->fd < 0) { + log_message(LOG_LEVEL_ERROR, "daemon module '%s': no auth transport available", + config->module); + return "authentication failed for the requested daemon module"; + } + if (!server_auth_handshake(gate_ctx->fd, config, module)) { + char* escaped_user = + config->auth_user ? output_escape(config->auth_user, config->eight_bit_output) : NULL; + log_message(LOG_LEVEL_ERROR, "daemon module '%s': authentication failed for user '%s'", + config->module, escaped_user ? escaped_user : "(none)"); + free(escaped_user); + return CONFIG_VALIDATE_ALREADY_TERMINATED; + } + char* escaped_user = output_escape(config->auth_user, config->eight_bit_output); + log_message(LOG_LEVEL_INFO, "daemon module '%s': user '%s' authenticated", config->module, + escaped_user ? escaped_user : ""); + free(escaped_user); + } + if (!configure_authorization(module->path)) { + log_message(LOG_LEVEL_ERROR, "daemon module '%s' path '%s' is not usable", config->module, + module->path ? module->path : "(null)"); + return "requested daemon module root is not usable"; + } + return NULL; /* accepted; authorized root is now the module's path */ } void handler(int file_descriptor) { SSL* ssl = io_get_ssl(); - Config* config = config_receive(file_descriptor); + ProtocolSession session; + protocol_session_init(&session, file_descriptor, file_descriptor); + protocol_session_set_ssl(&session, ssl); + protocol_session_bind(&session); + ModuleGateContext gate_ctx; + gate_ctx.ssl = ssl; + gate_ctx.fd = file_descriptor; + gate_ctx.super_mode_override = -1; + Config* config = config_receive_with_validate(file_descriptor, server_module_gate, &gate_ctx); if (config == NULL) { log_message(LOG_LEVEL_ERROR, "Failed to receive config"); close(file_descriptor); + protocol_session_unbind(); return; } + /* Apply the super-mode veto the gate decided on (operator --no-super, or a + * daemon module without the `client owner = yes` opt-in) exactly once, so + * every downstream gate (identity_apply_ownership via privilege_super_permitted, + * device-node creation) sees SUPER_MODE_OFF. The gate never mutated the + * received config. */ + if (gate_ctx.super_mode_override != -1) + config->super_mode = gate_ctx.super_mode_override; + protocol_set_8_bit_output(config->eight_bit_output); + if (!authorized_root) { + log_message(LOG_LEVEL_ERROR, "No server-side destination root configured"); + config_delete(config); + close(file_descriptor); + protocol_session_unbind(); + return; + } + if (!allow_unauthenticated && ssl == NULL) { + log_message(LOG_LEVEL_ERROR, "Rejected unauthenticated plaintext connection"); + config_delete(config); + close(file_descriptor); + protocol_session_unbind(); + return; + } + if (ssl && required_client_cn && !tls_client_identity_allowed(ssl)) { + log_message(LOG_LEVEL_ERROR, "Rejected TLS client with unauthorized identity"); + config_delete(config); + close(file_descriptor); + return; + } + /* Daemon mode: the module's root is the authorized root (installed by + server_module_gate), and the client's destination is a MODULE-RELATIVE + path. Reject an absolute destination up front so the module-relative + confinement contract is never eroded by a client that tries to address the + module root by absolute path. */ + if (g_daemon_conf && config->receive_root_directory && config->receive_root_directory[0] == '/') { + log_message(LOG_LEVEL_ERROR, "Rejected absolute daemon destination (must be relative to the " + "selected module root)"); + config_delete(config); + close(file_descriptor); + protocol_session_unbind(); + return; + } + char* destination = config->receive_root_directory; + char* joined_destination = NULL; + if (destination && destination[0] != '/') + joined_destination = path_cat(authorized_root, destination); + if (joined_destination) + destination = joined_destination; + if (!destination || has_path_traversal(destination) || + !path_is_within(authorized_root, destination)) { + log_message(LOG_LEVEL_ERROR, "Rejected destination outside authorized root"); + free(joined_destination); + config_delete(config); + close(file_descriptor); + return; + } + if (joined_destination) { + free(config->receive_root_directory); + config->receive_root_directory = joined_destination; + } + if (!config->receive_root_directory) { + config_delete(config); + close(file_descriptor); + protocol_session_unbind(); + return; + } + config->use_delete = config->use_delete && allow_delete; + /* --iconv (protocol 2.16.0): install the receiver-side wire->local conversion + now that the client's full CONVERT_SPEC has been received and validated, + before any received file name is decoded. The server's own --iconv (if + any) may override the local charset; a spec the client is known to have + validated cannot fail here unless the server's override names an + unsupported charset. */ + if (config->iconv_spec && !charset_wire_init_receiver(config->iconv_spec, server_iconv_spec)) { + log_message(LOG_LEVEL_ERROR, + "--iconv: unsupported charset conversion requested (LOCAL[,REMOTE])"); + config_delete(config); + close(file_descriptor); + protocol_session_unbind(); + return; + } + /* --delete-missing-args deletes destination mirrors receiver-side, so it is + deletion and stays gated by the same --allow-delete server policy. When + the server policy is off the flag is inert (the missing entries are still + skipped via its implied --ignore-missing-args, but nothing is deleted). */ + config->delete_missing_args = config->delete_missing_args && allow_delete; + /* --mkpath: create the destination root (and its missing leading components) + before anything else; without it the root must pre-exist. A failure here + aborts the connection cleanly before any file data is exchanged. */ + if (!ensure_receive_root(config)) { + char* escaped_root = output_escape(config->receive_root_directory, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "destination root is not available: %s", + escaped_root ? escaped_root : ""); + free(escaped_root); + config_delete(config); + close(file_descriptor); + protocol_session_unbind(); + return; + } + /* A --delay-updates transfer stages under a private 0700 directory inside + the receive root. Create it up front (wiping leftovers of any previously + interrupted delayed transfer) so a fully-skipped run also starts clean. */ + if (config->delay_updates) { + config->delay_context = delay_updates_context_create(config->receive_root_directory); + if (!config->delay_context || !delay_updates_prepare(config->delay_context)) { + log_message(LOG_LEVEL_ERROR, "Failed to initialize --delay-updates staging area"); + delay_updates_cleanup(config->delay_context); + config_delete(config); + close(file_descriptor); + protocol_session_unbind(); + return; + } + } + /* Preserve the negotiated identity policy for the fd-relative ownership + apply path. Each connection is its own forked process, so this + per-process snapshot never races another connection. A failed deep copy + (allocation failure) leaves the snapshot cleared, so refuse the connection + rather than silently applying the wrong ownership policy. */ + if (!identity_set_active(config)) { + log_message(LOG_LEVEL_ERROR, "Failed to activate identity policy"); + config_delete(config); + close(file_descriptor); + protocol_session_unbind(); + return; + } + /* Persist the negotiated --keep-dirlinks policy once, here at config-accept, + before any multithreaded receiver/writer threads are spawned, so the + fd-walk reads a stable value during the whole transfer (and never bleeds + across the per-connection forked processes). */ + file_set_keep_dirlinks(config->keep_dirlinks); + /* --trust-sender is a LOCAL receiver policy: it never crosses the wire (so a + wire peer can never enable it). The standalone server only honours it when + its own CLI was started with --trust-sender (the client forwards that switch + into the remote argv via --remote-option=--trust-sender; the server then + parses it here and applies the policy below). Set before any multithreaded + receiver/writer threads are spawned so the fd-walk reads a stable value + during the whole transfer, and never bleeds across the per-connection + forked processes. Off by default. */ + file_set_trust_sender(trust_sender); + /* Wave C MOTD: on the daemon listener path only, once the module gate + auth + have accepted and every destination check has passed, send the configured + `motd file` as the first server->client frame before any transfer data + (rsync sends its MOTD as the first thing from the server on a daemon + connection). Every daemon connection gets the frame -- an unset or + unreadable motd file sends an empty string -- so the client's read is + deterministic and an absent file is never an error. The --stdio SSH path + has no MOTD (g_daemon_conf is NULL there). No PROTOCOL_VERSION bump: the + frame is symmetric server->client in every 2.15.0 daemon build (see the + Wave C note in config.h). */ + if (g_daemon_conf) { + char* motd = motd_read_file(g_daemon_conf->global.motd_file); + if (!motd_send(file_descriptor, motd ? motd : "")) { + free(motd); + log_message(LOG_LEVEL_ERROR, "Failed to send daemon MOTD"); + config_delete(config); + close(file_descriptor); + protocol_session_unbind(); + identity_clear_active(); + return; + } + free(motd); + } if (config->use_multithreading) { Queue* q = queue_create(100, file_destroy); if (q == NULL) { config_delete(config); close(file_descriptor); + protocol_session_unbind(); + identity_clear_active(); return; } PipelineContextReceiver* context = @@ -132,24 +628,95 @@ void handler(int file_descriptor) { queue_destroy(q); config_delete(config); close(file_descriptor); + protocol_session_unbind(); + identity_clear_active(); return; } + protocol_session_set_max_alloc(&context->session, config->max_alloc); + atomic_store(&context->session.total_allocated_bytes, + atomic_load(&session.total_allocated_bytes)); + pipeline_context_receiver_set_queue_byte_limit(context, RECEIVER_QUEUE_MAX_BYTES); thrd_t receiver, writer; - if (thrd_create(&receiver, receive_thread, context) != thrd_success || - thrd_create(&writer, write_thread, context) != thrd_success) { - perror("Error creating Threads"); + bool receiver_created = thrd_create(&receiver, receive_thread, context) == thrd_success; + bool writer_created = false; + if (receiver_created) + writer_created = thrd_create(&writer, write_thread, context) == thrd_success; + if (!receiver_created || !writer_created) { + log_perror("Error creating Threads"); + if (receiver_created) { + mtx_lock(&context->mutex); + atomic_store(&context->cancelled, true); + cnd_broadcast(&context->condition_not_full); + cnd_broadcast(&context->condition_not_empty); + mtx_unlock(&context->mutex); + close(file_descriptor); + thrd_join(receiver, NULL); + } else { + close(file_descriptor); + } + if (writer_created) + thrd_join(writer, NULL); pipeline_context_receiver_destroy(context); - close(file_descriptor); + protocol_session_unbind(); + identity_clear_active(); return; } - thrd_join(receiver, NULL); - thrd_join(writer, NULL); - send_status(file_descriptor, STATUS_OK); + int receiver_result; + int writer_result; + thrd_join(receiver, &receiver_result); + thrd_join(writer, &writer_result); + bool transfer_ok = receiver_result == thrd_success && writer_result == thrd_success; + if (transfer_ok) { + /* Commit-style (late) deletion: receive_thread handed the keep-set + manifest here instead of deleting while write_thread might still be + draining, so by now every file is on disk and the whole transfer is + known to have succeeded. Remove the extras before publishing a + --delay-updates run; the walker skips the staging directory. */ + if (context->deferred_manifest) { + if (!manifest_delete_all(config, context->deferred_manifest)) { + transfer_ok = false; + } + delete_manifest_free(context->deferred_manifest); + context->deferred_manifest = NULL; + } + } + if (transfer_ok) { + /* --delay-updates: receive_thread has finished the whole protocol stream + (including manifest/delete handling) and write_thread has drained its + queue, so every staged file is complete. Publish atomically before the + success/outcome frame so a --remove-source-files sender only learns of + files that were actually installed. */ + if (config->delay_updates && config->delay_context && + !delay_updates_publish(config->delay_context, config)) { + transfer_ok = false; + } + /* P7 Wave D: all writers have joined and the late deletion (and + --delay-updates publication) has committed above, so it is finally safe + to stamp directory times; a directory's mtime must not be clobbered by + its children or by an extra removal. */ + if (transfer_ok) + dir_time_list_apply(&context->dir_times, config->receive_root_directory); + } + if (transfer_ok) { + if (!receiver_send_final_success(file_descriptor, config, &context->outcomes)) + transfer_ok = false; + } else { + send_status(file_descriptor, STATUS_ERROR); + } + if (!transfer_ok) { + log_message(LOG_LEVEL_ERROR, "Transfer failed"); + if (config->delay_updates && config->delay_context) + delay_updates_cleanup(config->delay_context); + } pipeline_context_receiver_destroy(context); } else { - receive_files(config, file_descriptor); + if (receiver_receive_files(config, file_descriptor) != 0) + log_message(LOG_LEVEL_ERROR, "Transfer failed"); config_delete(config); } + protocol_session_unbind(); + identity_clear_active(); + charset_wire_free(); close(file_descriptor); } @@ -158,95 +725,345 @@ static Server* g_server = NULL; static void cleanup(int sig) { (void)sig; - if (g_server) { + if (g_server) server_delete(&g_server); - } + daemon_conf_free(g_daemon_conf); + g_daemon_conf = NULL; + credentials_free(g_credentials); + g_credentials = NULL; _exit(0); } static void print_server_usage(void) { printf("FastSync Server\n"); - printf("Usage: fastsync-server [options]\n"); - printf("\n"); + printf("Usage: fastsync-server [options]\n\n"); printf("Options:\n"); printf(" --stdio Run in stdio mode (SSH transport)\n"); + printf(" --daemon Run as a persistent daemon listener using a module\n"); + printf(" config file (-p/config port; default 873)\n"); + printf(" --config=FILE Daemon config file (default: ~/.config/fastsync/\n"); + printf(" fastsyncd.conf, else /etc/fastsyncd.conf)\n"); + printf(" --dparam=KEY=VALUE Override one global config key on the command line\n"); + printf(" (port, motd file, address)\n"); + printf(" --no-detach Stay in the foreground (default detaches to\n"); + printf(" background when running --daemon)\n"); + printf(" --password-file=FILE Credential store for modules that declare\n"); + printf(" 'auth users' (line format:\n"); + printf(" user:$fastsync$1$pbkdf2-sha256$iters$salt$stored$server,\n"); + printf(" generated by --hash-credentials). Legacy\n"); + printf(" user:SHA256HEX lines are rejected. Requires\n"); + printf(" --daemon; an auth-required module with no store\n"); + printf(" refuses to start\n"); + printf(" --early-input=FILE Second credential store layered over\n"); + printf(" --password-file (same format); usually a secrets-\n"); + printf(" manager/process-substitution file. Requires --daemon\n"); printf(" -p TCP port (default: 8080, range: 1-65535)\n"); printf(" --tls Enable TLS encryption\n"); printf(" --cert TLS certificate file (PEM)\n"); printf(" --key TLS private key file (PEM)\n"); printf(" --ca TLS CA certificate file (PEM)\n"); + printf(" --client-cn TLS client certificate CN (mandatory with --tls)\n"); + printf(" --destination-root Authorized destination root (default: .)\n"); + printf(" --address Bind the listening socket to this address\n"); + printf(" -4, --ipv4 Bind an IPv4 socket (default)\n"); + printf(" -6, --ipv6 Bind an IPv6 socket\n"); + printf(" --allow-delete Permit manifest deletion\n"); + printf(" --trust-sender Trust the remote sender's file list\n"); + printf(" --no-super Operator veto: never attempt super-user activities\n"); + printf(" (ownership, device nodes) even as root, and refuse\n"); + printf(" any client --copy-as/--super request\n"); + printf(" --iconv=LOCAL[,REMOTE] Declare this server's LOCAL charset for file-name\n"); + printf(" conversion: received names are translated to this\n"); + printf(" charset (the wire charset still comes from the\n"); + printf(" client's CONVERT_SPEC). A name that cannot be\n"); + printf(" represented fails the run cleanly\n"); + printf(" --allow-unauthenticated Allow plaintext/anonymous network clients\n"); + printf(" (an auth-required module still accepts only opted-in\n"); + printf(" loopback plaintext; remote auth requires verified TLS)\n"); + printf(" --hash-credentials Read 's user:password lines and print\n"); + printf(" PBKDF2 credential-store lines to stdout, then exit.\n"); + printf(" Use the output as --password-file for --daemon;\n"); + printf(" redirect it to an owner-only (0600) file\n"); + printf(" --iterations N PBKDF2 iteration count for --hash-credentials\n"); + printf(" (default %u, range %u-%u)\n", CREDENTIAL_DEFAULT_ITERS, + CREDENTIAL_MIN_ITERS, CREDENTIAL_MAX_ITERS); printf(" -v, --verbose Enable debug logging\n"); printf(" --help Show this help\n"); } +/* Resolve the daemon config default: ~/.config/fastsync/fastsyncd.conf when it + * exists (or when HOME is set), otherwise /etc/fastsyncd.conf. Returns a + * pointer to a static buffer (never NULL). */ +static const char* default_daemon_config_path(void) { + const char* home = getenv("HOME"); + if (home && *home) { + static char user_path[PATH_MAX]; + int n = snprintf(user_path, sizeof(user_path), "%s/.config/fastsync/fastsyncd.conf", home); + if (n > 0 && (size_t)n < sizeof(user_path) && access(user_path, R_OK) == 0) + return user_path; + } + /* Fall back to the traditional system path. */ + return "/etc/fastsyncd.conf"; +} + +/* Detach from the controlling terminal: fork, exit the parent, and make the + * surviving child a session leader (setsid) with stdio redirected to + * /dev/null. The listening socket is already open (bound in main before this + * runs), so it is inherited by the background daemon. Returns true on + * success (in the daemon's own process). */ +static bool daemonize(void) { + pid_t pid = fork(); + if (pid < 0) + return false; + if (pid > 0) + _exit(0); + if (setsid() < 0) + return false; + pid = fork(); + if (pid < 0) + return false; + if (pid > 0) + _exit(0); + int devnull = open("/dev/null", O_RDWR); + if (devnull >= 0) { + dup2(devnull, STDIN_FILENO); + dup2(devnull, STDOUT_FILENO); + dup2(devnull, STDERR_FILENO); + if (devnull > STDERR_FILENO) + close(devnull); + } + /* Do not pin the launch CWD (module-relative 'path' entries would resolve + * against an unstable working directory) and drop the restrictive host umask + * so modules can create files/dirs with the modes the config requests. */ + if (chdir("/") != 0) + log_message(LOG_LEVEL_WARNING, "daemon: chdir to / failed: %s", strerror(errno)); + umask(0); + return true; +} + int main(int argc, char* argv[]) { - bool use_tls = false; - char* tls_cert = NULL; - char* tls_key = NULL; - char* tls_ca = NULL; - int port = 8080; - - signal(SIGPIPE, SIG_IGN); - for (int i = 1; i < argc; i++) { - if (strcmp(argv[i], "--help") == 0) { - print_server_usage(); - return 0; - } else if (strcmp(argv[i], "--stdio") == 0) { - io_set_fds(STDIN_FILENO, STDOUT_FILENO); - handler(STDIN_FILENO); - return 0; - } else if (strcmp(argv[i], "-v") == 0 || strcmp(argv[i], "--verbose") == 0) { - set_log_level(LOG_LEVEL_DEBUG); - } else if (strcmp(argv[i], "--tls") == 0) { - use_tls = true; - } else if (strcmp(argv[i], "--cert") == 0 && i + 1 < argc) { - tls_cert = argv[++i]; - } else if (strcmp(argv[i], "--key") == 0 && i + 1 < argc) { - tls_key = argv[++i]; - } else if (strcmp(argv[i], "--ca") == 0 && i + 1 < argc) { - tls_ca = argv[++i]; - } else if (strcmp(argv[i], "-p") == 0 && i + 1 < argc) { - char* end; - long p = strtol(argv[++i], &end, 10); - if (*end || p <= 0 || p > 65535) { - fprintf(stderr, "Error: invalid port '%s' (must be 1-65535)\n", argv[i]); - return 1; - } - port = (int)p; - } else if (argv[i][0] == '-') { - fprintf(stderr, "Unknown option: %s\n", argv[i]); - print_server_usage(); - return 1; - } + ServerCliOptions opts; + char cli_err[512]; + int parse_result = server_cli_parse(argc, argv, &opts, cli_err, sizeof(cli_err)); + if (parse_result == 1) { + print_server_usage(); + return 0; } - - if (tls_ca && !use_tls) { - log_message(LOG_LEVEL_WARNING, "--ca has no effect without --tls"); - } - - signal(SIGINT, cleanup); - signal(SIGTERM, cleanup); - g_server = server_create(port); - if (g_server == NULL) { - log_message(LOG_LEVEL_ERROR, "Failed to create server"); + if (parse_result < 0) { + server_cli_options_free(&opts); + fprintf(stderr, "Error: %s\n", cli_err); + print_server_usage(); return 1; } - if (use_tls) { - if (!tls_cert || !tls_key) { - fprintf(stderr, "Error: --tls requires --cert and --key\n"); - server_delete(&g_server); + + /* --hash-credentials: standalone offline tool; read user:password lines and + * emit new-format credential-store lines, then exit. */ + if (opts.hash_credentials_file) { + uint32_t iters = opts.hash_iterations_set ? opts.hash_iterations : CREDENTIAL_DEFAULT_ITERS; + /* The output is secret material: if it is redirected to a regular file, + * warn when that file is group/other-accessible (the store must be 0600). */ + struct stat out_st; + if (fstat(STDOUT_FILENO, &out_st) == 0 && S_ISREG(out_st.st_mode) && + (out_st.st_mode & (S_IRWXG | S_IRWXO)) != 0) + fprintf(stderr, + "Warning: credential-store output is a group/other-accessible file; restrict it to " + "mode 0600 (chmod 600)\n"); + char hash_err[512]; + if (credentials_hash_file(opts.hash_credentials_file, iters, stdout, hash_err, + sizeof(hash_err)) != 0) { + fprintf(stderr, "Error: %s\n", hash_err); + server_cli_options_free(&opts); return 1; } + server_cli_options_free(&opts); + return 0; + } + + int exit_code = 0; + signal(SIGPIPE, SIG_IGN); + if (opts.verbose) { + set_log_level(LOG_LEVEL_DEBUG); + set_log_debug_flags(LOG_DEBUG_ALL); + } + if (opts.tls_ca && !opts.use_tls) + log_message(LOG_LEVEL_WARNING, "--ca has no effect without --tls"); + /* Persist the parsed server policies into the process-global policy state + * BEFORE the stdio branch: an SSH-launched `--stdio` server (whose argv came + * from the client via --remote-option and friends) must honor --allow-delete, + * --trust-sender and --client-cn exactly like the standalone listener. */ + required_client_cn = opts.client_cn; + allow_delete = opts.allow_delete; + trust_sender = opts.trust_sender; + allow_unauthenticated = opts.allow_unauthenticated; + server_no_super = opts.no_super; + server_iconv_spec = opts.iconv_spec; + signal(SIGINT, cleanup); + signal(SIGTERM, cleanup); + + if (opts.stdio_mode) { + /* SSH authenticates the stdio transport outside of FastSync. */ + allow_unauthenticated = true; + if (!configure_authorization(opts.destination_root)) { + char* escaped = output_escape(opts.destination_root, false); + fprintf(stderr, "Error: invalid destination root '%s'\n", + escaped ? escaped : ""); + free(escaped); + server_cli_options_free(&opts); + return 1; + } + io_set_fds(STDIN_FILENO, STDOUT_FILENO); + handler(STDIN_FILENO); + release_authorization(); + server_cli_options_free(&opts); + return 0; + } + + int port = opts.port; + int bind_family = opts.bind_family; + const char* bind_address = opts.bind_address; + + if (opts.daemon_mode) { + const char* config_path = opts.config_path ? opts.config_path : default_daemon_config_path(); + g_daemon_conf = daemon_conf_load(config_path, cli_err, sizeof(cli_err)); + if (!g_daemon_conf) { + server_cli_options_free(&opts); + fprintf(stderr, "Error: %s\n", cli_err); + return 1; + } + for (int i = 0; i < opts.dparam_count; i++) { + if (daemon_conf_apply_dparam(g_daemon_conf, opts.dparams[i], cli_err, sizeof(cli_err)) != 0) { + fprintf(stderr, "Error: --dparam: %s\n", cli_err); + exit_code = 1; + goto out; + } + } + /* Effective port: -p (highest) > --dparam port > config port (default 873). */ + if (!opts.port_set) + port = g_daemon_conf->global.port; + if (!bind_address) + bind_address = g_daemon_conf->global.address; + if (g_daemon_conf->module_count == 0) + log_message(LOG_LEVEL_WARNING, + "daemon config has no modules; every connection will be refused"); + /* Surface the operator's client-chosen-ownership opt-in prominently: an + opted-in module lets its clients request arbitrary owner ids inside that + module root. */ + for (int i = 0; i < g_daemon_conf->module_count; i++) { + if (g_daemon_conf->modules[i].client_owner) + log_message(LOG_LEVEL_WARNING, + "daemon module '%s' allows client-chosen ownership and super-user device " + "activities (`client owner = yes`); clients may request arbitrary owner ids " + "and device nodes within that module root -- pair it with `auth users` " + "unless the module is intentionally open to the network", + g_daemon_conf->modules[i].name); + } + /* Daemon credential store (Wave B). --password-file and --early-input + * feed the same store, loaded BEFORE the listener forks so every + * connection child shares one read-only store. Fail closed at startup: a + * module that declares `auth users` without a store (or with an empty + * store) refuses to start rather than serving a module whose credentials + * can never be verified. */ + g_credentials = + credentials_load(opts.password_file, opts.early_input_file, cli_err, sizeof(cli_err)); + if (!g_credentials) { + server_cli_options_free(&opts); + fprintf(stderr, "Error: %s\n", cli_err); + return 1; + } + bool credential_source_given = opts.password_file != NULL || opts.early_input_file != NULL; + for (int i = 0; i < g_daemon_conf->module_count; i++) { + const DaemonModule* module = &g_daemon_conf->modules[i]; + if (module->auth_user_count == 0) + continue; + if (!credential_source_given) { + fprintf(stderr, + "Error: module '%s' declares 'auth users' but no credential store was given " + "(--password-file or --early-input); refusing to start (fail closed)\n", + module->name); + server_cli_options_free(&opts); + return 1; + } + if (credentials_store_size(g_credentials) == 0) { + fprintf(stderr, + "Error: module '%s' declares 'auth users' but the credential store is empty; " + "refusing to start (fail closed)\n", + module->name); + server_cli_options_free(&opts); + return 1; + } + for (int j = 0; j < module->auth_user_count; j++) { + if (!credentials_store_has(g_credentials, module->auth_users[j])) + log_message(LOG_LEVEL_WARNING, + "daemon module '%s': auth user '%s' has no credential store entry; that " + "user can never authenticate", + module->name, module->auth_users[j]); + } + } + } else { + if (!configure_authorization(opts.destination_root)) { + char* escaped = output_escape(opts.destination_root, false); + fprintf(stderr, "Error: invalid destination root '%s'\n", + escaped ? escaped : ""); + free(escaped); + server_cli_options_free(&opts); + return 1; + } + } + + ServerBindOptions bind_opts; + bind_opts.bind_address = bind_address; + bind_opts.family = bind_family; + g_server = server_create_ex(port, &bind_opts); + if (!g_server) { + log_message(LOG_LEVEL_ERROR, "Failed to create server"); + release_authorization(); + exit_code = 1; + goto out; + } + if (opts.use_tls) { + if (!opts.tls_cert || !opts.tls_key || !opts.tls_ca || !opts.client_cn) { + fprintf(stderr, "Error: --tls requires --cert, --key, --ca, and --client-cn\n"); + server_delete(&g_server); + release_authorization(); + exit_code = 1; + goto out; + } tls_global_init(); - if (!server_create_tls(g_server, tls_cert, tls_key, tls_ca)) { + if (!server_create_tls(g_server, opts.tls_cert, opts.tls_key, opts.tls_ca)) { log_message(LOG_LEVEL_ERROR, "Failed to set up TLS"); server_delete(&g_server); - return 1; + release_authorization(); + exit_code = 1; + goto out; } - server_listen_tls(g_server, handler); - } else { - server_listen(g_server, handler); } - return 0; + + /* Detach after the listening socket (and TLS context) exist so the + * background daemon inherits a fully-bound listener. --no-detach runs in + * the foreground, which is how tests drive the daemon. */ + if (opts.daemon_mode && !opts.no_detach) { + if (!daemonize()) { + log_message(LOG_LEVEL_ERROR, "Failed to daemonize"); + server_delete(&g_server); + release_authorization(); + exit_code = 1; + goto out; + } + } + + if (opts.use_tls) + server_listen_tls(g_server, handler); + else + server_listen(g_server, handler); + server_delete(&g_server); + release_authorization(); + +out: + daemon_conf_free(g_daemon_conf); + g_daemon_conf = NULL; + credentials_free(g_credentials); + g_credentials = NULL; + server_cli_options_free(&opts); + return exit_code; } -#endif /* !FASTSYNC_SERVER_AS_LIB */ +#endif diff --git a/src/server/server_cli.c b/src/server/server_cli.c new file mode 100644 index 0000000..794a97a --- /dev/null +++ b/src/server/server_cli.c @@ -0,0 +1,281 @@ +#include "server_cli.h" +#include "charset.h" +#include "credentials.h" +#include "utils.h" +#include +#include +#include +#include +#include +#include + +static void set_error(char* err, size_t err_size, const char* fmt, ...) { + if (!err || err_size == 0) + return; + va_list args; + va_start(args, fmt); + vsnprintf(err, err_size, fmt, args); + va_end(args); +} + +void server_cli_options_default(ServerCliOptions* opts) { + if (!opts) + return; + memset(opts, 0, sizeof(*opts)); + opts->destination_root = "."; + opts->port = 8080; + opts->bind_family = AF_UNSPEC; +} + +static bool arg_is(const char* arg, const char* name) { + return strcmp(arg, name) == 0; +} + +/* Match "--opt" against "--opt=value" / separate-value forms; on the "=" form + * *value receives the inline value. Returns true when the argument is the + * named option in either form. */ +static bool arg_has_value(const char* arg, const char* name, const char** value) { + if (strcmp(arg, name) == 0) + return true; /* separate form; caller takes the next argv slot */ + size_t name_len = strlen(name); + if (strncmp(arg, name, name_len) == 0 && arg[name_len] == '=') { + *value = arg + name_len + 1; + return true; + } + return false; +} + +static int parse_port_arg(const char* value, int* port, char* err, size_t err_size) { + char* end; + long p = strtol(value, &end, 10); + if (*end != '\0' || p <= 0 || p > 65535) { + char* escaped = output_escape(value, false); + set_error(err, err_size, "invalid port '%s' (must be 1-65535)", + escaped ? escaped : ""); + free(escaped); + return -1; + } + *port = (int)p; + return 0; +} + +int server_cli_parse(int argc, char* argv[], ServerCliOptions* opts, char* err, size_t err_size) { + if (err && err_size) + err[0] = '\0'; + server_cli_options_default(opts); + + for (int i = 1; i < argc; i++) { + const char* inline_value = NULL; + if (arg_is(argv[i], "--help")) { + opts->show_help = true; + return 1; + } else if (arg_is(argv[i], "--stdio")) { + opts->stdio_mode = true; + } else if (arg_is(argv[i], "--daemon")) { + opts->daemon_mode = true; + } else if (arg_is(argv[i], "--no-detach")) { + opts->no_detach = true; + } else if (arg_is(argv[i], "-v") || arg_is(argv[i], "--verbose")) { + opts->verbose = true; + } else if (arg_is(argv[i], "--tls")) { + opts->use_tls = true; + } else if (arg_is(argv[i], "--cert")) { + if (i + 1 >= argc) { + set_error(err, err_size, "missing argument for --cert"); + return -1; + } + opts->tls_cert = argv[++i]; + } else if (arg_is(argv[i], "--key")) { + if (i + 1 >= argc) { + set_error(err, err_size, "missing argument for --key"); + return -1; + } + opts->tls_key = argv[++i]; + } else if (arg_is(argv[i], "--ca")) { + if (i + 1 >= argc) { + set_error(err, err_size, "missing argument for --ca"); + return -1; + } + opts->tls_ca = argv[++i]; + } else if (arg_is(argv[i], "--client-cn")) { + if (i + 1 >= argc) { + set_error(err, err_size, "missing argument for --client-cn"); + return -1; + } + opts->client_cn = argv[++i]; + } else if (arg_is(argv[i], "--destination-root")) { + if (i + 1 >= argc) { + set_error(err, err_size, "missing argument for --destination-root"); + return -1; + } + opts->destination_root = argv[++i]; + opts->destination_root_set = true; + } else if (arg_has_value(argv[i], "--password-file", &inline_value)) { + if (!inline_value) { + if (i + 1 >= argc) { + set_error(err, err_size, "missing argument for --password-file"); + return -1; + } + inline_value = argv[++i]; + } + opts->password_file = inline_value; + } else if (arg_has_value(argv[i], "--early-input", &inline_value)) { + if (!inline_value) { + if (i + 1 >= argc) { + set_error(err, err_size, "missing argument for --early-input"); + return -1; + } + inline_value = argv[++i]; + } + opts->early_input_file = inline_value; + } else if (arg_has_value(argv[i], "--hash-credentials", &inline_value)) { + if (!inline_value) { + if (i + 1 >= argc) { + set_error(err, err_size, "missing argument for --hash-credentials"); + return -1; + } + inline_value = argv[++i]; + } + opts->hash_credentials_file = inline_value; + } else if (arg_has_value(argv[i], "--iterations", &inline_value)) { + if (!inline_value) { + if (i + 1 >= argc) { + set_error(err, err_size, "missing argument for --iterations"); + return -1; + } + inline_value = argv[++i]; + } + char* end = NULL; + long n = strtol(inline_value, &end, 10); + if (!end || *end != '\0' || n < (long)CREDENTIAL_MIN_ITERS || + n > (long)CREDENTIAL_MAX_ITERS) { + set_error(err, err_size, "--iterations must be in [%u,%u], got '%s'", CREDENTIAL_MIN_ITERS, + CREDENTIAL_MAX_ITERS, inline_value); + return -1; + } + opts->hash_iterations = (uint32_t)n; + opts->hash_iterations_set = true; + } else if (arg_is(argv[i], "--address")) { + if (i + 1 >= argc) { + set_error(err, err_size, "missing argument for --address"); + return -1; + } + opts->bind_address = argv[++i]; + } else if (arg_is(argv[i], "-4") || arg_is(argv[i], "--ipv4")) { + if (opts->bind_family == AF_INET6) { + set_error(err, err_size, "--ipv4 and --ipv6 are mutually exclusive"); + return -1; + } + opts->bind_family = AF_INET; + } else if (arg_is(argv[i], "-6") || arg_is(argv[i], "--ipv6")) { + if (opts->bind_family == AF_INET) { + set_error(err, err_size, "--ipv4 and --ipv6 are mutually exclusive"); + return -1; + } + opts->bind_family = AF_INET6; + } else if (arg_is(argv[i], "--allow-delete")) { + opts->allow_delete = true; + } else if (arg_is(argv[i], "--trust-sender")) { + opts->trust_sender = true; + } else if (arg_is(argv[i], "--no-super")) { + opts->no_super = true; + } else if (arg_is(argv[i], "--allow-unauthenticated")) { + opts->allow_unauthenticated = true; + } else if (arg_has_value(argv[i], "--iconv", &inline_value)) { + if (!inline_value) { + if (i + 1 >= argc) { + set_error(err, err_size, "missing argument for --iconv"); + return -1; + } + inline_value = argv[++i]; + } + opts->iconv_spec = inline_value; + } else if (arg_is(argv[i], "-p")) { + if (i + 1 >= argc) { + set_error(err, err_size, "missing argument for -p"); + return -1; + } + opts->port_set = true; + if (parse_port_arg(argv[++i], &opts->port, err, err_size) != 0) + return -1; + } else { + if (arg_has_value(argv[i], "--config", &inline_value)) { + if (!inline_value) { + if (i + 1 >= argc) { + set_error(err, err_size, "missing argument for --config"); + return -1; + } + inline_value = argv[++i]; + } + opts->config_path = inline_value; + } else if (arg_has_value(argv[i], "--dparam", &inline_value)) { + if (!inline_value) { + if (i + 1 >= argc) { + set_error(err, err_size, "missing argument for --dparam"); + return -1; + } + inline_value = argv[++i]; + } + const char** grown = + realloc((char**)opts->dparams, (size_t)(opts->dparam_count + 1) * sizeof(const char*)); + if (!grown) { + set_error(err, err_size, "out of memory parsing --dparam"); + return -1; + } + opts->dparams = grown; + opts->dparams[opts->dparam_count++] = inline_value; + } else if (argv[i][0] == '-') { + char* escaped = output_escape(argv[i], false); + set_error(err, err_size, "unknown option: %s", escaped ? escaped : ""); + free(escaped); + return -1; + } else { + set_error(err, err_size, "unexpected argument '%s'", argv[i]); + return -1; + } + } + } + + /* Cross-mode validation. */ + if (opts->stdio_mode && opts->daemon_mode) { + set_error(err, err_size, "--stdio and --daemon are mutually exclusive"); + return -1; + } + if (opts->daemon_mode && opts->destination_root_set) { + set_error(err, err_size, + "--destination-root cannot be combined with --daemon (module paths " + "replace it)"); + return -1; + } + if (!opts->daemon_mode && + (opts->config_path != NULL || opts->dparam_count > 0 || opts->no_detach || + opts->password_file != NULL || opts->early_input_file != NULL)) { + set_error(err, err_size, + "--config, --dparam, --no-detach, --password-file, and --early-input require " + "--daemon"); + return -1; + } + if (opts->hash_credentials_file != NULL && (opts->daemon_mode || opts->stdio_mode)) { + set_error(err, err_size, "--hash-credentials cannot be combined with --daemon or --stdio"); + return -1; + } + if (opts->hash_iterations_set && opts->hash_credentials_file == NULL) { + set_error(err, err_size, "--iterations requires --hash-credentials"); + return -1; + } + /* --iconv: reject a malformed CONVERT_SPEC or an unsupported charset name at + startup (a probe iconv_open is attempted). */ + if (opts->iconv_spec != NULL && !charset_spec_valid(opts->iconv_spec)) { + set_error(err, err_size, "--iconv requires LOCAL[,REMOTE] charset names supported by iconv"); + return -1; + } + return 0; +} + +void server_cli_options_free(ServerCliOptions* opts) { + if (!opts) + return; + free((char**)opts->dparams); + opts->dparams = NULL; + opts->dparam_count = 0; +} diff --git a/src/server/server_cli.h b/src/server/server_cli.h new file mode 100644 index 0000000..ab19a7f --- /dev/null +++ b/src/server/server_cli.h @@ -0,0 +1,68 @@ +#ifndef SERVER_CLI_H +#define SERVER_CLI_H + +#include +#include +#include + +/* Parsed fastsync-server command line. All string members are borrowed + * pointers into the original argv (valid for the life of the argv array the + * caller passed to server_cli_parse); dparams points at the raw --dparam + * argument strings. No member owns heap memory. */ +typedef struct ServerCliOptions { + bool stdio_mode; /* --stdio */ + bool daemon_mode; /* --daemon */ + bool no_detach; /* --no-detach */ + bool verbose; /* -v / --verbose */ + bool show_help; /* --help */ + bool use_tls; /* --tls */ + const char* tls_cert; /* --cert */ + const char* tls_key; /* --key */ + const char* tls_ca; /* --ca */ + const char* client_cn; /* --client-cn */ + bool destination_root_set; /* an explicit --destination-root was given */ + const char* destination_root; /* --destination-root value ("." if unset) */ + bool port_set; /* an explicit -p was given */ + int port; /* -p value (default 8080 when unset) */ + const char* config_path; /* --config value, or NULL */ + const char* password_file; /* --password-file value, or NULL (daemon) */ + const char* early_input_file; /* --early-input value, or NULL (daemon) */ + /* --hash-credentials=FILE: read `user:password` lines from FILE and print + * new-format credential-store lines to stdout, then exit. Standalone mode + * (mutually exclusive with --daemon/--stdio). */ + const char* hash_credentials_file; + bool hash_iterations_set; /* an explicit --iterations was given */ + uint32_t hash_iterations; /* --iterations value (default CREDENTIAL_DEFAULT_ITERS) */ + const char** dparams; /* raw --dparam override strings */ + int dparam_count; + const char* bind_address; /* --address */ + int bind_family; /* AF_UNSPEC / AF_INET / AF_INET6 */ + bool allow_delete; /* --allow-delete */ + bool trust_sender; /* --trust-sender */ + bool allow_unauthenticated; /* --allow-unauthenticated */ + /* --no-super: operator veto forcing SUPER_MODE_OFF for every connection, so + * the receiver never attempts super-user activities (ownership application, + * device-node creation) even when running as root. Applies to --stdio and + * --daemon alike; also makes the server refuse any client --copy-as. */ + bool no_super; /* --no-super */ + /* --iconv=CONVERT_SPEC: the server's own LOCAL charset declaration. The + * client's full spec rides the wire config frame anyway; when the server is + * started with its own --iconv, its LOCAL half overrides the local charset + * the client assumed so the server converts received names to ITS charset. + * Borrowed pointer into argv (never owns heap). */ + const char* iconv_spec; /* --iconv value, or NULL */ +} ServerCliOptions; + +/* Parse argc/argv into *opts. Zero-initialize *opts before calling (or use + * server_cli_options_default). Returns: + * 1 -- --help was requested (opts->show_help set; caller prints usage). + * 0 -- parsed successfully. + * -1 -- invalid arguments (err is filled with the reason). + */ +void server_cli_options_default(ServerCliOptions* opts); +int server_cli_parse(int argc, char* argv[], ServerCliOptions* opts, char* err, size_t err_size); +/* Release the only heap the parsed options own (the dparams pointer array; the + * strings it points at are borrowed from argv and are not freed). Safe to + * call on a zero-initialized/defaulted struct. */ +void server_cli_options_free(ServerCliOptions* opts); +#endif diff --git a/src/shared/array_list.c b/src/shared/array_list.c index 4df49d0..95799b0 100644 --- a/src/shared/array_list.c +++ b/src/shared/array_list.c @@ -1,16 +1,18 @@ +#include "log.h" #include "array_list.h" +#include "protocol.h" #include #include #include ArrayList* array_list_create(void (*item_destroyer)(void* item)) { - ArrayList* list = (ArrayList*)malloc(sizeof(ArrayList)); + ArrayList* list = (ArrayList*)protocol_alloc(sizeof(ArrayList)); if (list == NULL) { - perror("ERROR: Could not allocate memory for array list struct"); + log_perror("ERROR: Could not allocate memory for array list struct"); return NULL; } - list->items = malloc(INITIAL_ARRAY_SIZE * sizeof(void*)); + list->items = protocol_alloc(INITIAL_ARRAY_SIZE * sizeof(void*)); if (list->items == NULL) { free(list); return NULL; @@ -34,15 +36,15 @@ void array_list_delete(ArrayList* array_list) { free(array_list); } -bool array_list_extend(ArrayList* array_list) { +static bool array_list_extend(ArrayList* array_list) { if (array_list == NULL) return false; int new_capacity = array_list->capacity * 2; if (new_capacity == 0) new_capacity = INITIAL_ARRAY_SIZE; - void* new_items = realloc(array_list->items, new_capacity * sizeof(void*)); + void* new_items = protocol_realloc(array_list->items, new_capacity * sizeof(void*)); if (new_items == NULL) { - perror("ERROR: Could not reallocate memory for array list items"); + log_perror("ERROR: Could not reallocate memory for array list items"); return false; } array_list->items = new_items; @@ -66,9 +68,9 @@ void** array_list_to_array(const ArrayList* array_list) { if (array_list == NULL) { return NULL; } - void** array = malloc(array_list->size * sizeof(void*)); + void** array = protocol_alloc(array_list->size * sizeof(void*)); if (array == NULL) { - perror("Could not malloc space for array from array list!"); + log_perror("Could not malloc space for array from array list!"); return NULL; } memcpy(array, array_list->items, array_list->size * sizeof(void*)); diff --git a/src/shared/array_list.h b/src/shared/array_list.h index 9ecbe0b..69485bc 100644 --- a/src/shared/array_list.h +++ b/src/shared/array_list.h @@ -14,7 +14,6 @@ typedef struct ArrayList { ArrayList* array_list_create(void (*item_destroyer)(void* item)); void array_list_delete(ArrayList* array_list); -bool array_list_extend(ArrayList* array_list); bool array_list_add(ArrayList* array_list, void* item); void** array_list_to_array(const ArrayList* array_list); diff --git a/src/shared/batch.c b/src/shared/batch.c new file mode 100644 index 0000000..4a2ee38 --- /dev/null +++ b/src/shared/batch.c @@ -0,0 +1,160 @@ +#include "batch.h" +#include "data.h" +#include "file.h" +#include "file_receive.h" +#include "log.h" +#include +#include +#include +#include + +/* Serialization metadata mode for the batch stream, captured from the config at + * batch_write_header time. The header persists it into the file so a batch is + * self-describing: batch_read_apply re-reads it from the file (not from the + * reading config), so a batch written with -M is applied identically by an + * invoking process regardless of its own -M setting. The batch driver is a + * single sequential scan pass within one thread, so this module-level flag is + * safe. */ +static bool batch_metadata_mode = false; + +static bool write_all_bytes(int fd, const void* data, size_t size) { + const unsigned char* p = (const unsigned char*)data; + size_t done = 0; + while (done < size) { + ssize_t n = write(fd, p + done, size - done); + if (n < 0 && errno == EINTR) + continue; + if (n <= 0) + return false; + done += (size_t)n; + } + return true; +} + +bool batch_write_header(int fd, const Config* config) { + if (fd < 0) + return false; + batch_metadata_mode = config != NULL && config->use_metadata; + if (!write_all_bytes(fd, BATCH_MAGIC, BATCH_MAGIC_LEN)) + return false; + unsigned char version = BATCH_FORMAT_VERSION; + if (!write_all_bytes(fd, &version, 1)) + return false; + unsigned char mode = batch_metadata_mode ? 1 : 0; + return write_all_bytes(fd, &mode, 1); +} + +bool batch_write_chunk(int fd, Chunk* chunk) { + if (fd < 0 || chunk == NULL) + return false; + Data* serialized = chunk_serialize(chunk, batch_metadata_mode); + if (serialized == NULL) + return false; + bool ok = false; + unsigned long long length = (unsigned long long)serialized->size; + if (length > BATCH_MAX_RECORD) { + log_message(LOG_LEVEL_ERROR, "batch: record size %llu exceeds the %llu-byte cap", length, + (unsigned long long)BATCH_MAX_RECORD); + } else if (write_all_bytes(fd, &length, sizeof(length)) && + (length == 0 || write_all_bytes(fd, serialized->data, (size_t)length))) { + ok = true; + } + data_destroy(serialized); + return ok; +} + +/* Read exactly `size` bytes. Returns true on success. On reaching EOF, sets + * *clean_eof only when no bytes had been read yet (a clean boundary) and returns + * that value, so a truncated record (EOF mid-read) yields false. */ +static bool read_exact(int fd, void* data, size_t size, bool* clean_eof) { + unsigned char* p = (unsigned char*)data; + size_t done = 0; + while (done < size) { + ssize_t n = read(fd, p + done, size - done); + if (n < 0 && errno == EINTR) + continue; + if (n == 0) { + if (clean_eof) + *clean_eof = done == 0; + return done == 0; + } + if (n < 0) + return false; + done += (size_t)n; + } + if (clean_eof) + *clean_eof = false; + return true; +} + +int batch_read_apply(int fd, const Config* config, const char* dest_root) { + if (fd < 0 || dest_root == NULL || dest_root[0] == '\0') + return -1; + + char magic[BATCH_MAGIC_LEN]; + bool eof = false; + if (!read_exact(fd, magic, BATCH_MAGIC_LEN, &eof) || eof || + memcmp(magic, BATCH_MAGIC, BATCH_MAGIC_LEN) != 0) { + log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad magic)"); + return -1; + } + unsigned char version; + if (!read_exact(fd, &version, 1, &eof) || eof || version != BATCH_FORMAT_VERSION) { + log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad or missing format version)"); + return -1; + } + unsigned char mode; + if (!read_exact(fd, &mode, 1, &eof) || eof || (mode != 0 && mode != 1)) { + log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad metadata flag)"); + return -1; + } + bool use_metadata = mode == 1; + + while (1) { + unsigned long long length; + if (!read_exact(fd, &length, sizeof(length), &eof)) { + log_message(LOG_LEVEL_ERROR, "batch: truncated length prefix"); + return -1; + } + if (eof) + break; /* clean end of stream */ + if (length == 0 || length > BATCH_MAX_RECORD) { + log_message(LOG_LEVEL_ERROR, "batch: rejected record length %llu (valid range 1..%llu)", + length, (unsigned long long)BATCH_MAX_RECORD); + return -1; + } + char* record = (char*)malloc((size_t)length); + if (record == NULL) { + log_message(LOG_LEVEL_ERROR, "batch: could not allocate a %llu-byte record", length); + return -1; + } + if (!read_exact(fd, record, (size_t)length, &eof) || eof) { + log_message(LOG_LEVEL_ERROR, "batch: truncated chunk record"); + free(record); + return -1; + } + Data* data = data_create(record, (size_t)length); + if (data == NULL) + return -1; /* data_create frees `record` on failure */ + Chunk* chunk = chunk_deserialize(data, use_metadata); + data_destroy(data); + if (chunk == NULL) { + log_message(LOG_LEVEL_ERROR, "batch: rejected malformed chunk record"); + return -1; + } + for (int i = 0; i < chunk->element_count; i++) { + File* file = chunk->items[i]; + chunk->items[i] = NULL; + if (file == NULL) + continue; + FileSaveResult result = file_save_to_disk_full(dest_root, file, config); + file_destroy(file); + if (result == FILE_SAVE_ERROR) { + chunk_destroy(chunk); + return -1; + } + } + chunk_destroy(chunk); + } + return 0; +} \ No newline at end of file diff --git a/src/shared/batch.h b/src/shared/batch.h new file mode 100644 index 0000000..41e8579 --- /dev/null +++ b/src/shared/batch.h @@ -0,0 +1,28 @@ +#ifndef BATCH_H +#define BATCH_H +#include "chunk.h" +#include "config.h" + +/* Phase 6 residual-batch codec. A residual batch is a self-contained + * single-file record of a whole source tree: a magic+format-version header + * followed by length-prefixed chunk blobs (each built with chunk_serialize), + * byte-identical by construction. The batch is a client-only driver feature: + * it never crosses the wire, so there is no PROTOCOL_VERSION bump and no server + * change. */ + +#define BATCH_MAGIC "FSTRESBATCH" +#define BATCH_MAGIC_LEN 11 +#define BATCH_FORMAT_VERSION 1 +/* Max size of a single length-prefixed record (a whole serialized chunk, + * which can span several files). A single source file near the 64 MB wire + * limit plus per-file headers can produce a record slightly over 64 MB, so a + * large file just under the wire cap may be refused by the batch writer; this + * is documented upstream and the failure is clean (the partial batch is + * unlinked), never a truncated/corrupt batch. */ +#define BATCH_MAX_RECORD (64ULL * 1024 * 1024) + +bool batch_write_header(int fd, const Config* config); +bool batch_write_chunk(int fd, Chunk* chunk); +int batch_read_apply(int fd, const Config* config, const char* dest_root); + +#endif \ No newline at end of file diff --git a/src/shared/charset.c b/src/shared/charset.c new file mode 100644 index 0000000..75a2ac5 --- /dev/null +++ b/src/shared/charset.c @@ -0,0 +1,384 @@ +#include "charset.h" +#include "log.h" +#include "protocol.h" +#include "utils.h" +#include +#include +#include +#include + +typedef struct { + iconv_t cd; +} CharsetConversion; + +/* Process-wide wire conversion descriptor (one direction per process: a client + * only sends, a server only receives). CONCURRENCY CONTRACT: iconv_t is not + * guaranteed thread-safe, so every conversion MUST run on a single thread at a + * time. This holds today -- on the client the conversions run on the sender + * thread (in the -m pipeline chunk_serialize/send happen on the sender thread + * only), on the server on the receive-loop thread; the descriptor is + * initialized on one thread before any transfer thread spawns and torn down + * (charset_wire_free) only after all threads have joined. Do not add a + * concurrent conversion path (e.g. parallel chunk serialization) without + * guarding access with a mutex. */ +static CharsetConversion* g_wire_conv; + +/* Grow *buf to double capacity, freeing it on failure. realloc preserves the + * already-written prefix, so the caller only tracks its write offset. */ +static bool grow_charset_buffer(char** buf, size_t* cap) { + size_t new_cap = *cap * 2; + if (new_cap <= *cap) { + free(*buf); + *buf = NULL; + return false; + } + char* grown = realloc(*buf, new_cap); + if (!grown) { + free(*buf); + *buf = NULL; + return false; + } + *buf = grown; + *cap = new_cap; + return true; +} + +/* Throw away any pending shift state so a subsequent conversion starts clean. + * The flush output is discarded; for the stateless single-byte/UTF charsets + * this feature targets it is a no-op. */ +static void charset_conversion_reset(const CharsetConversion* conv) { + char scratch[64]; + char* sp = scratch; + size_t sl = sizeof(scratch); + (void)iconv(conv->cd, NULL, NULL, &sp, &sl); +} + +int charset_spec_parse(const char* spec, char** local_out, char** remote_out) { + if (!local_out || !remote_out) + return -1; + *local_out = NULL; + *remote_out = NULL; + if (!spec || spec[0] == '\0') + return -1; + char* dup = str_dup(spec); + if (!dup) + return -1; + char* comma = strchr(dup, ','); + if (comma) { + if (comma == dup || comma[1] == '\0') { + free(dup); + return -1; + } + *comma = '\0'; + *local_out = str_dup(dup); + *remote_out = str_dup(comma + 1); + free(dup); + } else { + *local_out = str_dup(dup); + *remote_out = str_dup(dup); + free(dup); + } + if (!*local_out || !*remote_out) { + free(*local_out); + free(*remote_out); + *local_out = NULL; + *remote_out = NULL; + return -1; + } + return 0; +} + +void* charset_conversion_open(const char* from_charset, const char* to_charset) { + if (!from_charset || !to_charset) + return NULL; + iconv_t cd = iconv_open(to_charset, from_charset); + if (cd == (iconv_t)-1) + return NULL; + CharsetConversion* conv = malloc(sizeof(CharsetConversion)); + if (!conv) { + iconv_close(cd); + return NULL; + } + conv->cd = cd; + return conv; +} + +void charset_conversion_close(void* conversion) { + if (!conversion) + return; + CharsetConversion* conv = (CharsetConversion*)conversion; + iconv_close(conv->cd); + free(conv); +} + +/* Probe a single conversion direction: the from/to charsets both open AND a + * representative ASCII name converts to a byte string containing no embedded + * NUL (so a target charset like UTF-16 that emits NUL bytes for ordinary ASCII + * names is rejected up front -- such an output would be silently truncated by + * the C-string wire helpers). */ +static bool direction_probe_valid(const char* from, const char* to) { + if (!from || !to) + return false; + void* conv = charset_conversion_open(from, to); + if (!conv) + return false; + bool ok = true; + char input = 'a'; + char* in_ptr = &input; + size_t in_left = 1; + char out_buf[64]; + char* out_ptr = out_buf; + size_t out_left = sizeof(out_buf); + if (iconv(((CharsetConversion*)conv)->cd, &in_ptr, &in_left, &out_ptr, &out_left) == (size_t)-1) + ok = false; + char flush_buf[64]; + char* flush_ptr = flush_buf; + size_t flush_left = sizeof(flush_buf); + if (ok && + iconv(((CharsetConversion*)conv)->cd, NULL, NULL, &flush_ptr, &flush_left) == (size_t)-1) + ok = false; + size_t produced = (size_t)(out_ptr - out_buf); + if (ok && produced > 0 && memchr(out_buf, '\0', produced) != NULL) + ok = false; + charset_conversion_close(conv); + return ok; +} + +bool charset_pair_valid(const char* local, const char* remote) { + /* Both ends convert in opposite directions with the same two charsets, so a + * valid spec must open (and be NUL-free) in BOTH directions: the sender + * opens local->remote, the receiver opens remote->local. */ + return direction_probe_valid(local, remote) && direction_probe_valid(remote, local); +} + +bool charset_spec_valid(const char* spec) { + if (!spec) + return true; + char* local; + char* remote; + if (charset_spec_parse(spec, &local, &remote) != 0) + return false; + bool ok = charset_pair_valid(local, remote); + free(local); + free(remote); + return ok; +} + +bool charset_spec_valid_direction(const char* from_charset, const char* to_charset) { + return direction_probe_valid(from_charset, to_charset); +} + +/* The receiver's real conversion is wire(client REMOTE) -> server-local (the + * server's own --iconv LOCAL half, or the client's LOCAL half when the server + * has no --iconv). A dedicated pre-ack check so an impossible direction is + * rejected before the connection instead of refusing mid-transfer. */ +bool charset_wire_receiver_spec_valid(const char* spec, const char* server_spec) { + if (!spec) + return true; + char* local; + char* remote; + if (charset_spec_parse(spec, &local, &remote) != 0) + return false; + const char* wire = remote; + const char* target_local = local; + char* server_local = NULL; + char* server_remote = NULL; + if (server_spec) { + if (charset_spec_parse(server_spec, &server_local, &server_remote) != 0) { + free(local); + free(remote); + return false; + } + target_local = server_local; + } + bool ok = charset_spec_valid_direction(wire, target_local); + free(server_local); + free(server_remote); + free(local); + free(remote); + return ok; +} + +char* charset_convert(const void* conversion, const char* in, int* err_out) { + if (!conversion || !in) + return NULL; + const CharsetConversion* conv = (const CharsetConversion*)conversion; + size_t in_len = strlen(in); + size_t cap = in_len + 16; + char* out = malloc(cap); + if (!out) + return NULL; + size_t in_left = in_len; + char* in_ptr = (char*)in; + size_t out_used = 0; + + while (in_left > 0) { + char* out_ptr = out + out_used; + size_t out_left = cap - out_used; + if (iconv(conv->cd, &in_ptr, &in_left, &out_ptr, &out_left) == (size_t)-1) { + if (errno != E2BIG) { + if (err_out) + *err_out = errno; + charset_conversion_reset(conv); + free(out); + return NULL; + } + /* Output exhausted but input remains. E2BIG does not roll the output + pointer back: the bytes iconv already emitted before the failure must + be preserved, so advance out_used before growing. */ + out_used = (size_t)(out_ptr - out); + if (!grow_charset_buffer(&out, &cap)) + return NULL; + continue; + } + out_used = (size_t)(out_ptr - out); + } + + /* Flush any pending shift state (a no-op for the stateless single-byte and + UTF charsets this feature targets, but keeps the descriptor clean). */ + for (;;) { + char* out_ptr = out + out_used; + size_t out_left = cap - out_used; + if (iconv(conv->cd, NULL, NULL, &out_ptr, &out_left) == (size_t)-1) { + if (errno != E2BIG) { + if (err_out) + *err_out = errno; + charset_conversion_reset(conv); + free(out); + return NULL; + } + out_used = (size_t)(out_ptr - out); + if (!grow_charset_buffer(&out, &cap)) + return NULL; + continue; + } + out_used = (size_t)(out_ptr - out); + break; + } + + /* A successful iconv call may legitimately consume the whole buffer (output + exactly fills cap), leaving no room for the terminator: guarantee headroom + before the final write. */ + if (out_used >= cap && !grow_charset_buffer(&out, &cap)) + return NULL; + + /* Defense in depth: a target charset that emits embedded NUL bytes would + truncate at the first NUL in the C-string wire helpers; fail cleanly + (validation already rejects such charsets up front). */ + if (memchr(out, '\0', out_used) != NULL) { + if (err_out) + *err_out = EILSEQ; + charset_conversion_reset(conv); + free(out); + return NULL; + } + + out[out_used] = '\0'; + return out; +} + +bool charset_wire_init_sender(const char* spec) { + charset_wire_free(); + if (!spec) + return true; + char* local; + char* remote; + if (charset_spec_parse(spec, &local, &remote) != 0) + return false; + void* conv = charset_conversion_open(local, remote); + free(local); + free(remote); + if (!conv) + return false; + g_wire_conv = (CharsetConversion*)conv; + return true; +} + +bool charset_wire_init_receiver(const char* spec, const char* server_spec) { + charset_wire_free(); + if (!spec) + return true; + char* local; + char* remote; + if (charset_spec_parse(spec, &local, &remote) != 0) + return false; + /* The wire charset is the client spec's REMOTE half; the local charset is + * the client spec's LOCAL half unless the server was itself started with + * --iconv naming a different local charset (the server halves above never + * travel, so the server's own flag is the only way its local charset can + * differ from what the client assumed). */ + const char* wire = remote; + const char* target_local = local; + char* server_local = NULL; + char* server_remote = NULL; + if (server_spec) { + if (charset_spec_parse(server_spec, &server_local, &server_remote) != 0) { + free(local); + free(remote); + return false; + } + target_local = server_local; + } + void* conv = charset_conversion_open(wire, target_local); + free(server_local); + free(server_remote); + free(local); + free(remote); + if (!conv) + return false; + g_wire_conv = (CharsetConversion*)conv; + return true; +} + +void charset_wire_free(void) { + if (g_wire_conv) { + charset_conversion_close(g_wire_conv); + g_wire_conv = NULL; + } +} + +bool charset_wire_active(void) { + return g_wire_conv != NULL; +} + +char* charset_wire_apply(const char* path) { + if (!g_wire_conv) + return str_dup(path); + return charset_convert(g_wire_conv, path, NULL); +} + +static void charset_convert_failure_log(const char* path) { + char* escaped = output_escape(path, false); + log_message(LOG_LEVEL_ERROR, "--iconv: cannot convert file name '%s' to the target charset", + escaped ? escaped : ""); + free(escaped); +} + +bool send_wire_str(int file_descriptor, const char* local_path) { + if (!g_wire_conv) + return send_str(file_descriptor, local_path); + char* wire = charset_wire_apply(local_path); + if (!wire) { + charset_convert_failure_log(local_path); + return false; + } + bool ok = send_str(file_descriptor, wire); + free(wire); + return ok; +} + +char* receive_wire_str(int file_descriptor) { + char* raw = receive_str(file_descriptor); + if (!raw) + return NULL; + if (!g_wire_conv) + return raw; + char* local = charset_convert(g_wire_conv, raw, NULL); + if (!local) { + charset_convert_failure_log(raw); + free(raw); + return NULL; + } + free(raw); + return local; +} \ No newline at end of file diff --git a/src/shared/charset.h b/src/shared/charset.h new file mode 100644 index 0000000..49cd0f3 --- /dev/null +++ b/src/shared/charset.h @@ -0,0 +1,85 @@ +#ifndef CHARSET_H +#define CHARSET_H + +#include +#include + +/* --iconv=CONVERT_SPEC file-name charset conversion (rsync compatibility). + * + * CONVERT_SPEC is "LOCAL[,REMOTE]": LOCAL is the charset of our own file + * names, REMOTE is the charset of the remote side's file names and defaults + * to LOCAL when the comma half is omitted. The sender converts every local + * path from LOCAL to REMOTE before it goes on the wire; the receiver converts + * every received path back from REMOTE to LOCAL. A NULL/disabled spec means + * identity with zero overhead (the common path never consults iconv). + * + * All helpers are friendly to the strict cold path: the wire conversion state + * is process-global (one direction per process -- a client only sends, a + * server only receives) and is initialized once, before any path is + * serialized, so conversion compiles to a single non-NULL check when disabled. + */ + +/* Parse CONVERT_SPEC into malloc'd LOCAL and REMOTE charset names (caller + * frees both). REMOTE is a separate copy of LOCAL when no comma is present. + * Returns 0 on success, -1 on a malformed spec (empty halves / missing value / + * allocation failure); nothing is allocated on the -1 path. Both output + * pointers are REQUIRED (non-NULL). */ +int charset_spec_parse(const char* spec, char** local_out, char** remote_out); + +/* True when a CONVERT_SPEC is well-formed AND its charsets are usable for this + * feature: each pair opens in a probe iconv_open in BOTH directions (a sender + * converts local->remote, the receiver converts remote->local) and converting + * a representative ASCII name emits no embedded NUL byte (a UTF-16-style NUL + * emitter would be silently truncated by the C-string wire helpers). A typo'd + * charset name is therefore rejected at startup, not mid-run. NULL (iconv + * disabled) is always valid. */ +bool charset_spec_valid(const char* spec); + +/* Probe a concrete from->to conversion pair without keeping the descriptor: + * both charsets open AND a representative ASCII name converts with no embedded + * NUL. Used for direction-specific validation (e.g. the receiver's exact + * wire->local direction including a server-side charset override). */ +bool charset_spec_valid_direction(const char* from_charset, const char* to_charset); +bool charset_pair_valid(const char* local, const char* remote); + +/* One-shot conversion of a NUL-terminated input to a malloc'd NUL-terminated + * result, or NULL on failure. On failure *err_out (when non-NULL) receives + * the iconv errno (EILSEQ/EINVAL = the input is not representable in the + * target charset). The caller must free the result. */ +char* charset_convert(const void* conversion, const char* in, int* err_out); + +/* Open a conversion descriptor for direction from_charset -> to_charset. + * Returns NULL (errno = EINVAL) when a charset name is unsupported. Freed + * with charset_conversion_close. */ +void* charset_conversion_open(const char* from_charset, const char* to_charset); +void charset_conversion_close(void* conversion); + +/* Process-wide wire conversion. charset_wire_init_sender (client side) opens + * LOCAL->REMOTE; charset_wire_init_receiver (server side) opens + * wire(REMOTE)->server-local. server_spec is the server's own --iconv, whose + * LOCAL half may override the local charset the client assumed; NULL reuses + * the client spec's LOCAL half. Both return false on an unsupported spec. + * The state is freed with charset_wire_free. */ +bool charset_wire_init_sender(const char* spec); +bool charset_wire_init_receiver(const char* spec, const char* server_spec); +void charset_wire_free(void); +bool charset_wire_active(void); + +/* Pre-ack receiver-direction sanity (see charset_wire_init_receiver): true + * when the exact wire->server-local conversion the receiver will use (client + * spec's REMOTE half into the server's own LOCAL half, or the client's LOCAL + * half when the server has no --iconv) opens and produces NUL-free output. */ +bool charset_wire_receiver_spec_valid(const char* spec, const char* server_spec); + +/* Convert a path across the wire in the process direction. Returns a malloc'd + * string, or NULL when the name cannot be represented in the target charset. */ +char* charset_wire_apply(const char* path); + +/* Convenience wire string I/O: encode+send_str / receive_str+decode. Both + * return false/NULL (logging a clear --iconv error) on conversion failure, so + * an unconvertible path FAILS the transfer cleanly instead of silently sending + * a mangled name. */ +bool send_wire_str(int file_descriptor, const char* local_path); +char* receive_wire_str(int file_descriptor); + +#endif \ No newline at end of file diff --git a/src/shared/checksum.c b/src/shared/checksum.c new file mode 100644 index 0000000..f17698f --- /dev/null +++ b/src/shared/checksum.c @@ -0,0 +1,75 @@ +#include "checksum.h" +#include +#include +#include + +/* delta.c owns the single XXH_IMPLEMENTATION that provides the xxHash symbols + * for the whole binary; this TU only needs the declarations. */ +#include + +bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t size, uint8_t* out, + size_t out_capacity, size_t* out_len) { + if (!out || !out_len || out_capacity < CHECKSUM_MAX_DIGEST_LEN) + return false; + if (data == NULL && size != 0) + return false; + + if (algo == CHECKSUM_ALGO_XXH64) { + uint64_t digest = XXH64(data, size, seed); + memcpy(out, &digest, sizeof(digest)); + *out_len = sizeof(digest); + return true; + } + + if (algo == CHECKSUM_ALGO_MD5) { + /* md5 takes no seed; the caller's seed is deliberately ignored (documented + * in RSYNC_COMPAT.md). OpenSSL's one-shot EVP_Digest needs a non-NULL + * buffer even for an empty input, so map a NULL data + size==0 to an empty + * buffer. */ + static const uint8_t empty = 0; + const void* input = data ? data : ∅ + unsigned int digest_len = 0; + if (EVP_Digest(input, size, out, &digest_len, EVP_md5(), NULL) != 1) + return false; + if (digest_len > out_capacity) + return false; + *out_len = digest_len; + return true; + } + + return false; +} + +int checksum_algo_from_name(const char* name) { + if (!name) + return -1; + if (strcasecmp(name, "xxh64") == 0 || strcasecmp(name, "xxhash") == 0) + return (int)CHECKSUM_ALGO_XXH64; + if (strcasecmp(name, "md5") == 0) + return (int)CHECKSUM_ALGO_MD5; + return -1; +} + +const char* checksum_algo_name(ChecksumAlgo algo) { + switch (algo) { + case CHECKSUM_ALGO_XXH64: + return "xxh64"; + case CHECKSUM_ALGO_MD5: + return "md5"; + } + return ""; +} + +bool checksum_algo_valid(int algo) { + return algo == (int)CHECKSUM_ALGO_XXH64 || algo == (int)CHECKSUM_ALGO_MD5; +} + +uint8_t checksum_digest_len(ChecksumAlgo algo) { + switch (algo) { + case CHECKSUM_ALGO_XXH64: + return 8; + case CHECKSUM_ALGO_MD5: + return 16; + } + return 0; +} \ No newline at end of file diff --git a/src/shared/checksum.h b/src/shared/checksum.h new file mode 100644 index 0000000..ddc6ee5 --- /dev/null +++ b/src/shared/checksum.h @@ -0,0 +1,45 @@ +#ifndef CHECKSUM_H +#define CHECKSUM_H + +#include +#include +#include + +/* Whole-file content-digest algorithms selectable with --checksum-choice and + * seeded with --checksum-seed. The ids are the values actually placed on the + * wire (config frame), so they must be kept stable and validated on receive. + * CHECKSUM_ALGO_XXH64 == 0 is the default and is byte-for-byte what FastSync + * computed before these options existed (xxHash64 with seed 0). */ +typedef enum { CHECKSUM_ALGO_XXH64 = 0, CHECKSUM_ALGO_MD5 = 1 } ChecksumAlgo; + +/* md5 digest is 16 bytes, the longest supported. */ +#define CHECKSUM_MAX_DIGEST_LEN 16 + +/* Compute the whole-file digest of the first `size` bytes of `data`. + * + * - CHECKSUM_ALGO_XXH64: xxHash64(data, size, seed) (full 64-bit seed). + * - CHECKSUM_ALGO_MD5: md5(data, size) via OpenSSL EVP. + * md5 has no seed, so `seed` is ignored (documented). + * - `size == 0` hashes the empty input (plus its seed), not a NULL input. + * + * Writes up to `out_capacity` bytes into `out`, storing the digest length in + * *out_len. Returns false on NULL out* or when the digest would not fit. + * Never writes more than CHECKSUM_MAX_DIGEST_LEN bytes. */ +bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t size, uint8_t* out, + size_t out_capacity, size_t* out_len); + +/* Resolve a --checksum-choice string (case-insensitive) to an algorithm id. + * Accepts "xxh64" and "xxhash" (both map to CHECKSUM_ALGO_XXH64, rsync's + * xxhash spelling) and "md5". Returns -1 for any unsupported name. */ +int checksum_algo_from_name(const char* name); + +/* Canonical name of an algorithm (used in CLI error messages). */ +const char* checksum_algo_name(ChecksumAlgo algo); + +/* True when `algo` is a supported id (used by config receive validation). */ +bool checksum_algo_valid(int algo); + +/* Digest length in bytes for an algorithm (xxx64 = 8, md5 = 16). */ +uint8_t checksum_digest_len(ChecksumAlgo algo); + +#endif /* CHECKSUM_H */ \ No newline at end of file diff --git a/src/shared/chmod.c b/src/shared/chmod.c new file mode 100644 index 0000000..bfea77b --- /dev/null +++ b/src/shared/chmod.c @@ -0,0 +1,90 @@ +#include "chmod.h" +#include +#include + +static bool parse_clause(mode_t* mode, const char* begin, const char* end) { + const char* p = begin; + unsigned who = 0; + while (p < end && strchr("ugoa", *p)) { + if (*p == 'a') + who = 7; + else + who |= *p == 'u' ? 1U : (*p == 'g' ? 2U : 4U); + p++; + } + if (who == 0) + who = 7; + if (p == end || (*p != '+' && *p != '-' && *p != '=')) + return false; + char operation = *p++; + mode_t bits = 0; + while (p < end) { + mode_t bit; + switch (*p++) { + case 'r': + bit = 4; + break; + case 'w': + bit = 2; + break; + case 'x': + bit = 1; + break; + default: + return false; + } + bits |= bit; + } + for (unsigned class_index = 0; class_index < 3; class_index++) { + unsigned class_bit = 1U << class_index; + if (!(who & class_bit)) + continue; + mode_t shift = (mode_t)((2U - class_index) * 3U); + mode_t mask = (mode_t)(7U << shift); + mode_t class_bits = (mode_t)(bits << shift); + if (operation == '+') + *mode |= class_bits; + else if (operation == '-') + *mode &= ~class_bits; + else + *mode = (*mode & ~mask) | class_bits; + } + return true; +} + +bool chmod_apply(mode_t mode, const char* spec, mode_t* result) { + if (!spec || !*spec || !result) + return false; + bool numeric = true; + size_t length = strlen(spec); + if (length > 4) + numeric = false; + for (size_t i = 0; i < length && numeric; i++) + numeric = spec[i] >= '0' && spec[i] <= '7'; + if (numeric) { + if (length == 0 || length > 4) + return false; + mode_t parsed = 0; + for (size_t i = 0; i < length; i++) + parsed = (mode_t)((parsed << 3) | (spec[i] - '0')); + *result = parsed; + return true; + } + + mode_t changed = mode; + const char* begin = spec; + while (*begin) { + const char* end = strchr(begin, ','); + if (!end) + end = begin + strlen(begin); + if (!parse_clause(&changed, begin, end)) + return false; + if (*end == '\0') + break; + begin = end + 1; + if (!*begin) + return false; + } + *result = changed; + return true; +} diff --git a/src/shared/chmod.h b/src/shared/chmod.h new file mode 100644 index 0000000..c3b259c --- /dev/null +++ b/src/shared/chmod.h @@ -0,0 +1,10 @@ +#ifndef CHMOD_H +#define CHMOD_H + +#include +#include + +/* Apply the supported rsync --chmod syntax to a permission mode. */ +bool chmod_apply(mode_t mode, const char* spec, mode_t* result); + +#endif diff --git a/src/shared/chunk.c b/src/shared/chunk.c index 6c5adbd..c5baeb4 100644 --- a/src/shared/chunk.c +++ b/src/shared/chunk.c @@ -1,9 +1,12 @@ #include +#include +#include #include #include #include #include "array_list.h" +#include "charset.h" #include "chunk.h" #include "compression.h" #include "data.h" @@ -11,18 +14,33 @@ #include "log.h" #include "metadata.h" #include "protocol.h" +#include "utils.h" + +/* Maximum individual file data size within a chunk (64 MB) */ +#define MAX_FILE_DATA_SIZE (64ULL * 1024 * 1024) +#define MAX_FILES_PER_CHUNK 65536U Chunk* chunk_create(File** items, int element_count) { - Chunk* chunk = (Chunk*)malloc(sizeof(Chunk)); + if (element_count < 0 || (element_count > 0 && items == NULL)) + return NULL; + Chunk* chunk = (Chunk*)protocol_alloc(sizeof(Chunk)); if (chunk == NULL) { - perror("ERROR: Could not allocate memory for chunk structure"); + log_perror("ERROR: Could not allocate memory for chunk structure"); return NULL; } - chunk->items = (File**)malloc(element_count * sizeof(File*)); - if (chunk->items == NULL) { - free(chunk); - return NULL; + if (element_count == 0) { + chunk->items = NULL; + } else { + if ((size_t)element_count > SIZE_MAX / sizeof(File*)) { + free(chunk); + return NULL; + } + chunk->items = (File**)protocol_alloc((size_t)element_count * sizeof(File*)); + if (chunk->items == NULL) { + free(chunk); + return NULL; + } } for (int i = 0; i < element_count; i++) { @@ -46,16 +64,79 @@ void chunk_destroy(void* item) { free(chunk); } +/* --iconv: a chunk blob carries wire-charset path/target bytes. Encode the + * sender-side path (a no-op copy when iconv is disabled) so the blob is in the + * same charset as every other wire string. */ +static char* chunk_encode_wire(const char* path) { + if (!charset_wire_active()) + return str_dup(path); + return charset_wire_apply(path); +} + static unsigned long long per_file_serialize_size(File* file, bool use_metadata) { - return sizeof(size_t) + strlen(file->path) + - (use_metadata ? sizeof(int) + (file->metadata ? FILE_METADATA_WIRE_SIZE : 0) : 0) + - sizeof(size_t) + file->data->size; + unsigned long long size = sizeof(size_t); + char* wire_path = chunk_encode_wire(file_wire_path(file)); + if (!wire_path) + return 0; + size_t path_len = strlen(wire_path); + free(wire_path); + unsigned long long metadata_size = + use_metadata ? sizeof(int) + (file->metadata ? FILE_METADATA_WIRE_SIZE : 0) : 0; + if ((unsigned long long)path_len > ULLONG_MAX - size) + return 0; + size += path_len; + if (metadata_size > ULLONG_MAX - size) + return 0; + size += metadata_size; + /* Entry type marker: 0 = regular file, 1 = explicit directory entry, + 2 = symlink entry (carries its target string), 3 = special/device node + (recreated by the receiver). */ + if (sizeof(int) > ULLONG_MAX - size) + return 0; + size += sizeof(int); + /* A special node also carries its rdev major/minor. */ + if (file->is_special) { + if (2 * sizeof(int32_t) > ULLONG_MAX - size) + return 0; + size += 2 * sizeof(int32_t); + } + if (sizeof(size_t) > ULLONG_MAX - size) + return 0; + size += sizeof(size_t); + if ((unsigned long long)file->data->size > ULLONG_MAX - size) + return 0; + size += file->data->size; + /* Symlink entries append the target string (length-prefixed). */ + if (file->is_symlink) { + char* wire_target = chunk_encode_wire(file->symlink_target ? file->symlink_target : ""); + if (!wire_target) + return 0; + size_t target_len = strlen(wire_target); + free(wire_target); + if (sizeof(size_t) > ULLONG_MAX - size) + return 0; + size += sizeof(size_t); + if ((unsigned long long)target_len > ULLONG_MAX - size) + return 0; + size += target_len; + } + return size; } Data* chunk_serialize(Chunk* chunk, bool use_metadata) { + if (!chunk || chunk->element_count < 0 || (chunk->element_count > 0 && chunk->items == NULL)) + return NULL; unsigned long long data_size = 0; for (int i = 0; i < chunk->element_count; i++) { - data_size += per_file_serialize_size(chunk->items[i], use_metadata); + if (!chunk->items[i] || !chunk->items[i]->path || !chunk->items[i]->data || + (chunk->items[i]->data->size > 0 && !chunk->items[i]->data->data) || + chunk->items[i]->path[0] == '\0' || has_path_traversal(chunk->items[i]->path) || + (file_wire_path(chunk->items[i]))[0] == '\0') + return NULL; + unsigned long long file_size = per_file_serialize_size(chunk->items[i], use_metadata); + if (file_size == 0 || file_size > ULLONG_MAX - data_size || data_size + file_size > SIZE_MAX) + return NULL; + data_size += file_size; } Data* data = data_create_empty(data_size); if (data == NULL) { @@ -65,11 +146,30 @@ Data* chunk_serialize(Chunk* chunk, bool use_metadata) { char* data_pointer = data->data; for (int i = 0; i < chunk->element_count; i++) { File* file = chunk->items[i]; - size_t path_len = strlen(file->path); + char* wire_path = chunk_encode_wire(file_wire_path(file)); + if (wire_path == NULL) { + data_destroy(data); + return NULL; + } + size_t path_len = strlen(wire_path); memcpy(data_pointer, &path_len, sizeof(size_t)); data_pointer += sizeof(size_t); - memcpy(data_pointer, file->path, path_len); + memcpy(data_pointer, wire_path, path_len); data_pointer += path_len; + free(wire_path); + + int entry_type = file->is_dir ? 1 : (file->is_symlink ? 2 : (file->is_special ? 3 : 0)); + memcpy(data_pointer, &entry_type, sizeof(int)); + data_pointer += sizeof(int); + + if (file->is_special) { + int32_t special_major = file->rdev_major; + int32_t special_minor = file->rdev_minor; + memcpy(data_pointer, &special_major, sizeof(special_major)); + data_pointer += sizeof(special_major); + memcpy(data_pointer, &special_minor, sizeof(special_minor)); + data_pointer += sizeof(special_minor); + } if (use_metadata) metadata_to_buf(&data_pointer, file->metadata); @@ -77,18 +177,43 @@ Data* chunk_serialize(Chunk* chunk, bool use_metadata) { size_t file_data_size = file->data->size; memcpy(data_pointer, &file_data_size, sizeof(size_t)); data_pointer += sizeof(size_t); - memcpy(data_pointer, file->data->data, file_data_size); + if (file_data_size > 0) + memcpy(data_pointer, file->data->data, file_data_size); data_pointer += file_data_size; + + if (file->is_symlink) { + char* wire_target = chunk_encode_wire(file->symlink_target ? file->symlink_target : ""); + if (wire_target == NULL) { + data_destroy(data); + return NULL; + } + size_t target_len = strlen(wire_target); + memcpy(data_pointer, &target_len, sizeof(size_t)); + data_pointer += sizeof(size_t); + if (target_len > 0) + memcpy(data_pointer, wire_target, target_len); + data_pointer += target_len; + free(wire_target); + } } return data; } Chunk* chunk_deserialize(Data* data, bool use_metadata) { + if (!data || (!data->data && data->size != 0)) + return NULL; ArrayList* files = array_list_create(file_destroy); + if (files == NULL) + return NULL; char* data_pointer = data->data; size_t remaining_size = data->size; while (remaining_size > 0) { + if ((unsigned int)files->size >= MAX_FILES_PER_CHUNK) { + log_message(LOG_LEVEL_ERROR, "Chunk contains too many files"); + array_list_delete(files); + return NULL; + } if (remaining_size < sizeof(size_t)) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for path length"); array_list_delete(files); @@ -100,48 +225,141 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) { data_pointer += sizeof(size_t); remaining_size -= sizeof(size_t); - if (remaining_size < path_len) { + if (path_len > SIZE_MAX - 1 || remaining_size < path_len) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for path"); array_list_delete(files); return NULL; } - char* path = malloc(path_len + 1); + if (path_len == SIZE_MAX) { + array_list_delete(files); + return NULL; + } + char* path = protocol_alloc(path_len + 1); if (path == NULL) { - perror("Could not allocate memory for file path"); + log_perror("Could not allocate memory for file path"); array_list_delete(files); return NULL; } memcpy(path, data_pointer, path_len); path[path_len] = '\0'; + if (memchr(path, '\0', path_len) != NULL) { + free(path); + array_list_delete(files); + return NULL; + } data_pointer += path_len; remaining_size -= path_len; + /* --iconv: the blob holds the wire charset; translate it to the receiver's + local charset before validation and creation so the destination gets the + local name. A name that cannot be decoded fails the file cleanly. */ + if (charset_wire_active()) { + char* local_path = charset_wire_apply(path); + free(path); + if (local_path == NULL) { + log_message(LOG_LEVEL_ERROR, + "--iconv: received chunk file name cannot be converted to the local charset"); + array_list_delete(files); + return NULL; + } + path = local_path; + path_len = strlen(path); + } + + if (path_len == 0 || has_path_traversal(path)) { + free(path); + array_list_delete(files); + return NULL; + } + File* file = file_create(path); free(path); + if (file == NULL) { + array_list_delete(files); + return NULL; + } + + if (remaining_size < sizeof(int)) { + log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for entry type"); + file_destroy(file); + array_list_delete(files); + return NULL; + } + int entry_type; + memcpy(&entry_type, data_pointer, sizeof(int)); + if (entry_type != 0 && entry_type != 1 && entry_type != 2 && entry_type != 3) { + log_message(LOG_LEVEL_ERROR, "Invalid chunk format: bad entry type"); + file_destroy(file); + array_list_delete(files); + return NULL; + } + file->is_dir = entry_type == 1; + file->is_symlink = entry_type == 2; + file->is_special = entry_type == 3; + data_pointer += sizeof(int); + remaining_size -= sizeof(int); + + if (file->is_special) { + if (remaining_size < 2 * (int32_t)sizeof(int32_t)) { + log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for special rdev"); + file_destroy(file); + array_list_delete(files); + return NULL; + } + int32_t special_major, special_minor; + memcpy(&special_major, data_pointer, sizeof(special_major)); + data_pointer += sizeof(special_major); + memcpy(&special_minor, data_pointer, sizeof(special_minor)); + data_pointer += sizeof(special_minor); + remaining_size -= 2 * sizeof(int32_t); + /* Reject an out-of-range/negative rdev here as a malformed chunk (the + same 0xffff / 0x00ffffff bounds file_special_rdev_valid uses), so a + bogus large-but-positive rdev is refused cleanly instead of being + deferred to the creation site where it would abort after the frame. */ + if (special_major < 0 || special_minor < 0 || special_major > 0xffff || + special_minor > 0x00ffffff) { + log_message(LOG_LEVEL_ERROR, "Invalid chunk format: out-of-range special rdev"); + file_destroy(file); + array_list_delete(files); + return NULL; + } + file->rdev_major = special_major; + file->rdev_minor = special_minor; + } if (use_metadata) { if (remaining_size < sizeof(int)) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for metadata"); + file_destroy(file); array_list_delete(files); return NULL; } // Peek at present flag to determine total size needed before reading int present_flag; memcpy(&present_flag, data_pointer, sizeof(int)); - if (present_flag && remaining_size < sizeof(int) + FILE_METADATA_WIRE_SIZE) { + if ((present_flag != 0 && present_flag != 1) || + (present_flag == 1 && remaining_size < sizeof(int) + FILE_METADATA_WIRE_SIZE)) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for metadata body"); + file_destroy(file); array_list_delete(files); return NULL; } file->metadata = metadata_from_buf(&data_pointer); remaining_size -= sizeof(int); - if (file->metadata) + if (present_flag == 1) { + if (file->metadata == NULL) { + file_destroy(file); + array_list_delete(files); + return NULL; + } remaining_size -= FILE_METADATA_WIRE_SIZE; + } } if (remaining_size < sizeof(size_t)) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for data size"); + file_destroy(file); array_list_delete(files); return NULL; } @@ -153,29 +371,111 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) { if (remaining_size < file_data_size) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for file content"); + file_destroy(file); array_list_delete(files); return NULL; } - void* file_data = malloc(file_data_size); + // Reject individual file data larger than the maximum allowed size. + if (file_data_size > MAX_FILE_DATA_SIZE) { + log_message(LOG_LEVEL_ERROR, "File data size %zu exceeds maximum %llu", file_data_size, + (unsigned long long)MAX_FILE_DATA_SIZE); + file_destroy(file); + array_list_delete(files); + return NULL; + } + + size_t allocation_size = file_data_size > 0 ? file_data_size : 1; + void* file_data = protocol_alloc(allocation_size); if (file_data == NULL) { - perror("Could not allocate memory for file data"); + log_perror("Could not allocate memory for file data"); + file_destroy(file); array_list_delete(files); return NULL; } memcpy(file_data, data_pointer, file_data_size); + Data* replacement = data_create(file_data, file_data_size); + if (replacement == NULL) { + file_destroy(file); + array_list_delete(files); + return NULL; + } data_destroy(file->data); - file->data = data_create(file_data, file_data_size); + file->data = replacement; data_pointer += file_data_size; remaining_size -= file_data_size; - array_list_add(files, file); + if (file->is_symlink) { + if (remaining_size < sizeof(size_t)) { + log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for symlink target"); + file_destroy(file); + array_list_delete(files); + return NULL; + } + size_t target_len; + memcpy(&target_len, data_pointer, sizeof(size_t)); + data_pointer += sizeof(size_t); + remaining_size -= sizeof(size_t); + if (target_len == 0 || remaining_size < target_len) { + log_message(LOG_LEVEL_ERROR, "Invalid chunk format: bad symlink target"); + file_destroy(file); + array_list_delete(files); + return NULL; + } + char* target = protocol_alloc(target_len + 1); + if (!target) { + log_perror("Could not allocate memory for symlink target"); + file_destroy(file); + array_list_delete(files); + return NULL; + } + memcpy(target, data_pointer, target_len); + target[target_len] = '\0'; + if (memchr(target, '\0', target_len) != NULL) { + free(target); + file_destroy(file); + array_list_delete(files); + return NULL; + } + /* The symlink target also rides the wire charset; decode it to the local + charset like the path (a target is a path). */ + if (charset_wire_active()) { + char* local_target = charset_wire_apply(target); + free(target); + if (local_target == NULL) { + log_message(LOG_LEVEL_ERROR, + "--iconv: received chunk symlink target cannot be converted to the local " + "charset"); + file_destroy(file); + array_list_delete(files); + return NULL; + } + target = local_target; + } + file->symlink_target = target; + data_pointer += target_len; + remaining_size -= target_len; + } + + if (!array_list_add(files, file)) { + file_destroy(file); + array_list_delete(files); + return NULL; + } } File** file_array = (File**)array_list_to_array(files); + if (files->size > 0 && file_array == NULL) { + array_list_delete(files); + return NULL; + } Chunk* chunk = chunk_create(file_array, files->size); free(file_array); + if (chunk == NULL) { + array_list_delete(files); + return NULL; + } files->item_destroyer = NULL; array_list_delete(files); @@ -183,33 +483,47 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) { } Data* chunk_compress(Chunk* chunk, int compression_level, bool use_metadata) { + return chunk_compress_with_threads(chunk, compression_level, use_metadata, 0); +} + +Data* chunk_compress_with_threads(Chunk* chunk, int compression_level, bool use_metadata, + int compression_threads) { log_message(LOG_LEVEL_DEBUG, "Starting to compress chunk"); Data* serialized = chunk_serialize(chunk, use_metadata); if (serialized == NULL) return NULL; - Data* compressed = data_compress(serialized, compression_level); + Data* compressed = data_compress_with_threads(serialized, compression_level, compression_threads); data_destroy(serialized); if (compressed == NULL) return NULL; - log_message(LOG_LEVEL_DEBUG, "Chunk successfully compressed"); + log_debug_message(LOG_DEBUG_PACK, "Chunk successfully compressed"); return compressed; } Chunk* receive_chunk_data(int fd, const Config* config) { - Data* chunk_data = receive_data(fd); + Data* chunk_data = receive_data_limited(fd, MAX_CHUNK_SIZE); if (chunk_data == NULL) { log_message(LOG_LEVEL_ERROR, "Failed to receive chunk data"); return NULL; } Data* data_to_process = chunk_data; if (config->use_compression) { - data_to_process = data_decompress(chunk_data); + data_to_process = data_decompress_limited(chunk_data, MAX_CHUNK_SIZE); data_destroy(chunk_data); if (data_to_process == NULL) { log_message(LOG_LEVEL_ERROR, "Failed to decompress chunk"); return NULL; } } + + // Reject chunks larger than the maximum allowed size to prevent OOM. + if (data_to_process->size > MAX_CHUNK_SIZE) { + log_message(LOG_LEVEL_ERROR, "Chunk size %zu exceeds maximum %llu", data_to_process->size, + (unsigned long long)MAX_CHUNK_SIZE); + data_destroy(data_to_process); + return NULL; + } + Chunk* chunk = chunk_deserialize(data_to_process, config->use_metadata); data_destroy(data_to_process); if (chunk == NULL) diff --git a/src/shared/chunk.h b/src/shared/chunk.h index dbba7b7..2c04e85 100644 --- a/src/shared/chunk.h +++ b/src/shared/chunk.h @@ -19,6 +19,8 @@ void chunk_destroy(void* chunk); Data* chunk_serialize(Chunk* chunk, bool use_metadata); Chunk* chunk_deserialize(Data* data, bool use_metadata); Data* chunk_compress(Chunk* chunk, int compression_level, bool use_metadata); +Data* chunk_compress_with_threads(Chunk* chunk, int compression_level, bool use_metadata, + int compression_threads); Chunk* receive_chunk_data(int fd, const Config* config); #endif diff --git a/src/shared/compression.c b/src/shared/compression.c index c1eb1f1..e4d5b95 100644 --- a/src/shared/compression.c +++ b/src/shared/compression.c @@ -1,30 +1,53 @@ #include "compression.h" #include "data.h" #include "log.h" -#include "stdlib.h" -#include "string.h" +#include "protocol.h" +#include +#include +#include +#include #include -#include "zstd.h" +#include +#include #define INITIAL_DECOMPRESS_BUF_SIZE (1024 * 1024) +#define MAX_DECOMPRESSED_SIZE (100ULL * 1024 * 1024) /* 100 MB hard ceiling */ -static const char* SKIP_COMPRESSION_EXTENSIONS[] = {".jpg", ".jpeg", ".png", ".gif", ".mp4", ".mkv", - ".zip", ".gz", ".xz", ".zst", NULL}; +static char* SKIP_COMPRESSION_EXTENSIONS[] = {".jpg", ".jpeg", ".png", ".gif", ".mp4", ".mkv", + ".zip", ".gz", ".xz", ".zst", NULL}; bool compression_should_skip(const char* path) { + return compression_should_skip_with_suffixes(path, NULL, -1); +} + +bool compression_should_skip_with_suffixes(const char* path, char* const* suffixes, int count) { if (!path) return false; const char* dot = strrchr(path, '.'); if (!dot) return false; - for (int i = 0; SKIP_COMPRESSION_EXTENSIONS[i]; i++) { - if (strcasecmp(dot, SKIP_COMPRESSION_EXTENSIONS[i]) == 0) + if (count < 0) { + suffixes = SKIP_COMPRESSION_EXTENSIONS; + count = 0; + while (SKIP_COMPRESSION_EXTENSIONS[count]) + count++; + } + for (int i = 0; i < count; i++) { + if (strcasecmp(dot, suffixes[i]) == 0) return true; } return false; } Data* data_compress(Data* data_to_compress, int compression_level) { + return data_compress_with_threads(data_to_compress, compression_level, 0); +} + +Data* data_compress_with_threads(Data* data_to_compress, int compression_level, + int compression_threads) { + if (!data_to_compress || (!data_to_compress->data && data_to_compress->size != 0) || + compression_threads < 0 || compression_threads > COMPRESSION_MAX_THREADS) + return NULL; log_message(LOG_LEVEL_DEBUG, "Starting to compress data"); size_t dst_size = ZSTD_compressBound(data_to_compress->size); Data* compressed_data = data_create_empty(dst_size); @@ -46,6 +69,30 @@ Data* data_compress(Data* data_to_compress, int compression_level) { return NULL; } + if (compression_threads > 0) { + long online_cpus = sysconf(_SC_NPROCESSORS_ONLN); + int available_threads = online_cpus > 0 && online_cpus < compression_threads + ? (int)online_cpus + : compression_threads; + zret = ZSTD_CCtx_setParameter(cctx, ZSTD_c_nbWorkers, available_threads); + if (ZSTD_isError(zret)) { + log_message(LOG_LEVEL_ERROR, "Failed to set compression threads: %s", + ZSTD_getErrorName(zret)); + ZSTD_freeCCtx(cctx); + data_destroy(compressed_data); + return NULL; + } + /* Streaming compression needs the source size before threaded mode can end a frame. */ + zret = ZSTD_CCtx_setPledgedSrcSize(cctx, data_to_compress->size); + if (ZSTD_isError(zret)) { + log_message(LOG_LEVEL_ERROR, "Failed to set compression source size: %s", + ZSTD_getErrorName(zret)); + ZSTD_freeCCtx(cctx); + data_destroy(compressed_data); + return NULL; + } + } + ZSTD_inBuffer input = {data_to_compress->data, data_to_compress->size, 0}; ZSTD_outBuffer output = {compressed_data->data, dst_size, 0}; @@ -63,13 +110,16 @@ Data* data_compress(Data* data_to_compress, int compression_level) { compressed_data->size = output.pos; ZSTD_freeCCtx(cctx); - log_message(LOG_LEVEL_DEBUG, "Data succesfully compressed from %zu to %zu", - data_to_compress->size, compressed_data->size); + log_debug_message(LOG_DEBUG_UTIL, "Data succesfully compressed from %zu to %zu", + data_to_compress->size, compressed_data->size); return compressed_data; } -Data* data_decompress(Data* compressed_data) { - log_message(LOG_LEVEL_DEBUG, "Start to decompress data"); +Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) { + if (!compressed_data || (!compressed_data->data && compressed_data->size != 0) || + maximum_size == 0) + return NULL; + log_debug_message(LOG_DEBUG_UTIL, "Start to decompress data"); unsigned long long dst_size = ZSTD_getFrameContentSize(compressed_data->data, compressed_data->size); if (ZSTD_isError(dst_size)) { @@ -81,10 +131,18 @@ Data* data_decompress(Data* compressed_data) { // ZSTD_CONTENTSIZE_UNKNOWN (~2^64) can cause massive allocation; // fall back to a conservative estimate (3x compressed size) when unknown. if (dst_size == ZSTD_CONTENTSIZE_UNKNOWN) { + if (compressed_data->size > ULLONG_MAX / 3) + return NULL; dst_size = compressed_data->size * 3; if (dst_size < INITIAL_DECOMPRESS_BUF_SIZE) dst_size = INITIAL_DECOMPRESS_BUF_SIZE; } + unsigned long long hard_limit = + maximum_size < MAX_DECOMPRESSED_SIZE ? maximum_size : MAX_DECOMPRESSED_SIZE; + if (dst_size > hard_limit) { + log_message(LOG_LEVEL_ERROR, "Declared decompressed size exceeds %llu bytes", hard_limit); + return NULL; + } ZSTD_DCtx* dctx = ZSTD_createDCtx(); if (!dctx) { @@ -93,6 +151,8 @@ Data* data_decompress(Data* compressed_data) { } size_t buf_size = (dst_size > 0) ? (size_t)dst_size : INITIAL_DECOMPRESS_BUF_SIZE; + if (buf_size > maximum_size) + buf_size = maximum_size; Data* uncompressed_data = data_create_empty(buf_size); if (!uncompressed_data) { log_message(LOG_LEVEL_ERROR, "Failed to allocate decompression buffer"); @@ -113,8 +173,17 @@ Data* data_decompress(Data* compressed_data) { return NULL; } if (ret > 0 && output.pos == output.size) { + if (buf_size >= hard_limit || buf_size > SIZE_MAX / 2) { + log_message(LOG_LEVEL_ERROR, "Decompressed data exceeds %llu bytes", + (unsigned long long)MAX_DECOMPRESSED_SIZE); + ZSTD_freeDCtx(dctx); + data_destroy(uncompressed_data); + return NULL; + } buf_size *= 2; - void* new_data = realloc(uncompressed_data->data, buf_size); + if (buf_size > hard_limit) + buf_size = (size_t)hard_limit; + void* new_data = protocol_realloc(uncompressed_data->data, buf_size); if (!new_data) { log_message(LOG_LEVEL_ERROR, "Failed to grow decompression buffer"); ZSTD_freeDCtx(dctx); @@ -130,6 +199,10 @@ Data* data_decompress(Data* compressed_data) { uncompressed_data->size = output.pos; ZSTD_freeDCtx(dctx); - log_message(LOG_LEVEL_DEBUG, "Decompressed data successfully"); + log_debug_message(LOG_DEBUG_UTIL, "Decompressed data successfully"); return uncompressed_data; } + +Data* data_decompress(Data* compressed_data) { + return data_decompress_limited(compressed_data, MAX_DECOMPRESSED_SIZE); +} diff --git a/src/shared/compression.h b/src/shared/compression.h index d392f25..b30622d 100644 --- a/src/shared/compression.h +++ b/src/shared/compression.h @@ -4,8 +4,14 @@ #include "data.h" #include +#define COMPRESSION_MAX_THREADS 64 + Data* data_compress(Data* data_to_compress, int compression_level); +Data* data_compress_with_threads(Data* data_to_compress, int compression_level, + int compression_threads); Data* data_decompress(Data* compressed_data); +Data* data_decompress_limited(Data* compressed_data, size_t maximum_size); bool compression_should_skip(const char* path); +bool compression_should_skip_with_suffixes(const char* path, char* const* suffixes, int count); #endif diff --git a/src/shared/config.c b/src/shared/config.c index e31df05..61ea67c 100644 --- a/src/shared/config.c +++ b/src/shared/config.c @@ -1,5 +1,12 @@ #include "config.h" +#include "charset.h" +#include "chmod.h" +#include "credentials.h" +#include "daemon_conf.h" +#include "delay_updates.h" #include "delta.h" +#include "file_list.h" +#include "identity.h" #include "log.h" #include "protocol.h" #include "utils.h" @@ -7,11 +14,10 @@ #include #include #include +#include +#include -Config* config_create(void) { - Config* config = malloc(sizeof(Config)); - if (!config) - return NULL; +static void config_set_defaults(Config* config) { config->version = str_dup(PROTOCOL_VERSION); config->send_directory = NULL; config->receive_root_directory = NULL; @@ -20,15 +26,24 @@ Config* config_create(void) { config->use_chunk_serialization = false; config->use_compression = false; config->use_metadata = false; + config->use_executability = false; + config->metadata_explicitly_disabled = false; config->show_progress = false; config->dry_run = false; + config->remove_source_files = false; config->use_delete = false; config->compression_level = 5; + config->compression_threads = 0; config->use_sendfile = false; config->chunk_size = DEFAULT_CHUNK_SIZE; config->ssh_port = 22; config->transport = TRANSPORT_TCP; config->ssh_destination = NULL; + config->module = NULL; + config->auth_user = NULL; + config->auth_password = NULL; + config->password_file = NULL; + config->iconv_spec = NULL; config->fastsync_server_path = NULL; config->exclude_patterns = NULL; config->exclude_count = 0; @@ -36,8 +51,14 @@ Config* config_create(void) { config->include_count = 0; config->max_size = 0; config->min_size = 0; + config->max_alloc = DEFAULT_MAX_ALLOC; config->use_incremental = false; + config->ignore_times = false; + config->size_only = false; config->use_delta = false; + config->whole_file = false; + config->fuzzy = false; + config->modify_window = 0; config->delta_block_size = DELTA_BLOCK_SIZE_DEFAULT; config->delta_max_file_size = DELTA_MAX_FILE_SIZE; config->use_tls = false; @@ -60,39 +81,414 @@ Config* config_create(void) { config->copy_links = false; config->safe_links = false; config->copy_unsafe_links = false; + config->copy_dirlinks = false; + config->munge_links = false; + config->keep_dirlinks = false; config->preserve_hard_links = false; config->preserve_acls = false; config->preserve_xattrs = false; config->preserve_devices = false; config->preserve_sparse = false; + config->preserve_specials = false; + config->copy_devices = false; + config->write_devices = false; config->itemize_changes = false; config->out_format = NULL; + config->log_file_format = NULL; config->info_level = 0; config->debug_level = 0; config->list_only = false; config->human_readable = false; + config->eight_bit_output = false; + config->existing = false; + config->ignore_existing = false; config->update = false; config->inplace = false; + config->delay_updates = false; + config->use_fsync = false; config->append = false; config->append_verify = false; + config->preallocate = false; config->delete_excluded = false; config->delete_after = false; - config->max_delete = 0; + config->max_delete = -1; + config->ignore_errors = false; + config->force_delete = false; + config->ignore_missing_args = false; + config->delete_missing_args = false; config->filters = NULL; config->files_from = NULL; + config->files_from_set = NULL; + config->from0 = false; config->cvs_exclude = false; + config->per_dir_filter = false; config->prune_empty_dirs = false; + config->one_file_system = false; config->relative = false; + config->no_implied_dirs = false; + config->dirs = false; + config->mkpath = false; config->rsh_command = NULL; - config->rsync_path = NULL; + config->blocking_io = false; + config->outbuf = OUTBUF_BLOCK; + config->old_args = false; config->temp_dir = NULL; - config->compare_dest = NULL; - config->copy_dest = NULL; - config->link_dest = NULL; + config->remote_options = NULL; + config->remote_option_count = 0; + config->basis_dirs = NULL; + config->basis_count = 0; + config->partial_dir = NULL; + config->suffix = NULL; + config->delete_before = false; + config->delete_during = false; + config->delete_delay = false; + config->address = NULL; + config->bind_address = NULL; + config->ipv6 = false; + config->ipv4 = false; + config->sockopts = NULL; + config->sockopt_count = 0; + config->daemon = false; + config->daemon_config = NULL; + config->server_mode = false; + config->no_motd = false; + config->checksum = false; + config->checksum_algo = CHECKSUM_ALGO_XXH64; + config->checksum_seed = 0; + config->compress_choice = NULL; + config->chmod_spec = NULL; + config->skip_compress_suffixes = NULL; + config->skip_compress_count = 0; + config->skip_compress_set = false; + config->numeric_ids = false; + config->chown_uid_set = false; + config->chown_uid = 0; + config->chown_gid_set = false; + config->chown_gid = 0; + config->usermap = NULL; + config->usermap_count = 0; + config->groupmap = NULL; + config->groupmap_count = 0; + config->super_mode = SUPER_MODE_AUTO; + config->delay_context = NULL; + config->preserve_atimes = false; + config->preserve_crtimes = false; + config->omit_dir_times = false; + config->omit_link_times = false; + config->open_noatime = false; + config->use_xattrs = false; + config->fake_super = false; + config->copy_as_set = false; + config->copy_as_uid = 0; + config->copy_as_gid = 0; + config->trust_sender = false; + config->stop_after_mins = 0; + config->stop_at = 0; + config->stop_at_set = false; + config->write_batch = NULL; + config->only_write_batch = NULL; + config->read_batch = NULL; +} + +static bool valid_wire_bool(int value) { + return value == 0 || value == 1; +} + +static bool receive_wire_bool(int fd, bool* value) { + int wire_value; + if (!receive_int(fd, &wire_value) || !valid_wire_bool(wire_value)) + return false; + *value = wire_value != 0; + return true; +} + +static bool validate_received_config(const Config* config) { + return valid_wire_bool(config->save_to_disk) && valid_wire_bool(config->use_multithreading) && + valid_wire_bool(config->use_chunk_serialization) && + valid_wire_bool(config->use_compression) && valid_wire_bool(config->use_metadata) && + valid_wire_bool(config->use_executability) && valid_wire_bool(config->use_sendfile) && + valid_wire_bool(config->use_delete) && valid_wire_bool(config->use_incremental) && + valid_wire_bool(config->size_only) && valid_wire_bool(config->ignore_times) && + valid_wire_bool(config->use_delta) && valid_wire_bool(config->backup) && + valid_wire_bool(config->fuzzy) && valid_wire_bool(config->remove_source_files) && + valid_wire_bool(config->follow_symlinks) && valid_wire_bool(config->copy_links) && + valid_wire_bool(config->safe_links) && valid_wire_bool(config->copy_unsafe_links) && + valid_wire_bool(config->preserve_hard_links) && valid_wire_bool(config->preserve_acls) && + valid_wire_bool(config->preserve_xattrs) && valid_wire_bool(config->preserve_devices) && + valid_wire_bool(config->preserve_sparse) && valid_wire_bool(config->preserve_specials) && + valid_wire_bool(config->copy_devices) && valid_wire_bool(config->write_devices) && + valid_wire_bool(config->ignore_existing) && valid_wire_bool(config->existing) && + valid_wire_bool(config->update) && valid_wire_bool(config->inplace) && + valid_wire_bool(config->append) && valid_wire_bool(config->use_fsync) && + valid_wire_bool(config->append_verify) && valid_wire_bool(config->delete_excluded) && + valid_wire_bool(config->force_delete) && valid_wire_bool(config->delete_missing_args) && + valid_wire_bool(config->delete_after) && valid_wire_bool(config->preallocate) && + valid_wire_bool(config->delete_delay) && valid_wire_bool(config->delete_during) && + valid_wire_bool(config->relative) && valid_wire_bool(config->prune_empty_dirs) && + valid_wire_bool(config->delay_updates) && valid_wire_bool(config->mkpath) && + !(config->delay_updates && config->inplace) && + !(config->delay_updates && delay_updates_staging_name_conflict(config->backup_dir)) && + valid_wire_bool(config->partial) && valid_wire_bool(config->delete_before) && + valid_wire_bool(config->checksum) && valid_wire_bool(config->eight_bit_output) && + checksum_algo_valid(config->checksum_algo) && config_has_valid_delete_timing(config) && + identity_wire_valid(config) && + !(config->skip_compress_set && config->use_chunk_serialization) && + /* --append / --append-verify tail resume needs the per-file check, + which chunk serialization -s disables: reject on the receiver too + so a -s sender cannot negotiate an inert append mode. */ + !((config->append || config->append_verify) && config->use_chunk_serialization) && + !(config->preserve_hard_links && config->use_chunk_serialization) && + !(config->preserve_hard_links && (config->append || config->append_verify)) && + /* The xattr block rides the per-file streaming frame, which -s drops. */ + !((config->preserve_xattrs || config->preserve_acls) && config->use_chunk_serialization) && + valid_wire_bool(config->preserve_atimes) && valid_wire_bool(config->preserve_crtimes) && + valid_wire_bool(config->omit_dir_times) && valid_wire_bool(config->omit_link_times) && + valid_wire_bool(config->munge_links) && valid_wire_bool(config->keep_dirlinks) && + valid_wire_bool(config->fake_super) && + (!config->copy_as_set || (config->copy_as_uid >= 0 && config->copy_as_gid >= 0)) && + /* --copy-as forces ownership through the metadata path; without + metadata it would pass the privilege gate but silently chown + nothing. Refuse the frame instead. */ + (!config->copy_as_set || config->use_metadata) && + (!config->use_compression || + (config->compression_level >= 1 && config->compression_level <= 22)) && + config->chunk_size > 0 && config->chunk_size <= MAX_CHUNK_SIZE && + config->delta_block_size >= DELTA_BLOCK_SIZE_MIN && + config->delta_block_size <= DELTA_BLOCK_SIZE_MAX && + config->delta_max_file_size <= DELTA_MAX_FILE_SIZE && config->modify_window >= 0 && + config->max_delete >= -1 && config->skip_compress_count >= 0 && + config->skip_compress_count <= 10000 && config->max_alloc > 0 && + (!config->chmod_spec || !*config->chmod_spec || + chmod_apply(0, config->chmod_spec, &(mode_t){0})) && + /* The received --iconv CONVERT_SPEC is untrusted input that drives + the receiver's path decoding: reject a malformed spec or an + unsupported charset name so the run is refused up front instead of + every received file name failing mid-transfer. A NULL spec (iconv + disabled) is always accepted. */ + (!config->iconv_spec || charset_spec_valid(config->iconv_spec)) && + /* --super / --no-super: the received tri-state must be one of the + defined values (AUTO/ON/OFF); anything else is a malformed frame. */ + config->super_mode >= SUPER_MODE_AUTO && config->super_mode <= SUPER_MODE_OFF; +} + +Config* config_create(void) { + Config* config = malloc(sizeof(Config)); + if (!config) + return NULL; + config_set_defaults(config); return config; } -bool is_remote_dest(const char* s) { +bool config_delete_timing_early(const Config* config) { + if (!config) + return false; + return config->delete_before || config->delete_during; +} + +/* A delete-timing flag is only meaningful together with --delete. At most one + of the four flags may be set; several simultaneous timings are a client bug + and are rejected on both ends. */ +bool config_has_valid_delete_timing(const Config* config) { + if (!config) + return false; + if (!config->use_delete) + return !config->delete_before && !config->delete_during && !config->delete_delay && + !config->delete_after; + int timing_count = (config->delete_before ? 1 : 0) + (config->delete_during ? 1 : 0) + + (config->delete_delay ? 1 : 0) + (config->delete_after ? 1 : 0); + return timing_count <= 1; +} + +bool config_has_basis(const Config* config) { + return config && config->basis_count > 0; +} + +/* A basis-dir path travels from the client to the receiver and is resolved + * below the destination root, so it must be a non-empty relative path with no + * "." or ".." component and no traversal: an absolute or escaping path would + * make the receiver read or link files outside its authorized root. + * + * Returns a malloc'd CANONICAL copy of an accepted path, or NULL when the path + * is rejected. Canonicalization collapses interior empty components ("a//b" -> + * "a/b"), drops "." components and trailing "/"s, so validation, the delete + * walker prefix match and the receiver's basis lookup all agree on one form. + * The normalizer is the single source of truth for both config_basis_path_valid + * and config_basis_append. */ +static char* basis_path_normalize(const char* path) { + if (!path || path[0] == '\0' || path[0] == '/' || has_path_traversal(path)) + return NULL; + if (strcmp(path, ".") == 0) + return NULL; + char* dup = str_dup(path); + if (!dup) + return NULL; + size_t out_len = 0; + char* out = malloc(strlen(path) + 1); + if (!out) { + free(dup); + return NULL; + } + char* saveptr = NULL; + bool ok = true; + for (char* part = strtok_r(dup, "/", &saveptr); part; part = strtok_r(NULL, "/", &saveptr)) { + if (strcmp(part, "..") == 0) { + ok = false; + break; + } + if (strcmp(part, ".") == 0) + continue; + if (out_len > 0) + out[out_len++] = '/'; + size_t len = strlen(part); + memcpy(out + out_len, part, len); + out_len += len; + } + free(dup); + if (!ok || out_len == 0) { + free(out); + return NULL; + } + out[out_len] = '\0'; + return out; +} + +bool config_basis_path_valid(const char* path) { + char* normalized = basis_path_normalize(path); + if (!normalized) + return false; + free(normalized); + return true; +} + +int config_basis_append(Config* config, BasisDestType type, const char* path) { + if (!config || + (type != BASIS_DEST_COMPARE && type != BASIS_DEST_COPY && type != BASIS_DEST_LINK) || + config->basis_count >= MAX_BASIS_DIRS) + return -1; + char* normalized = basis_path_normalize(path); + if (!normalized) + return -1; + BasisDest* grown = realloc(config->basis_dirs, (config->basis_count + 1) * sizeof(BasisDest)); + if (!grown) { + free(normalized); + return -1; + } + config->basis_dirs = grown; + config->basis_dirs[config->basis_count].type = type; + config->basis_dirs[config->basis_count].path = normalized; + config->basis_count++; + return 0; +} + +/* Strict --sockopts allowlist: map an option NAME to its SockOptId, or -1 when + * the name is not on the allowlist. The list is intentionally closed so an + * unknown option is an error, never a silent no-op. */ +static int sockopt_id_from_name(const char* name) { + if (strcmp(name, "TCP_NODELAY") == 0) + return SOCKOPT_TCP_NODELAY; + if (strcmp(name, "SO_KEEPALIVE") == 0) + return SOCKOPT_SO_KEEPALIVE; + if (strcmp(name, "SO_RCVBUF") == 0) + return SOCKOPT_SO_RCVBUF; + if (strcmp(name, "SO_SNDBUF") == 0) + return SOCKOPT_SO_SNDBUF; + if (strcmp(name, "SO_REUSEADDR") == 0) + return SOCKOPT_SO_REUSEADDR; + return -1; +} + +static bool sockopt_is_boolean(SockOptId id) { + return id == SOCKOPT_TCP_NODELAY || id == SOCKOPT_SO_KEEPALIVE || id == SOCKOPT_SO_REUSEADDR; +} + +/* Parse one SockOptId's value. Booleans accept only 0/1 (a numeric "on" is + * rejected rather than coerced); buffer sizes accept any non-negative int. + * Returns 0 on success, -1 on a bad value. */ +static int sockopt_parse_value(SockOptId id, const char* value, int* out) { + if (sockopt_is_boolean(id)) { + if (strcmp(value, "0") == 0) { + *out = 0; + return 0; + } + if (strcmp(value, "1") == 0) { + *out = 1; + return 0; + } + return -1; + } + if (!value || *value == '\0') + return -1; + char* end; + errno = 0; + long v = strtol(value, &end, 10); + if (errno != 0 || *end != '\0' || v < 0 || v > INT_MAX) + return -1; + *out = (int)v; + return 0; +} + +int config_sockopts_parse(const char* spec, SockOptEntry** out, int* out_count) { + if (!spec || *spec == '\0' || !out || !out_count) + return -1; + char* copy = str_dup(spec); + if (!copy) + return -1; + + int count = 0; + int capacity = 0; + SockOptEntry* entries = NULL; + char* saveptr = NULL; + bool ok = true; + for (const char* token = strtok_r(copy, ",", &saveptr); token != NULL; + token = strtok_r(NULL, ",", &saveptr)) { + if (*token == '\0') { + ok = false; /* empty entry: a stray/trailing comma */ + break; + } + char* eq = strchr(token, '='); + if (eq) + *eq = '\0'; + int id = sockopt_id_from_name(token); + if (id < 0) { + ok = false; /* unknown option name */ + break; + } + int val; + /* rsync's --sockopts are OPT=VAL; a value is required for every option, so + * a bare option name (no '=') is rejected rather than coerced. */ + if (eq == NULL || eq[1] == '\0') { + ok = false; /* missing '=' or missing value */ + break; + } + if (sockopt_parse_value((SockOptId)id, eq + 1, &val) != 0) { + ok = false; /* bad value for an allowed option */ + break; + } + if (count == capacity) { + int new_cap = capacity == 0 ? 4 : capacity * 2; + SockOptEntry* grown = realloc(entries, (size_t)new_cap * sizeof(SockOptEntry)); + if (!grown) { + ok = false; + break; + } + entries = grown; + capacity = new_cap; + } + entries[count].id = (SockOptId)id; + entries[count].value = val; + count++; + } + free(copy); + if (!ok) { + free(entries); + return -1; + } + *out = entries; + *out_count = count; + return 0; +} + +bool config_is_remote_dest(const char* s) { if (s == NULL) return false; const char* colon = strchr(s, ':'); @@ -107,8 +503,124 @@ bool is_remote_dest(const char* s) { return true; } +/* Daemon destination detection: rsync's host::module[/path] marker is a "::" + * immediately after the host part (the first ':' is immediately followed by a + * second ':'), with no '/' before it. A single ':' (host:path) stays the SSH + * form even when the path itself later contains colons, and a "[::1]"-style + * bracketed IPv6 literal is not recognized as a daemon destination this wave + * (its first "::" is inside the brackets). */ +bool config_is_daemon_dest(const char* s) { + if (s == NULL) + return false; + const char* colon = strchr(s, ':'); + if (colon == NULL || colon == s || colon[1] != ':') + return false; + for (const char* p = s; p < colon; p++) { + if (*p == '/') + return false; + } + return true; +} + +/* Log an escaped message with an 8-bit-safe output policy and return -1 (the + * caller-visible parse failure code). */ +static int daemon_dest_parse_error(const char* message, const char* detail) { + char* escaped = output_escape(detail ? detail : "", false); + log_message(LOG_LEVEL_ERROR, "%s: %s", message, escaped ? escaped : ""); + free(escaped); + return -1; +} + +int config_parse_daemon_dest(Config* config) { + if (!config || !config->receive_root_directory) + return 0; + const char* dest = config->receive_root_directory; + if (!config_is_daemon_dest(dest)) + return 0; + + const char* colon = strchr(dest, ':'); + /* user@host::module names a daemon auth user. FastSync takes the username + * from the --password-file (its first user:password line) so there is a + * single source of truth; an @user that could contradict it is rejected + * with a pointer to the supported form. */ + if (memchr(dest, '@', (size_t)(colon - dest)) != NULL) + return daemon_dest_parse_error( + "daemon destination user@host::module is not supported: supply the username with " + "--password-file (first line: user:password)", + dest); + const char* host_start = dest; + + const char* module_and_path = colon + 2; + if (*module_and_path == '\0') + return daemon_dest_parse_error("daemon destination is missing its module name", dest); + const char* slash = strchr(module_and_path, '/'); + size_t module_len = slash ? (size_t)(slash - module_and_path) : strlen(module_and_path); + char* module = malloc(module_len + 1); + if (!module) + return daemon_dest_parse_error("out of memory parsing daemon destination", dest); + memcpy(module, module_and_path, module_len); + module[module_len] = '\0'; + if (!daemon_module_name_valid(module)) { + free(module); + return daemon_dest_parse_error( + "invalid daemon module name (must be 1-200 chars of [A-Za-z0-9._-])", dest); + } + + const char* path = slash ? slash + 1 : ""; + while (*path == '/') + path++; /* normalize "mod//a" to "mod/a"; keeps path module-relative */ + if (has_path_traversal(path)) { + free(module); + return daemon_dest_parse_error("daemon destination path must not contain '..'", dest); + } + + size_t host_len = (size_t)(colon - host_start); + char* host = malloc(host_len + 1); + if (!host) { + free(module); + return daemon_dest_parse_error("out of memory parsing daemon destination", dest); + } + memcpy(host, host_start, host_len); + host[host_len] = '\0'; + if (*host == '\0') { + free(host); + free(module); + return daemon_dest_parse_error("daemon destination has no host", dest); + } + + char* path_dup = str_dup(path); + if (!path_dup) { + free(host); + free(module); + return daemon_dest_parse_error("out of memory parsing daemon destination", dest); + } + + free(config->server_host); + config->server_host = host; + free(config->module); + config->module = module; + free(config->receive_root_directory); + config->receive_root_directory = path_dup; + config->transport = TRANSPORT_TCP; + return 1; +} + +int config_parse_transport_dest(Config* config) { + if (!config || !config->receive_root_directory) + return 0; + /* Daemon (host::module[/path]) first: the single-colon SSH parser would + * otherwise mis-split the double colon. Returns 1 (parsed as daemon), 0 + * (not daemon syntax -> try SSH below), or -1 (invalid daemon destination, + * already logged). */ + int daemon_ret = config_parse_daemon_dest(config); + if (daemon_ret != 0) + return daemon_ret; + config_parse_ssh_dest(config); + return 0; +} + void config_parse_ssh_dest(Config* config) { - if (!is_remote_dest(config->receive_root_directory)) + if (!config_is_remote_dest(config->receive_root_directory)) return; config->transport = TRANSPORT_SSH; config->ssh_destination = str_dup(config->receive_root_directory); @@ -118,11 +630,38 @@ void config_parse_ssh_dest(Config* config) { config->receive_root_directory = path; } +void config_burn_auth(Config* config) { + if (!config) + return; + if (config->auth_password) { + credentials_burn(config->auth_password, strlen(config->auth_password)); + free(config->auth_password); + config->auth_password = NULL; + } + if (config->auth_user) { + free(config->auth_user); + config->auth_user = NULL; + } +} + void config_delete(Config* config) { + if (config == NULL) + return; + if (config->log_file) { + fclose(config->log_file); + config->log_file = NULL; + } free(config->version); free(config->send_directory); free(config->receive_root_directory); free(config->ssh_destination); + free(config->module); + config_burn_auth(config); + free(config->password_file); + free(config->iconv_spec); + free(config->write_batch); + free(config->only_write_batch); + free(config->read_batch); free(config->fastsync_server_path); for (int i = 0; i < config->exclude_count; i++) free(config->exclude_patterns[i]); @@ -136,97 +675,695 @@ void config_delete(Config* config) { free(config->backup_dir); free(config->server_host); free(config->out_format); + free(config->log_file_format); free(config->files_from); + file_list_destroy((FileListSet*)config->files_from_set); free(config->rsh_command); - free(config->rsync_path); free(config->temp_dir); - free(config->compare_dest); - free(config->copy_dest); - free(config->link_dest); + if (config->remote_options) { + for (int i = 0; i < config->remote_option_count; i++) + free(config->remote_options[i]); + free(config->remote_options); + } + config->remote_options = NULL; + config->remote_option_count = 0; + for (int i = 0; i < config->basis_count; i++) { + free(config->basis_dirs[i].path); + config->basis_dirs[i].path = NULL; + } + free(config->basis_dirs); + config->basis_dirs = NULL; + config->basis_count = 0; + free(config->partial_dir); + free(config->suffix); + free(config->address); + free(config->bind_address); + free(config->sockopts); + free(config->daemon_config); + free(config->compress_choice); + free(config->chmod_spec); + if (config->skip_compress_suffixes) { + for (int i = 0; i < config->skip_compress_count; i++) + free(config->skip_compress_suffixes[i]); + free(config->skip_compress_suffixes); + } + free(config->usermap); + config->usermap = NULL; + config->usermap_count = 0; + free(config->groupmap); + config->groupmap = NULL; + config->groupmap_count = 0; if (config->filters) { array_list_delete(config->filters); } + /* A --delay-updates staging tree is transient receiver state: remove any + leftovers on every exit path (success already emptied it). */ + if (config->delay_context) + delay_updates_cleanup(config->delay_context); + delay_updates_context_destroy(config->delay_context); + config->delay_context = NULL; free(config); } +/* Each helper is deliberately ordered to match the wire format. Keep the + * helper call order in config_send and config_receive unchanged when adding + * fields. */ +static bool send_core_fields(int fd, const Config* c) { + if (!send_str(fd, c->version) || !send_int(fd, c->eight_bit_output)) + return false; + protocol_set_8_bit_output(c->eight_bit_output); + if (!send_n_data(fd, &c->max_alloc, sizeof(c->max_alloc))) + return false; + return send_str(fd, c->send_directory) && send_str(fd, c->receive_root_directory) && + send_int(fd, c->save_to_disk) && send_int(fd, c->use_multithreading) && + send_int(fd, c->use_chunk_serialization) && send_int(fd, c->use_compression) && + send_int(fd, c->use_metadata) && send_int(fd, c->use_executability) && + send_int(fd, c->compression_level) && + send_n_data(fd, &c->chunk_size, sizeof(c->chunk_size)) && send_int(fd, c->use_sendfile); +} + +static bool send_delta_fields(int fd, const Config* c) { + return send_int(fd, c->use_delete) && send_int(fd, c->use_incremental) && + send_int(fd, c->size_only) && send_int(fd, c->ignore_times) && + send_int(fd, c->use_delta && !c->whole_file) && + send_n_data(fd, &c->delta_block_size, sizeof(c->delta_block_size)) && + send_n_data(fd, &c->delta_max_file_size, sizeof(unsigned long long)); +} + +static bool send_file_options(int fd, const Config* c) { + /* Device/special preservation flags cross the wire so the receiver knows a + * special/device entry must be recreated. Trailing fields; protocol 2.13.0. */ + return send_int(fd, c->backup) && send_str(fd, c->backup_dir ? c->backup_dir : "") && + send_int(fd, c->remove_source_files) && send_int(fd, c->follow_symlinks) && + send_int(fd, c->copy_links) && send_int(fd, c->safe_links) && + send_int(fd, c->copy_unsafe_links) && send_int(fd, c->preserve_hard_links) && + send_int(fd, c->preserve_acls) && send_int(fd, c->preserve_xattrs) && + send_int(fd, c->preserve_devices) && send_int(fd, c->preserve_sparse) && + send_int(fd, c->preserve_specials) && send_int(fd, c->copy_devices) && + send_int(fd, c->write_devices); +} + +static bool send_selection_options(int fd, const Config* c) { + return send_int(fd, c->ignore_existing) && send_int(fd, c->existing) && send_int(fd, c->update) && + send_int(fd, c->inplace) && send_int(fd, c->delay_updates) && send_int(fd, c->append) && + send_int(fd, c->use_fsync) && send_int(fd, c->append_verify) && + send_int(fd, c->delete_excluded) && send_int(fd, c->force_delete) && + send_int(fd, c->delete_missing_args) && send_int(fd, c->delete_after) && + send_int(fd, c->preallocate) && send_n_data(fd, &c->max_delete, sizeof(c->max_delete)) && + send_int(fd, c->relative) && send_int(fd, c->prune_empty_dirs) && + send_int(fd, c->mkpath) && send_int(fd, c->delete_during) && send_int(fd, c->delete_delay); +} + +static bool send_skip_compress_options(int fd, const Config* c) { + if (!send_int(fd, c->skip_compress_set) || !send_int(fd, c->skip_compress_count)) + return false; + for (int i = 0; i < c->skip_compress_count; i++) { + if (!send_str(fd, c->skip_compress_suffixes[i])) + return false; + } + return true; +} + +static bool send_resume_options(int fd, const Config* c) { + return send_str(fd, c->temp_dir ? c->temp_dir : "") && send_int(fd, c->partial) && + send_str(fd, c->partial_dir ? c->partial_dir : "") && + send_str(fd, c->suffix ? c->suffix : "") && send_int(fd, c->delete_before) && + send_int(fd, c->checksum) && send_int(fd, c->modify_window) && + send_str(fd, c->compress_choice ? c->compress_choice : "") && + send_str(fd, c->chmod_spec ? c->chmod_spec : "") && send_skip_compress_options(fd, c); +} + +static bool send_basis_options(int fd, const Config* c) { + if (!send_int(fd, c->basis_count)) + return false; + for (int i = 0; i < c->basis_count; i++) { + if (!send_int(fd, (int)c->basis_dirs[i].type) || + !send_str(fd, c->basis_dirs[i].path ? c->basis_dirs[i].path : "")) + return false; + } + return true; +} + +/* -y/--fuzzy (receiver-side similar-file basis selection). Trailing field on + * the config frame; protocol 2.9.0. */ +static bool send_fuzzy_option(int fd, const Config* c) { + return send_int(fd, c->fuzzy); +} + +/* --checksum-choice/--cc + --checksum-seed. The algorithm id and seed travel + * with the config so the receiver hashes the on-disk old file with the same + * parameters the sender used for its digest (see checksum.h). Trailing fields + * on the config frame; protocol 2.10.0. */ +static bool send_checksum_options(int fd, const Config* c) { + return send_int(fd, c->checksum_algo) && + send_n_data(fd, &c->checksum_seed, sizeof(c->checksum_seed)); +} + +static bool receive_core_fields(int fd, Config* c) { + int value; + if (!receive_wire_bool(fd, &c->eight_bit_output)) + return false; + protocol_set_8_bit_output(c->eight_bit_output); + if (!receive_n_data(fd, &c->max_alloc, sizeof(c->max_alloc)) || c->max_alloc == 0) + return false; + if (c->max_alloc > MAX_SERVER_ALLOC) + c->max_alloc = MAX_SERVER_ALLOC; + protocol_session_set_max_alloc(NULL, c->max_alloc); + c->send_directory = receive_str(fd); + c->receive_root_directory = receive_str(fd); + if (!c->send_directory || !c->receive_root_directory) + return false; + if (!receive_wire_bool(fd, &c->save_to_disk) || !receive_wire_bool(fd, &c->use_multithreading) || + !receive_wire_bool(fd, &c->use_chunk_serialization) || + !receive_wire_bool(fd, &c->use_compression) || !receive_wire_bool(fd, &c->use_metadata) || + !receive_wire_bool(fd, &c->use_executability)) + return false; + if (!receive_int(fd, &value)) + return false; + c->compression_level = value; + if (!receive_n_data(fd, &c->chunk_size, sizeof(c->chunk_size))) + return false; + if (!receive_wire_bool(fd, &c->use_sendfile)) + return false; + return true; +} + +static bool receive_delta_fields(int fd, Config* c) { + if (!receive_wire_bool(fd, &c->use_delete)) + return false; + if (!receive_wire_bool(fd, &c->use_incremental)) + return false; + if (!receive_wire_bool(fd, &c->size_only)) + return false; + if (!receive_wire_bool(fd, &c->ignore_times)) + return false; + if (!receive_wire_bool(fd, &c->use_delta)) + return false; + return receive_n_data(fd, &c->delta_block_size, sizeof(c->delta_block_size)) && + receive_n_data(fd, &c->delta_max_file_size, sizeof(unsigned long long)); +} + +static bool receive_file_options(int fd, Config* c) { + if (!receive_wire_bool(fd, &c->backup)) + return false; + char* backup_dir = receive_str(fd); + if (!backup_dir) + return false; + if (*backup_dir != '\0') { + c->backup_dir = backup_dir; + } else { + /* The sender serializes an unset (NULL) string as "", so canonicalize the + empty wire value back to NULL to preserve NULL-vs-empty semantics. */ + free(backup_dir); + } + if (!receive_wire_bool(fd, &c->remove_source_files)) + return false; + bool* flags[] = {&c->follow_symlinks, &c->copy_links, &c->safe_links, + &c->copy_unsafe_links, &c->preserve_hard_links, &c->preserve_acls, + &c->preserve_xattrs, &c->preserve_devices, &c->preserve_sparse, + &c->preserve_specials, &c->copy_devices, &c->write_devices}; + for (size_t i = 0; i < sizeof(flags) / sizeof(flags[0]); i++) { + if (!receive_wire_bool(fd, flags[i])) + return false; + } + return true; +} + +static bool receive_selection_options(int fd, Config* c) { + bool* flags[] = {&c->ignore_existing, + &c->existing, + &c->update, + &c->inplace, + &c->delay_updates, + &c->append, + &c->use_fsync, + &c->append_verify, + &c->delete_excluded, + &c->force_delete, + &c->delete_missing_args, + &c->delete_after, + &c->preallocate}; + for (size_t i = 0; i < sizeof(flags) / sizeof(flags[0]); i++) { + if (!receive_wire_bool(fd, flags[i])) + return false; + } + if (!receive_n_data(fd, &c->max_delete, sizeof(c->max_delete))) + return false; + if (!receive_wire_bool(fd, &c->relative)) + return false; + if (!receive_wire_bool(fd, &c->prune_empty_dirs)) + return false; + if (!receive_wire_bool(fd, &c->mkpath)) + return false; + if (!receive_wire_bool(fd, &c->delete_during)) + return false; + return receive_wire_bool(fd, &c->delete_delay); +} + +static bool receive_resume_options(int fd, Config* c) { + char* temp_dir = receive_str(fd); + if (!temp_dir) + return false; + if (*temp_dir != '\0') { + c->temp_dir = temp_dir; + } else { + free(temp_dir); + } + if (!receive_wire_bool(fd, &c->partial)) + return false; + /* These options have NULL client defaults, so the sender transmits an empty + string for "unset". Canonicalize the empty wire value back to NULL so + receivers observe exactly what the client configured (plain --backup, for + example, must not look like --backup-dir ""). */ + char* partial_dir = receive_str(fd); + if (!partial_dir) + return false; + if (*partial_dir != '\0') { + c->partial_dir = partial_dir; + } else { + free(partial_dir); + } + char* suffix = receive_str(fd); + if (!suffix) + return false; + if (*suffix != '\0') { + c->suffix = suffix; + } else { + free(suffix); + } + if (!receive_wire_bool(fd, &c->delete_before)) + return false; + if (!receive_wire_bool(fd, &c->checksum)) + return false; + if (!receive_n_data(fd, &c->modify_window, sizeof(c->modify_window))) + return false; + c->compress_choice = receive_str(fd); + if (!c->compress_choice) + return false; + c->chmod_spec = receive_str(fd); + if (!c->chmod_spec || !receive_wire_bool(fd, &c->skip_compress_set) || + !receive_int(fd, &c->skip_compress_count) || c->skip_compress_count < 0 || + c->skip_compress_count > 10000) + return false; + if (c->skip_compress_count > 0) { + c->skip_compress_suffixes = calloc((size_t)c->skip_compress_count, sizeof(char*)); + if (!c->skip_compress_suffixes) + return false; + for (int i = 0; i < c->skip_compress_count; i++) { + c->skip_compress_suffixes[i] = receive_str(fd); + if (!c->skip_compress_suffixes[i]) + return false; + } + } + return true; +} + +static bool receive_basis_options(int fd, Config* c) { + int count; + if (!receive_int(fd, &count)) + return false; + if (count < 0 || count > MAX_BASIS_DIRS) + return false; + for (int i = 0; i < count; i++) { + int type; + if (!receive_int(fd, &type) || type <= BASIS_DEST_NONE || type > BASIS_DEST_LINK) + return false; + char* path = receive_str(fd); + if (!path) + return false; + /* config_basis_append validates and canonicalizes the path; a rejected + path (absolute / traversal / empty) drops the whole connection. */ + bool ok = config_basis_append(c, (BasisDestType)type, path) == 0; + free(path); + if (!ok) + return false; + } + return true; +} + +static bool receive_fuzzy_option(int fd, Config* c) { + return receive_wire_bool(fd, &c->fuzzy); +} + +static bool receive_checksum_options(int fd, Config* c) { + int algo; + if (!receive_int(fd, &algo) || !checksum_algo_valid(algo)) + return false; + c->checksum_algo = algo; + return receive_n_data(fd, &c->checksum_seed, sizeof(c->checksum_seed)); +} + +/* --numeric-ids / --usermap / --groupmap / --chown (identity mapping). The + * receiver needs these to apply the ownership the client requested, so they + * cross the config frame. Trailing fields; protocol 2.11.0. */ +static bool send_identity_map(int fd, const IdentityMap* map, int count) { + if (!send_int(fd, count)) + return false; + for (int i = 0; i < count; i++) { + if (!send_int(fd, map[i].from) || !send_int(fd, map[i].to)) + return false; + } + return true; +} + +static bool send_identity_options(int fd, const Config* c) { + return send_int(fd, c->numeric_ids) && send_int(fd, c->chown_uid_set) && + send_int(fd, c->chown_uid) && send_int(fd, c->chown_gid_set) && + send_int(fd, c->chown_gid) && send_identity_map(fd, c->usermap, c->usermap_count) && + send_identity_map(fd, c->groupmap, c->groupmap_count); +} + +static bool receive_identity_map(int fd, int* pcount, IdentityMap** pmap) { + int count; + if (!receive_int(fd, &count) || count < 0 || count > MAX_IDENTITY_MAP) + return false; + if (count > 0) { + IdentityMap* map = calloc((size_t)count, sizeof(IdentityMap)); + if (!map) + return false; + for (int i = 0; i < count; i++) { + if (!receive_int(fd, &map[i].from) || !receive_int(fd, &map[i].to)) { + free(map); + return false; + } + } + *pmap = map; + } + *pcount = count; + return true; +} + +static bool receive_identity_options(int fd, Config* c) { + int numeric_ids; + if (!receive_int(fd, &numeric_ids) || !valid_wire_bool(numeric_ids)) + return false; + c->numeric_ids = numeric_ids != 0; + if (!receive_wire_bool(fd, &c->chown_uid_set) || !receive_int(fd, &c->chown_uid) || + !receive_wire_bool(fd, &c->chown_gid_set) || !receive_int(fd, &c->chown_gid)) + return false; + if (c->chown_uid < IDENTITY_MATCH_ANY || c->chown_gid < IDENTITY_MATCH_ANY) + return false; + return receive_identity_map(fd, &c->usermap_count, &c->usermap) && + receive_identity_map(fd, &c->groupmap_count, &c->groupmap); +} + +/* -U/--atimes, -N/--crtimes (affect both sender capture and receiver apply) + * and -O/--omit-dir-times, -J/--omit-link-times (receiver-side prefs) all cross + * the wire so the receiver knows what to apply / suppress. --open-noatime is + * client-only (it only governs the sender's source reads) and is never + * serialized. Trailing fields; protocol 2.12.0. */ +static bool send_metadata_times_options(int fd, const Config* c) { + return send_int(fd, c->preserve_atimes) && send_int(fd, c->preserve_crtimes) && + send_int(fd, c->omit_dir_times) && send_int(fd, c->omit_link_times); +} + +static bool receive_metadata_times_options(int fd, Config* c) { + return receive_wire_bool(fd, &c->preserve_atimes) && + receive_wire_bool(fd, &c->preserve_crtimes) && receive_wire_bool(fd, &c->omit_dir_times) && + receive_wire_bool(fd, &c->omit_link_times); +} + +/* Phase 4 symlink-trust: --munge-links and -K/--keep-dirlinks. Both CROSS the + * wire (the receiver unmunges symlink targets and, with -K, follows an in-root + * destination symlink-to-directory). -k/--copy-dirlinks is sender-only and is + * never serialized. Trailing fields; protocol 2.13.0. */ +static bool send_symlink_trust_options(int fd, const Config* c) { + return send_int(fd, c->munge_links) && send_int(fd, c->keep_dirlinks); +} + +static bool receive_symlink_trust_options(int fd, Config* c) { + return receive_wire_bool(fd, &c->munge_links) && receive_wire_bool(fd, &c->keep_dirlinks); +} + +/* -X/--xattrs, -A/--acls, --fake-super (Phase-4). The receiver learns + * preserve_xattrs/preserve_acls from the earlier file-options block and + * recomputes the derived use_xattrs there; only --fake-super (receiver-side + * behavior) needs an extra wire bit. Trailing field; protocol 2.13.0. */ +static bool send_phase4_xattr_options(int fd, const Config* c) { + return send_int(fd, c->fake_super); +} + +static bool receive_phase4_xattr_options(int fd, Config* c) { + if (!receive_wire_bool(fd, &c->fake_super)) + return false; + c->use_xattrs = c->preserve_acls || c->preserve_xattrs; + return true; +} + +/* Daemon module selection (Wave A, protocol 2.15.0). Trailing string on the + * config frame, sent after the Phase-4 xattr block and before the ack. The + * client composes it from a host::module/path destination; an unset module is + * serialized as "" and canonicalized back to NULL on receive so the two never + * look different to a peer. */ +static bool send_daemon_module(int fd, const Config* c) { + return send_str(fd, c->module ? c->module : ""); +} + +static bool receive_daemon_module(int fd, Config* c) { + char* module = receive_str(fd); + if (!module) + return false; + /* Guard against a hostile client flooding the log with an over-long module + * name: only an empty string (module-less) or a valid module name + * (bounded by DAEMON_MAX_MODULE_NAME) is accepted. This is an input + * guard, not a wire-format change. */ + if (*module != '\0' && !daemon_module_name_valid(module)) { + log_message(LOG_LEVEL_WARNING, "Daemon client sent an invalid or over-long module name"); + free(module); + send_status(fd, STATUS_ERROR); + return false; + } + if (*module != '\0') { + c->module = module; + } else { + free(module); + } + return true; +} + +/* Daemon password credentials (A7 remediation, protocol 2.19.0). A single + * presence int is followed, when set, by ONLY the username; the password is + * never serialized. The daemon answers an auth-required module with the SCRAM + * challenge (see the auth exchange below). */ +static bool send_daemon_auth(int fd, const Config* c) { + bool present = c->auth_user != NULL && c->auth_user[0] != '\0'; + if (!send_int(fd, present ? 1 : 0)) + return false; + if (!present) + return true; + /* Redacted send: the username must never reach a --verbose debug log. */ + return send_str_redacted(fd, c->auth_user); +} + +static bool receive_daemon_auth(int fd, Config* c) { + int present; + if (!receive_int(fd, &present) || !valid_wire_bool(present)) + return false; + if (!present) + return true; + /* Redacted receive: never log the incoming username body. */ + char* user = receive_str_redacted(fd); + if (!user) + return false; + if (!credentials_username_valid(user)) { + free(user); + log_message(LOG_LEVEL_WARNING, "Daemon client sent malformed auth credentials"); + return false; + } + c->auth_user = user; + return true; +} + +/* Client half of the SCRAM challenge/response (A7 remediation). Called by + * config_send after the config frame is written and the server answered + * STATUS_AUTH_CHALLENGE. The plaintext password lives only in + * config->auth_password and every derived buffer is wiped on the way out. */ +static bool client_auth_exchange(int fd, const Config* c) { + if (!c->auth_user || !c->auth_password) + return false; + int iters = 0; + if (!receive_int(fd, &iters)) + return false; + if (iters < (int)CREDENTIAL_MIN_ITERS || iters > (int)CREDENTIAL_MAX_ITERS) { + log_message(LOG_LEVEL_ERROR, "Daemon sent an out-of-range auth iteration count"); + return false; + } + char* salt_b64 = receive_str(fd); + char* snonce_b64 = receive_str(fd); + uint8_t salt[CREDENTIAL_SALT_LEN]; + uint8_t snonce[CREDENTIAL_NONCE_LEN]; + uint8_t cnonce[CREDENTIAL_NONCE_LEN]; + size_t salt_len = 0; + size_t snonce_len = 0; + bool ok = salt_b64 && snonce_b64 && + credentials_b64_decode(salt_b64, salt, sizeof(salt), &salt_len) && + salt_len == CREDENTIAL_SALT_LEN && + credentials_b64_decode(snonce_b64, snonce, sizeof(snonce), &snonce_len) && + snonce_len == CREDENTIAL_NONCE_LEN && credentials_random_bytes(cnonce, sizeof(cnonce)); + credentials_burn(salt_b64, salt_b64 ? strlen(salt_b64) : 0); + credentials_burn(snonce_b64, snonce_b64 ? strlen(snonce_b64) : 0); + free(salt_b64); + free(snonce_b64); + if (!ok) { + log_message(LOG_LEVEL_ERROR, "Daemon sent a malformed auth challenge"); + return false; + } + uint8_t client_key[CREDENTIAL_KEY_LEN]; + uint8_t stored_key[CREDENTIAL_KEY_LEN]; + uint8_t server_key[CREDENTIAL_KEY_LEN]; + uint8_t auth_msg[CREDENTIAL_AUTH_MESSAGE_MAX]; + size_t msg_len = 0; + uint8_t proof[CREDENTIAL_KEY_LEN]; + uint8_t expected_sig[CREDENTIAL_KEY_LEN]; + ok = credentials_compute_keys(c->auth_password, salt, (uint32_t)iters, client_key, stored_key, + server_key) && + credentials_build_auth_message(c->auth_user, snonce, cnonce, auth_msg, sizeof(auth_msg), + &msg_len) && + credentials_client_proof(client_key, stored_key, server_key, auth_msg, msg_len, proof, + expected_sig); + char cnonce_b64[45]; + char proof_b64[45]; + if (ok) + ok = credentials_b64_encode(cnonce, sizeof(cnonce), cnonce_b64, sizeof(cnonce_b64)) && + credentials_b64_encode(proof, sizeof(proof), proof_b64, sizeof(proof_b64)); + if (!ok) { + log_message(LOG_LEVEL_ERROR, "Failed to compute the daemon auth response"); + } else { + ok = send_status(fd, STATUS_AUTH_RESPONSE) && send_str_redacted(fd, cnonce_b64) && + send_str_redacted(fd, proof_b64); + } + if (ok) { + Status status = STATUS_ERROR; + char* sig_b64 = NULL; + uint8_t sig[CREDENTIAL_KEY_LEN]; + size_t sig_len = 0; + ok = receive_status(fd, &status) && status == STATUS_AUTH_OK && + (sig_b64 = receive_str_redacted(fd)) != NULL && + credentials_b64_decode(sig_b64, sig, sizeof(sig), &sig_len) && + sig_len == CREDENTIAL_KEY_LEN && + credentials_secure_equal((const char*)sig, (const char*)expected_sig, CREDENTIAL_KEY_LEN); + if (!ok) + log_message(LOG_LEVEL_ERROR, "Daemon authentication failed"); + credentials_burn(sig_b64, sig_b64 ? strlen(sig_b64) : 0); + credentials_burn((char*)sig, sizeof(sig)); + free(sig_b64); + } + credentials_burn((char*)client_key, sizeof(client_key)); + credentials_burn((char*)stored_key, sizeof(stored_key)); + credentials_burn((char*)server_key, sizeof(server_key)); + credentials_burn((char*)auth_msg, sizeof(auth_msg)); + credentials_burn((char*)proof, sizeof(proof)); + credentials_burn((char*)expected_sig, sizeof(expected_sig)); + credentials_burn((char*)salt, sizeof(salt)); + credentials_burn((char*)snonce, sizeof(snonce)); + credentials_burn((char*)cnonce, sizeof(cnonce)); + credentials_burn(cnonce_b64, sizeof(cnonce_b64)); + credentials_burn(proof_b64, sizeof(proof_b64)); + return ok; +} + +/* --iconv CONVERT_SPEC (protocol 2.16.0). Trailing string on the config frame, + * sent after the Wave A/B daemon-auth block and before the ack, so the + * receiver knows the wire charset before the first file name arrives. The full + * spec travels (LOCAL,REMOTE) and each end derives its own LOCAL and the wire + * (REMOTE) charset symmetrically; an unset spec is serialized as "" and + * canonicalized back to NULL on receive. */ +static bool send_iconv_spec(int fd, const Config* c) { + return send_str(fd, c->iconv_spec ? c->iconv_spec : ""); +} + +static bool receive_iconv_spec(int fd, Config* c) { + char* spec = receive_str(fd); + if (!spec) + return false; + if (*spec == '\0') { + free(spec); + c->iconv_spec = NULL; + return true; + } + c->iconv_spec = spec; + return true; +} + +/* --super / --no-super privilege policy (P7 Wave E, protocol 2.18.0). One + * trailing int on the config frame, sent after the --iconv spec and before the + * STATUS_OK ack, so the receiver knows whether it may attempt super-user + * activities (ownership application, char/block device-node creation) that are + * already confined below the authorized receive root. The received value is + * validated to the SUPER_MODE_AUTO..SUPER_MODE_OFF range (also re-checked by + * validate_received_config). */ +static bool send_privilege_options(int fd, const Config* c) { + return send_int(fd, c->super_mode); +} + +static bool receive_privilege_options(int fd, Config* c) { + int mode; + if (!receive_int(fd, &mode) || mode < SUPER_MODE_AUTO || mode > SUPER_MODE_OFF) + return false; + c->super_mode = mode; + return true; +} + +/* --copy-as=USER[:GROUP] (P7 Wave E, protocol 2.18.0). Trailing block on the + * config frame, sent after the --super int and before the ack: a presence int, + * then (when set) the target uid and gid as int32. The receiver forces the + * ownership of every entry it writes to these ids through the confined + * fd-relative identity path and requires privilege; both ids are validated + * `>= 0` on receive so a hostile peer cannot smuggle a negative (sentinel) + * value into the ownership path. */ +static bool send_copy_as_options(int fd, const Config* c) { + if (!send_int(fd, c->copy_as_set ? 1 : 0)) + return false; + if (!c->copy_as_set) + return true; + return send_int(fd, c->copy_as_uid) && send_int(fd, c->copy_as_gid); +} + +static bool receive_copy_as_options(int fd, Config* c) { + int present; + if (!receive_int(fd, &present) || !valid_wire_bool(present)) + return false; + if (!present) { + c->copy_as_set = false; + return true; + } + int uid, gid; + if (!receive_int(fd, &uid) || !receive_int(fd, &gid) || uid < 0 || gid < 0) + return false; + c->copy_as_set = true; + c->copy_as_uid = uid; + c->copy_as_gid = gid; + return true; +} + bool config_send(int file_descriptor, const Config* config) { - if (!send_str(file_descriptor, config->version)) - return false; - if (!send_str(file_descriptor, config->send_directory)) - return false; - if (!send_str(file_descriptor, config->receive_root_directory)) - return false; - if (!send_int(file_descriptor, config->save_to_disk)) - return false; - if (!send_int(file_descriptor, config->use_multithreading)) - return false; - if (!send_int(file_descriptor, config->use_chunk_serialization)) - return false; - if (!send_int(file_descriptor, config->use_compression)) - return false; - if (!send_int(file_descriptor, config->use_metadata)) - return false; - if (!send_int(file_descriptor, config->compression_level)) - return false; - if (!send_n_data(file_descriptor, &config->chunk_size, sizeof(config->chunk_size))) - return false; - if (!send_int(file_descriptor, config->use_sendfile)) - return false; - if (!send_int(file_descriptor, config->use_delete)) - return false; - if (!send_int(file_descriptor, config->use_incremental)) - return false; - if (!send_int(file_descriptor, config->use_delta)) - return false; - if (!send_int(file_descriptor, (int)config->delta_block_size)) - return false; - if (!send_n_data(file_descriptor, &config->delta_max_file_size, sizeof(unsigned long long))) - return false; - if (!send_int(file_descriptor, config->backup)) - return false; - if (!send_str(file_descriptor, config->backup_dir ? config->backup_dir : "")) - return false; - if (!send_int(file_descriptor, config->follow_symlinks)) - return false; - if (!send_int(file_descriptor, config->copy_links)) - return false; - if (!send_int(file_descriptor, config->safe_links)) - return false; - if (!send_int(file_descriptor, config->copy_unsafe_links)) - return false; - if (!send_int(file_descriptor, config->preserve_hard_links)) - return false; - if (!send_int(file_descriptor, config->preserve_acls)) - return false; - if (!send_int(file_descriptor, config->preserve_xattrs)) - return false; - if (!send_int(file_descriptor, config->preserve_devices)) - return false; - if (!send_int(file_descriptor, config->preserve_sparse)) - return false; - if (!send_int(file_descriptor, config->update)) - return false; - if (!send_int(file_descriptor, config->inplace)) - return false; - if (!send_int(file_descriptor, config->append)) - return false; - if (!send_int(file_descriptor, config->append_verify)) - return false; - if (!send_int(file_descriptor, config->delete_excluded)) - return false; - if (!send_int(file_descriptor, config->delete_after)) - return false; - if (!send_n_data(file_descriptor, &config->max_delete, sizeof(config->max_delete))) - return false; - if (!send_int(file_descriptor, config->relative)) - return false; - if (!send_int(file_descriptor, config->prune_empty_dirs)) - return false; - if (!send_str(file_descriptor, config->temp_dir ? config->temp_dir : "")) + protocol_session_set_max_alloc(NULL, config->max_alloc); + if (!send_core_fields(file_descriptor, config) || !send_delta_fields(file_descriptor, config) || + !send_file_options(file_descriptor, config) || + !send_selection_options(file_descriptor, config) || + !send_resume_options(file_descriptor, config) || + !send_basis_options(file_descriptor, config) || !send_fuzzy_option(file_descriptor, config) || + !send_checksum_options(file_descriptor, config) || + !send_identity_options(file_descriptor, config) || + !send_metadata_times_options(file_descriptor, config) || + !send_symlink_trust_options(file_descriptor, config) || + !send_phase4_xattr_options(file_descriptor, config) || + !send_daemon_module(file_descriptor, config) || !send_daemon_auth(file_descriptor, config) || + !send_iconv_spec(file_descriptor, config) || + !send_privilege_options(file_descriptor, config) || + !send_copy_as_options(file_descriptor, config)) return false; Status status; if (!receive_status(file_descriptor, &status)) return false; + if (status == STATUS_AUTH_CHALLENGE) { + /* Daemon auth (protocol 2.19.0): run the SCRAM exchange, then wait for the + * ordinary STATUS_OK the server sends once authentication succeeded. */ + if (!client_auth_exchange(file_descriptor, config)) + return false; + if (!receive_status(file_descriptor, &status)) + return false; + } if (status != STATUS_OK) { log_message(LOG_LEVEL_ERROR, "Error transmitting config"); return false; @@ -234,206 +1371,80 @@ bool config_send(int file_descriptor, const Config* config) { return true; } -Config* config_receive(int file_descriptor) { - Config* config = (Config*)malloc(sizeof(Config)); - if (config == NULL) +Config* config_receive_with_validate(int file_descriptor, ConfigValidateFunc validate, + void* context) { + Config* config = config_create(); + if (!config) return NULL; - memset(config, 0, sizeof(*config)); + free(config->version); config->version = receive_str(file_descriptor); - if (!config->version) { - free(config); - return NULL; - } + if (!config->version) + goto error; if (strcmp(config->version, PROTOCOL_VERSION) != 0) { - fprintf(stderr, "Protocol version mismatch: client=%s, server=%s\n", config->version, - PROTOCOL_VERSION); - free(config->version); - free(config); + char* escaped_version = output_escape(config->version, false); + fprintf(stderr, "Protocol version mismatch: client=%s, server=%s\n", + escaped_version ? escaped_version : "", PROTOCOL_VERSION); + free(escaped_version); send_status(file_descriptor, STATUS_ERROR); - return NULL; + goto error; } - config->send_directory = receive_str(file_descriptor); - if (!config->send_directory) { - free(config->version); - free(config); - return NULL; + if (!receive_core_fields(file_descriptor, config) || + !receive_delta_fields(file_descriptor, config) || + !receive_file_options(file_descriptor, config) || + !receive_selection_options(file_descriptor, config) || + !receive_resume_options(file_descriptor, config) || + !receive_basis_options(file_descriptor, config) || + !receive_fuzzy_option(file_descriptor, config) || + !receive_checksum_options(file_descriptor, config) || + !receive_identity_options(file_descriptor, config) || + !receive_metadata_times_options(file_descriptor, config) || + !receive_symlink_trust_options(file_descriptor, config) || + !receive_phase4_xattr_options(file_descriptor, config) || + !receive_daemon_module(file_descriptor, config) || + !receive_daemon_auth(file_descriptor, config) || + !receive_iconv_spec(file_descriptor, config) || + !receive_privilege_options(file_descriptor, config) || + !receive_copy_as_options(file_descriptor, config)) + goto error; + if (config->compress_choice[0] != '\0' && strcmp(config->compress_choice, "zstd") != 0 && + strcmp(config->compress_choice, "none") != 0) { + char* escaped_choice = output_escape(config->compress_choice, config->eight_bit_output); + fprintf(stderr, "Unsupported compression choice: %s\n", + escaped_choice ? escaped_choice : ""); + free(escaped_choice); + send_status(file_descriptor, STATUS_ERROR); + goto error; } - config->receive_root_directory = receive_str(file_descriptor); - if (!config->receive_root_directory) { - free(config->version); - free(config->send_directory); - free(config); - return NULL; + if (!validate_received_config(config)) { + fprintf(stderr, "Invalid configuration received from client\n"); + send_status(file_descriptor, STATUS_ERROR); + goto error; + } + if (validate) { + const char* rejection = validate(config, context); + if (rejection != NULL) { + /* Daemon module gate (unknown module / read-only module / auth-required + * module): refuse BEFORE the STATUS_OK so the client aborts at the + * config handshake and no file data is ever exchanged. The auth + * handshake already sent STATUS_AUTH_FAILED when it failed, signalled by + * the CONFIG_VALIDATE_ALREADY_TERMINATED sentinel, so no second status is + * written. */ + if (rejection != CONFIG_VALIDATE_ALREADY_TERMINATED) { + fprintf(stderr, "%s\n", rejection); + send_status(file_descriptor, STATUS_ERROR); + } + goto error; + } } - int tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->save_to_disk = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->use_multithreading = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->use_chunk_serialization = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->use_compression = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->use_metadata = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->compression_level = tmp; - if (!receive_n_data(file_descriptor, &config->chunk_size, sizeof(config->chunk_size))) - goto error; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->use_sendfile = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->use_delete = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->use_incremental = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->use_delta = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->delta_block_size = (uint32_t)tmp; - if (!receive_n_data(file_descriptor, &config->delta_max_file_size, sizeof(unsigned long long))) - goto error; - config->show_progress = false; - config->dry_run = false; - config->ssh_port = 22; - config->transport = TRANSPORT_TCP; - config->ssh_destination = NULL; - config->fastsync_server_path = NULL; - config->exclude_patterns = NULL; - config->exclude_count = 0; - config->include_patterns = NULL; - config->include_count = 0; - config->max_size = 0; - config->min_size = 0; - config->use_tls = false; - config->tls_cert = NULL; - config->tls_key = NULL; - config->tls_ca = NULL; - config->timeout = 30; - config->contimeout = 10; - config->quiet = false; - config->stats = false; - config->max_depth = 0; - config->log_file = NULL; - config->queue_size = 100; - config->follow_symlinks = false; - config->copy_links = false; - config->safe_links = false; - config->copy_unsafe_links = false; - config->preserve_hard_links = false; - config->preserve_acls = false; - config->preserve_xattrs = false; - config->preserve_devices = false; - config->preserve_sparse = false; - config->itemize_changes = false; - config->out_format = NULL; - config->info_level = 0; - config->debug_level = 0; - config->list_only = false; - config->human_readable = false; - config->update = false; - config->inplace = false; - config->append = false; - config->append_verify = false; - config->delete_excluded = false; - config->delete_after = false; - config->max_delete = 0; - config->filters = NULL; - config->files_from = NULL; - config->cvs_exclude = false; - config->prune_empty_dirs = false; - config->relative = false; - config->rsh_command = NULL; - config->rsync_path = NULL; - config->temp_dir = NULL; - config->compare_dest = NULL; - config->copy_dest = NULL; - config->link_dest = NULL; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->backup = tmp; - config->backup_dir = receive_str(file_descriptor); - if (config->backup_dir == NULL) - goto error; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->follow_symlinks = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->copy_links = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->safe_links = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->copy_unsafe_links = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->preserve_hard_links = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->preserve_acls = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->preserve_xattrs = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->preserve_devices = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->preserve_sparse = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->update = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->inplace = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->append = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->append_verify = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->delete_excluded = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->delete_after = tmp; - if (!receive_n_data(file_descriptor, &config->max_delete, sizeof(config->max_delete))) - goto error; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->relative = tmp; - if (!receive_int(file_descriptor, &tmp)) - goto error; - config->prune_empty_dirs = tmp; - config->temp_dir = receive_str(file_descriptor); - if (config->temp_dir == NULL) - goto error; - config->server_host = str_dup("127.0.0.1"); - config->server_port = 8080; if (!send_status(file_descriptor, STATUS_OK)) goto error; return config; error: - free(config->version); - free(config->send_directory); - free(config->receive_root_directory); - free(config->server_host); - free(config->backup_dir); - free(config->temp_dir); - free(config); + config_delete(config); return NULL; } + +Config* config_receive(int file_descriptor) { + return config_receive_with_validate(file_descriptor, NULL, NULL); +} diff --git a/src/shared/config.h b/src/shared/config.h index e178318..da3770e 100644 --- a/src/shared/config.h +++ b/src/shared/config.h @@ -2,12 +2,72 @@ #define CONFIG_H #include "array_list.h" +#include "checksum.h" #include #include #include +#include typedef enum { TRANSPORT_TCP, TRANSPORT_SSH } TransportType; +/* --outbuf stdout/stderr buffering style (client-only launch concern, never + * crosses the wire). OUTBUF_BLOCK is the default, matching the stdio default + * (fully buffered when output is not a terminal). */ +typedef enum { + OUTBUF_BLOCK = 0, /* _IOFBF */ + OUTBUF_LINE, /* _IOLBF */ + OUTBUF_NONE /* _IONBF */ +} OutbufMode; + +/* Receiver-side staging state for --delay-updates. Forward-declared here so + Config can carry it; the concrete type lives in delay_updates.h. */ +typedef struct DelayUpdatesContext DelayUpdatesContext; + +/* Alternate basis-directory modes (--compare-dest / --copy-dest / + * --link-dest). Each flag adds one entry to the ordered Config->basis_dirs + * list; the receiver consults entries in command-line order and stops at the + * first exact match, mirroring rsync's basis-dir priority rules. */ +typedef enum { + BASIS_DEST_NONE = 0, + BASIS_DEST_COMPARE, /* compare only: never copies, never materializes */ + BASIS_DEST_COPY, /* local copy of the matched basis file */ + BASIS_DEST_LINK /* hard link to the matched basis file */ +} BasisDestType; + +typedef struct BasisDest { + BasisDestType type; + char* path; /* relative to the destination root (receiver-confined) */ +} BasisDest; + +/* One resolved FROM:TO identity-mapping rule (--usermap / --groupmap). Both + * fields are numeric ids. IDENTITY_MATCH_ANY (-1) in `from` is rsync's '*' + * wildcard (matches any transmitted id); IDENTITY_CURRENT (-1) in `to` makes + * the receiver resolve the receiving process's own current euid/egid at apply + * time. Names are resolved to numbers at parse time on the client (see + * identity.h for the exact subset). */ +typedef struct { + int32_t from; + int32_t to; +} IdentityMap; + +/* --sockopts=OPTIONS allowlist. Only these option names are accepted; anything + * else is rejected (never silently ignored). TCP_NODELAY, SO_KEEPALIVE and + * SO_REUSEADDR are boolean options (value 0/1); SO_RCVBUF and SO_SNDBUF take a + * non-negative byte count. All are applied as int-sized setsockopt values. */ +typedef enum { + SOCKOPT_TCP_NODELAY = 0, + SOCKOPT_SO_KEEPALIVE, + SOCKOPT_SO_RCVBUF, + SOCKOPT_SO_SNDBUF, + SOCKOPT_SO_REUSEADDR, + SOCKOPT_COUNT +} SockOptId; + +typedef struct { + SockOptId id; /* allowlist index */ + int value; /* 0/1 for booleans, byte count for SO_RCVBUF/SO_SNDBUF */ +} SockOptEntry; + typedef struct Config { char* version; char* send_directory; @@ -18,23 +78,70 @@ typedef struct Config { bool use_compression; bool use_sendfile; bool use_metadata; + bool use_executability; + bool metadata_explicitly_disabled; bool show_progress; bool dry_run; + bool remove_source_files; bool use_delete; int compression_level; + int compression_threads; unsigned long long chunk_size; int ssh_port; TransportType transport; char* ssh_destination; + /* Daemon module selection (Wave A, protocol 2.15.0). Client-composed from a + * host::module/path destination; NULL or "" means "no module" (the ordinary + * standalone-server path). Crosses the wire as a trailing config-frame + * string so the daemon can look the module up in its own config and confine + * the connection to the module's root (never a client-chosen root). */ + char* module; + /* Daemon password authentication (A7 remediation, protocol 2.19.0). + * Client-composed from a --password-file whose first meaningful line is + * `user:password`: the client sends ONLY the username in the config frame + * (auth_user); the literal password is kept in auth_password CLIENT-SIDE for + * the duration of the SCRAM challenge/response and is NEVER serialized. Both + * are NULL when the client has no credentials to present; a module WITHOUT + * `auth users` stays open and the server ignores any credentials that do + * arrive (the client sends them opportunistically and the server decides). */ + char* auth_user; + char* auth_password; + /* Client-only path of --password-file (never crosses the wire; it is read to + * populate auth_user/auth_password before connecting). */ + char* password_file; char* fastsync_server_path; + /* --iconv=CONVERT_SPEC (protocol 2.16.0, rsync compatibility): convert the + * charset of FILE NAMES at the wire boundary. CONVERT_SPEC is + * "LOCAL[,REMOTE]": LOCAL is the charset of our own file names, REMOTE is + * the remote side's charset and defaults to LOCAL. The sender converts + * every path LOCAL->REMOTE before transmitting it; the receiver converts + * every received path back REMOTE->LOCAL before creating/writing it. The + * FULL SPEC crosses the wire as a trailing config-frame string so each end + * derives its own LOCAL and the wire (REMOTE) charset symmetrically. NULL + * (or "") means no conversion: identity with zero overhead. See charset.c + * and the PROTOCOL_VERSION note below. */ + char* iconv_spec; char** exclude_patterns; int exclude_count; char** include_patterns; int include_count; unsigned long long max_size; unsigned long long min_size; + unsigned long long max_alloc; bool use_incremental; + bool ignore_times; + bool size_only; bool use_delta; + bool whole_file; + /* -y/--fuzzy: when a file must be transferred and the destination holds no + * usable file at the exact path, the receiver may reuse a SIMILAR-named + * existing regular file in the same destination directory as the delta + * basis so the sender transmits only the differences. Crosses the wire + * (the receiver performs the candidate search); the CLI implies + * --incremental + --delta because the similar-basis only matters on the + * receiver-driven delta path. Off by default. */ + bool fuzzy; + int modify_window; uint32_t delta_block_size; unsigned long long delta_max_file_size; bool use_tls; @@ -59,6 +166,15 @@ typedef struct Config { bool copy_links; bool safe_links; bool copy_unsafe_links; + /* Phase 4 symlink-trust. -k/--copy-dirlinks and --munge-links are + * CLIENT/sender-side only (they decide how the SENDER scans and rewrites + * symlinks; the receiver never reads them), so they never cross the wire. + * -K/--keep-dirlinks is a RECEIVER-side policy (follow an in-root destination + * symlink-to-directory as a directory) and CROSSES the wire along with + * --munge-links (so the receiver knows to unmunge). */ + bool copy_dirlinks; /* client-only, sender-side (-k) */ + bool munge_links; /* crosses the wire */ + bool keep_dirlinks; /* crosses the wire (-K) */ // Issue #121: Extended metadata preservation bool preserve_hard_links; @@ -66,50 +182,554 @@ typedef struct Config { bool preserve_xattrs; bool preserve_devices; bool preserve_sparse; + /* Phase 4 special/devices: preserve special files (FIFOs, sockets) and device + * nodes on the destination by recreating them (mknod/mkfifo) instead of + * transferring content. preserve_specials mirrors rsync --specials (the + * special-file half of -D); preserve_devices mirrors --devices (the device + * half of -D); both CROSS the wire so the receiver knows a special/device + * entry must be recreated rather than written as a regular file. */ + bool preserve_specials; + /* --copy-devices: copy the CONTENT of a source device as an ordinary regular + * file on the destination (rsync's non-privileged safe mode), instead of + * recreating the device node. CROSSES the wire (receiver treats the entry as + * a regular file, which is the default, so this is belt-and-braces). */ + bool copy_devices; + /* --write-devices: write the received data directly INTO an existing device + * node on the destination instead of creating a regular file. Dangeroud; + * see RSYNC_COMPAT.md for the tight gating. CROSSES the wire. */ + bool write_devices; // Issue #122: Output/logging options bool itemize_changes; char* out_format; + char* log_file_format; int info_level; int debug_level; bool list_only; bool human_readable; + bool eight_bit_output; // Issue #127: Transfer modes + bool existing; + bool ignore_existing; bool update; bool inplace; + bool delay_updates; + bool use_fsync; bool append; bool append_verify; + /* --preallocate: allocates the destination file's full expected space up + * front (before any data is written) so a transfer that would overflow disk + * fails fast at allocation time and the file is laid out contiguously, + * avoiding fragmentation. Receiver-side, crosses the wire. */ + bool preallocate; // Issue #128: Extended delete options + /* --delete-excluded: also delete destination entries that were excluded on + * the source. Default (off) matches rsync: excluded paths are protected from + * deletion. Crosses the wire (the sender encodes the choice by whether it + * transmits a protected-prefix list with the keep-set manifest). */ bool delete_excluded; bool delete_after; + /* --max-delete=NUM: the receiver refuses to delete more than NUM entries per + * run (all-or-nothing: when the extras would exceed NUM nothing is removed and + * the transfer fails with a distinct error). -1 == no client limit (the + * server hard bound MAX_SERVER_DELETE_COUNT still applies). */ int max_delete; + /* --ignore-errors (client-only, never serialized): a sender-side source I/O + * error (an unreadable directory during the scan) normally aborts the run so + * no deletion happens; with --ignore-errors the scan continues and the + * (partial) keep-set is still transmitted so the deletion runs. */ + bool ignore_errors; + /* --force (receiver-side): a regular file may replace a destination + * directory by removing that (possibly non-empty, symlink-safe) directory + * tree first, instead of failing the write. Crosses the wire. */ + bool force_delete; + /* --ignore-missing-args (client-only, never serialized): a --files-from + * entry that does not exist under the source is silently skipped instead of + * failing the run. Sender-side only: nothing is sent for it and it never + * enters the keep-set. Implied by --delete-missing-args. */ + bool ignore_missing_args; + /* --delete-missing-args: implies --ignore-missing-args; additionally each + * missing entry's destination mirror (computed like a present entry's wire + * path) is deleted receiver-side. Crosses the wire and is gated by the + * server's --allow-delete policy like --delete. rsync-parity: independent + * of ordinary --delete processing (it does not imply --delete); a non-empty + * directory mirror is only removed with --force or --delete in effect, and + * the missing-args deletions are not counted toward --max-delete. */ + bool delete_missing_args; - // Issue #129: Advanced file selection - ArrayList* filters; - char* files_from; - bool cvs_exclude; + // Issue #129: Advanced file selection. These fields are CLIENT-ONLY: they are + // never serialized to the wire (the receiver must not learn them). + ArrayList* filters; /* --filter=RULE rule strings, in order */ + char* files_from; /* --files-from path (may be NULL) */ + void* files_from_set; /* parsed FileListSet* allow-set, or NULL */ + bool from0; /* -0/--from0: NUL-delimited *-from files */ + bool cvs_exclude; /* -C/--cvs-exclude: standard CVS ignore set */ + bool per_dir_filter; /* -F: apply per-directory .rsync-filter files */ bool prune_empty_dirs; + bool one_file_system; /* -x/--one-file-system: do not cross filesystem boundaries */ + /* -R/--relative: crosses the wire; with --files-from listed entries keep + * their bare relative destination path (no source-root mirror prefix). */ bool relative; + /* --no-implied-dirs: client-only. With -R + --files-from, refuse to place a + * listed file whose ancestor directory is not itself explicitly listed. */ + bool no_implied_dirs; + /* -d/--dirs: client-only. Transfer the directory entries named by the + * source argument / --files-from list without recursing into contents. */ + bool dirs; + /* --mkpath: crosses the wire. Tells the server to create the destination + * root directory (and missing leading components below its authorized root) + * at connection start instead of requiring it to already exist. */ + bool mkpath; // Issue #130: Remote shell/connection options + /* -e/--rsh: the remote-shell program used to establish the SSH transport. + * NULL means the default "ssh". Client-only launch concern: NEVER crosses + * the wire (it is not meaningful to the daemon/server handshake). */ char* rsh_command; - char* rsync_path; + /* --blocking-io: leave the SSH transport socket without + * SO_RCVTIMEO/SO_SNDTIMEO so it blocks naturally instead of timing out. + * Client-only launch concern: NEVER crosses the wire. */ + bool blocking_io; + /* --outbuf mode (OutbufMode): stdout/stderr buffering. Client-only launch + * concern: NEVER crosses the wire. */ + int outbuf; + bool old_args; char* temp_dir; - char* compare_dest; - char* copy_dest; - char* link_dest; + /* --remote-option=OPT (Phase 5, long form only): one or more extra command-line + * options to append to the REMOTE server invocation over SSH. CLIENT-ONLY: + * they are composed into the remote command line by ssh_build_remote_command() + * (each valid word is shell-escaped with the same quoting boundary as the + * server path), and are NEVER serialized into the binary config frame. They + * do NOT cross the wire and are never parsed on the receiver process. */ + char** remote_options; + int remote_option_count; + /* Alternate basis directories, ordered by command-line appearance. Each + * entry's type selects compare/copy/link behavior on an exact match. These + * cross the wire so the receiver can consult them; they are interpreted + * relative to the destination root and confined there. */ + BasisDest* basis_dirs; + int basis_count; + + // PR #174: Partial transfer resumption + char* partial_dir; + + // PR #178: Backup versioning + char* suffix; + + // PR #179: Delete policies + bool delete_before; + + /* rsync deletion-timing family (real from Phase 3). At most one of + delete_before / delete_during / delete_delay / delete_after may be set, and + only together with use_delete (the CLI implies --delete for each of them). + delete_before and delete_during select the EARLY engine mode: the keep-set + manifest is transmitted before any file data and extras are removed then, + acknowledged, before the first data byte. delete_delay and delete_after + select the LATE commit mode: extras are removed only after the whole + transfer has succeeded (plain --delete keeps this mode). The exact + semantics and the divergences from rsync are documented in RSYNC_COMPAT.md + and in config_delete_timing_early() below. */ + bool delete_during; + bool delete_delay; + + // PR #181: IPv6 and bind address + char* address; + char* bind_address; + bool ipv6; + bool ipv4; + /* --sockopts=OPTIONS (Phase 5, Wave B): strict allowlist of TCP/socket + * options applied via setsockopt after socket() and before connect()/bind(). + * These are LOCAL socket concerns: they never cross the wire config frame. + * .address is the outgoing/source bind address (--address); .bind_address is + * reserved for daemon-side binding and is not wired yet. */ + SockOptEntry* sockopts; + int sockopt_count; + + // PR #182: Daemon/server mode + bool daemon; + char* daemon_config; + bool server_mode; + /* --no-motd (Wave C): CLIENT-ONLY, never crosses the wire. Suppresses + * DISPLAY of the daemon's MOTD; the daemon still sends the MOTD frame, so + * the client reads and discards it to keep the stream in sync. rsync's + * --no-motd is likewise a client-side display switch. Default false (the + * MOTD is shown when a daemon offers one). */ + bool no_motd; + + // PR #183: Checksum comparison + bool checksum; + + // PR #184: Compression algorithm negotiation + char* compress_choice; + char* chmod_spec; + + /* --checksum-choice / --cc and --checksum-seed. checksum_algo is the id of + * the whole-file content-digest algorithm used by the per-file --incremental + * handshake (sender computes it, receiver compares it to skip unchanged + * files) and by the basis-dir content verification. checksum_seed is passed + * to xxHash64 (and to the delta block strong hash, low 32 bits); md5 has no + * seed so it is ignored there. Both cross the wire: the receiver MUST hash + * the on-disk old file with the same algorithm and seed to reach a matching + * digest. Defaults (XXH64 / seed 0) reproduce the pre-existing behavior + * byte-for-byte. */ + int checksum_algo; /* ChecksumAlgo, default CHECKSUM_ALGO_XXH64 */ + uint64_t checksum_seed; /* default 0 */ + + char** skip_compress_suffixes; + int skip_compress_count; + bool skip_compress_set; + + // Issue #131: Identity mapping. These configure whether and how the receiver + // applies ownership when it is actually preserved/applied. ALL of them cross + // the wire (protocol 2.11.0) so the receiver resolves and applies ownership + // with the exact policy the client requested. Plain -M/--preserve still does + // NOT apply ownership (FastSync's deliberate conservative default); it is + // only attempted when at least one of these is set (see identity.h). + /* --numeric-ids: no name lookup, use the transmitted numeric ids raw. */ + bool numeric_ids; + /* --chown USER (owner) override; IDENTITY_CURRENT = the receiver's euid. */ + bool chown_uid_set; + int32_t chown_uid; + /* --chown :GROUP (group) override; IDENTITY_CURRENT = the receiver's egid. */ + bool chown_gid_set; + int32_t chown_gid; + /* --usermap / --groupmap entries, in order (first match wins). */ + IdentityMap* usermap; + int usermap_count; + IdentityMap* groupmap; + int groupmap_count; + + /* --super / --no-super (P7 Wave E, protocol 2.18.0): receiver-side privilege + * policy for super-user activities confined below the authorized receive + * root. SUPER_MODE_AUTO (default) preserves the pre-existing best-effort + * behavior: the confined super-user operation is ALWAYS attempted and an + * unprivileged attempt is refused by the kernel and skipped per entry. + * SUPER_MODE_ON (--super) explicitly REQUESTS those activities (char/block + * device-node creation, --write-devices); it does NOT imply --numeric-ids and + * never enables ownership application on its own. SUPER_MODE_OFF + * (--no-super) FORBIDS them even when running as root. FastSync NEVER + * elevates privileges (no setuid/seteuid/setgid) and never bypasses the + * fd-relative confinement (file_open_secure_parent, O_NOFOLLOW, root checks); + * --super only permits an attempt that is already confined. Crosses the wire + * as a trailing int so the receiver can enforce the policy. See + * privilege_super_permitted() and identity_ownership_requested() in + * identity.h. */ + int super_mode; + + // Receiver-side runtime staging registry for --delay-updates. Never sent + // over the wire and never set on the sender side. + DelayUpdatesContext* delay_context; + + // Phase 4: metadata time preservation. -U/--atimes and -N/--crtimes capture + // and transmit the source access / birth time (both sender and receiver + // effect, so they CROSS the wire). --omit-dir-times/-O and + // --omit-link-times/-J are receiver-side prefs (CROSS the wire). Their + // exact capture/transmit/apply semantics are documented in RSYNC_COMPAT.md. + /* -U/--atimes: preserve source access times on the destination. */ + bool preserve_atimes; + /* -N/--crtimes: capture+transmit source birth time; see RSYNC_COMPAT for the + * receiver not-applied divergence. */ + bool preserve_crtimes; + /* -O/--omit-dir-times: do not apply mtimes to directories. */ + bool omit_dir_times; + /* -J/--omit-link-times: do not apply times to symlinks. */ + bool omit_link_times; + /* --open-noatime: CLIENT-ONLY (never crosses the wire). The sender opens + * source files with O_NOATIME so reading for transfer does not bump the + * source access time. */ + bool open_noatime; + + // Phase 4: xattr / ACL / fake-super preservation. + /* -X/--xattrs and -A/--acls toggle the sender's capture and the receiver's + * application of per-file extended attributes (xattrs). Both cross the wire: + * the sender only transmits the bounded, whitelisted attribute set it + * captures and the receiver re-validates namespaces/sizes before applying + * fd-relative. With neither set (the default) no xattr block is sent, so the + * wire is byte-identical to prior protocol versions for unaffected runs. */ + /* true when preserve_xattrs || preserve_acls; the sender/receiver gate the + * xattr wire block on this single flag. */ + bool use_xattrs; + /* --fake-super: receiver-only. When set, each written file additionally gets + * a reserved user.fastsync.stat xattr recording the source uid/gid/mode/mtime + * so a later privileged restore could re-apply them. Crosses the wire. */ + bool fake_super; + /* --copy-as=USER[:GROUP] (P7 Wave E, protocol 2.18.0). Safe-subset + * implementation, a documented divergence from rsync's real identity switch: + * the receiver does NOT change its process credentials (FastSync's receiver + * is multithreaded, so a setuid/seteuid drop would be unsafe). Instead the + * receiver FORCES the ownership of every entry it writes to copy_as_uid / + * copy_as_gid through the existing confined, fd-relative identity path + * (fchown/fchownat), which REQUIRES receiver privilege (root); an + * unprivileged receiver REFUSES the whole transfer up front at the config + * handshake (never a silent wrong-ownership result). All three fields CROSS + * the wire as a trailing config-frame block so the receiver learns the + * requested ids; see the PROTOCOL_VERSION note below. */ + bool copy_as_set; + int32_t copy_as_uid; + int32_t copy_as_gid; + + // Phase 5: --trust-sender + /* Long-form-only, receiver-local policy. rsync's --trust-sender tells the + * receiving side to trust that the sender already produced a sane file list, + * relaxing the receiver's own up-front re-validation of every incoming path. + * In FastSync the receiver normally double-checks each transmitted file-list + * entry (empty / ".." path-traversal rejection) and refuses to materialize a + * symlink whose target could escape the receive root. When trust_sender is + * set, those redundant list-level re-checks are SKIPPED: the receiving side + * trusts the sender's list instead of re-validating it (fewer checks, faster, + * potentially unsafe, matching rsync). It is a LOCAL receiver policy and is + * NEVER serialized into the config frame (it exists only on the process that + * actually receives the file list). Even under trust_sender the low-level + * fd-relative confinement primitives (file_open_secure_parent, the O_NOFOLLOW + * parent walk, leaf/destination confinement) are deliberately KEPT as a hard + * floor, so a hostile sender still cannot write or link outside the + * authorized root (see the phase-5 notes in RSYNC_COMPAT.md). Off by + * default; only relaxes validation when explicitly requested. */ + bool trust_sender; + + // Phase 6: --stop-after / --stop-at + /* Client-only sender-side transfer stop deadlines. --stop-after=MINS stops + * the transfer after a number of elapsed minutes (checked against + * CLOCK_MONOTONIC so clock changes do not skew it); --stop-at=TIME stops at + * an absolute wall-clock time (HH:MM, HH:MM:SS, or now+N[smhd]). At the + * deadline the run stops elegantly at the next chunk/file boundary and the + * completion tail still runs (exit 0). Both are LOCAL to the sending + * process and are NEVER serialized into the config frame. */ + int stop_after_mins; /* --stop-after=MINS minutes; 0 when unset */ + time_t stop_at; /* --stop-at=... absolute wall-clock deadline */ + bool stop_at_set; /* true when --stop-at was given */ + + // Phase 6: --write-batch / --only-write-batch / --read-batch + /* Client-only residual-batch paths. A residual batch is a self-contained + * single-file record of the whole source tree (full file images using the + * chunk codec), independent of any live server. --write-batch=FILE runs the + * normal live transfer AND additionally emits the batch FILE; + * --only-write-batch=FILE emits FILE only (no destination, no server); + * --read-batch=FILE applies FILE to the destination (no source, no server). + * All three are LOCAL to the driving process and are NEVER serialized into + * the config frame (the batch paths bypass the transport entirely). */ + char* write_batch; /* --write-batch=FILE path, or NULL */ + char* only_write_batch; /* --only-write-batch=FILE path, or NULL */ + char* read_batch; /* --read-batch=FILE path, or NULL */ } Config; -#define PROTOCOL_VERSION "1.3.0" +/* Phase 5 (remote-option wave): 2.13.0 -> 2.14.0. + * + * WHY the bump, grounded in the wire: the binary config-frame layout is + * UNCHANGED by this wave (neither --remote-option nor --trust-sender adds a + * serialized field; see the field comments above). --remote-option is + * forwarded to the remote server over the SSH remote-command line + * (ssh_build_remote_command) and --trust-sender is a purely local receiver + * policy, so there is no new frame byte to negotiate. The bump is still the + * correct release marker for Phase 5 because the client-to-server INVOCATION + * surface changed: a client that composes remote-options expects a server that + * knows how to honor them, and the only safe way to express "this feature set + * is one coordinated release" is the strict same-version handshake FastSync + * already performs for every release. A 2.14 client against a 2.13 server + * fails the version check cleanly up front (rather than the remote server + * rejecting an unfamiliar forwarded argv at a confusing later point), which is + * exactly what the lockstep convention of this project requires. */ +/* Daemon Wave A: 2.14.0 -> 2.15.0. + * + * WHY the bump, grounded in the wire: this wave really does add a serialized + * field to the binary config frame. The client sends its requested daemon + * module name (Config->module) as a new trailing string on the frame (sent + * after the Phase-4 xattr block and before the STATUS_OK/STATUS_ERROR ack, in + * config_send/config_receive), and the daemon reads it to select which module + * root confines the connection. Any config-frame layout change must bump the + * protocol version because a peer that does not parse the new trailing bytes + * would desynchronize on the frame boundary; the strict same-version handshake + * (config_receive rejects a mismatched version before parsing anything else) + * is what keeps a 2.15 client and a 2.14 server from ever reaching that state. + * + * NOTE: daemon module-selection bump owned by Wave A (2.15.0); later daemon + * waves (auth, motd) must not bump PROTOCOL_VERSION. Wave B (auth) added the + * credential fields (auth_user + password digest) as further trailing + * config-frame strings AFTER the Wave A module string, with a presence int + * prefix. This is not a new frame version: sender and receiver of a 2.15.0 + * build always read and write the same full layout (the strict same-version + * handshake rejects any other version before a byte of the frame is parsed), + * so a peer can never desynchronize on the added tail. The 2.15.0 release + * ships Wave A + Wave B together; the bump stays owned by Wave A. + * + * Wave C (MOTD) adds NO config-frame field and no version bump either. On the + * daemon listener path only, the server sends one MOTD string frame AFTER the + * config-frame STATUS_OK (server.c handler), and every 2.15.0 daemon client + * reads that frame right after the ack (client_send.c) -- symmetric + * server->client in every build, so the strict same-version handshake keeps the + * two peers in lockstep and nothing can desynchronize. The --stdio SSH path + * sends/reads no MOTD at all. + * + * --iconv Wave (P6): 2.15.0 -> 2.16.0. + * + * WHY the bump, grounded in the wire: the --iconv feature adds a serialized + * field to the binary config frame. The client sends the full CONVERT_SPEC + * (Config->iconv_spec) as a new trailing string AFTER the Wave A/B daemon-auth + * block (in config_send/config_receive), so the receiver knows the wire charset + * (the REMOTE half) before the first file name arrives. Any config-frame + * layout change must bump the protocol version: a peer that does not parse the + * new trailing bytes would desynchronize on the frame boundary, and the strict + * same-version handshake (config_receive rejects a mismatched version before + * parsing anything else) is what keeps a 2.16 client and a 2.15 server from + * ever reaching that state. + * + * Times Wave (P7 Wave D): 2.16.0 -> 2.17.0. + * + * WHY the bump, grounded in the wire: this wave makes -O/--omit-dir-times and + * -J/--omit-link-times REAL by adding directory and symlink time preservation. + * The config-frame LAYOUT is unchanged (the omit flags already crossed the + * wire), but the FRAME STREAM gains a new terminal frame: after all file data + * and the optional delete manifest, the sender transmits STATUS_DIR_TIMES + * frame(s) (each a count followed by (path, metadata) pairs, chunked so no + * frame exceeds the receiver's MAX_MANIFEST_ENTRIES bound) carrying every + * source directory's captured times, so the receiver can apply them AFTER all of a + * directory's children have been written (writing a child bumps the parent's + * mtime). Symlink entries already carry their metadata on the STATUS_SYMLINK + * frame; the receiver now applies it (utimensat/lchown with + * AT_SYMLINK_NOFOLLOW) unless -J is set. Any change to the frame sequence must + * bump the protocol version: a 2.16 peer that does not know STATUS_DIR_TIMES + * would desynchronize on the unknown frame, and the strict same-version + * handshake (config_receive rejects a mismatched version before parsing + * anything else) is what keeps a 2.17 client and a 2.16 server from ever + * reaching that state. + * + * Privilege Wave (P7 Wave E): 2.17.0 -> 2.18.0. + * + * WHY the bump, grounded in the wire: this wave adds the receiver-side + * privilege flags --super/--no-super and --copy-as=USER[:GROUP]. The + * config-frame layout gains two new trailing blocks AFTER the --iconv + * CONVERT_SPEC string, in this fixed order: (1) send_privilege_options / + * receive_privilege_options send one int (Config->super_mode, 0..2), then + * (2) send_copy_as_options / receive_copy_as_options send a presence int and, + * when set, the target uid and gid (both int32). The receiver uses + * super_mode to decide whether it may attempt super-user activities + * (ownership application, char/block device-node creation) already confined + * below the authorized receive root, and the copy-as ids to force the + * ownership of every entry it writes (the safe-subset --copy-as model). The + * receiver REQUIRES privilege for copy-as: an unprivileged receiver refuses + * the transfer at the config handshake (server_module_gate) instead of silently + * ignoring the flag. Any config-frame layout change must bump the protocol + * version: a peer that does not parse the new trailing bytes would + * desynchronize on the frame boundary, and the strict same-version handshake + * (config_receive rejects a mismatched version before parsing anything else) is + * what keeps a 2.18 client and a 2.17 server from ever reaching that state. + * --super never elevates privileges; it only permits a confined attempt, and + * --copy-as never switches process credentials (see RSYNC_COMPAT.md). + * + * A7 Auth Wave: 2.18.0 -> 2.19.0. + * + * WHY the bump, grounded in the wire: the daemon auth block on the config frame + * loses the hard-wired password digest (it becomes `[int present][str_redacted + * username]`), and the frame stream gains the SCRAM challenge/response + * (STATUS_AUTH_CHALLENGE -> STATUS_AUTH_RESPONSE -> STATUS_AUTH_OK) between the + * config frame and the STATUS_OK ack. A 2.18 peer would desynchronize on both + * the shorter auth block and the new status frames, so the strict same-version + * handshake (config_receive rejects a mismatched version before parsing + * anything else) is what keeps a 2.19 client and a 2.18 server from ever + * reaching that state. SECURITY: a 2.19 store holds a salted PBKDF2 verifier + * and cannot verify (and refuses to load) a legacy unsalted-SHA-256 store line, + * so an old bearer digest can never be replayed against a 2.19 daemon. */ +#define PROTOCOL_VERSION "2.19.0" #define DEFAULT_CHUNK_SIZE (10 * 1024 * 1024) +/* Upper bound on total basis-dir entries (rsync caps --link-dest at 20). */ +#define MAX_BASIS_DIRS 64 + +/* Identity-mapping sentinels and bounds (see identity.h for semantics). + * IDENTITY_MATCH_ANY is a usermap/groupmap FROM '*' (matches any id); + * IDENTITY_CURRENT is a chown / map TO '*' (resolve to the receiver's current + * euid/egid at apply time). */ +#define IDENTITY_MATCH_ANY (-1) +#define IDENTITY_CURRENT (-1) +#define MAX_IDENTITY_MAP 128 + +/* --super / --no-super tri-state (Config->super_mode). AUTO (default) and ON + * both permit a confined super-user attempt (AUTO preserves FastSync's + * historical best-effort behavior; an unprivileged attempt is refused by the + * kernel and skipped per entry); OFF forbids the attempt even for root. See + * privilege_super_mode_permitted() in identity.h. */ +#define SUPER_MODE_AUTO 0 +#define SUPER_MODE_ON 1 +#define SUPER_MODE_OFF 2 Config* config_create(void); void config_delete(Config* config); + +/* Wipe the client-side plaintext auth password (and username) from a Config + * before it is freed or handed off. Safe on a NULL/empty Config and idempotent + * (it clears the pointers after burning). config_delete calls this + * automatically; a caller that drops a Config earlier may call it explicitly. */ +void config_burn_auth(Config* config); + bool config_send(int file_descriptor, const Config* config); Config* config_receive(int file_descriptor); -bool is_remote_dest(const char* s); +bool config_is_remote_dest(const char* s); void config_parse_ssh_dest(Config* config); +/* A ConfigValidateFunc may return this sentinel to tell + * config_receive_with_validate that the callback ALREADY sent a terminal status + * frame (e.g. STATUS_AUTH_FAILED, then closed) and the frame must be abandoned + * without an additional STATUS_ERROR. A normal rejection returns a message + * string (logged, then STATUS_ERROR); NULL accepts. */ +#define CONFIG_VALIDATE_ALREADY_TERMINATED ((const char*)-1) + +/* Server-side config-frame gate (daemon module selection, Wave A). A server + * that needs to make an accept/reject decision about a received Config BEFORE + * it sends the STATUS_OK ack (so a rejected connection is refused cleanly with + * no data transferred) passes a callback here; it runs after the frame parses + * and validates but before the STATUS_OK/STATUS_ERROR ack. Return NULL to + * accept the connection; return a non-NULL message to reject it (the message + * is logged server-side and STATUS_ERROR is sent in place of STATUS_OK), or the + * CONFIG_VALIDATE_ALREADY_TERMINATED sentinel when the callback already sent + * its own terminal status. The callback runs in the connection's own process, + * so it may set up per-module process state (e.g. the authorized root) and + * drive the daemon auth handshake. context is an opaque caller pointer. */ +typedef const char* (*ConfigValidateFunc)(const Config* config, void* context); +Config* config_receive_with_validate(int file_descriptor, ConfigValidateFunc validate, + void* context); + +/* Daemon-destination (host::module[/path]) helpers, Wave A. config_is_remote_dest + * recognizes the ordinary rsync-style single-colon host:path form used by the + * SSH transport; config_is_daemon_dest recognizes the double-colon form that + * selects a daemon module over TCP. config_parse_transport_dest is the single + * entry point main() uses: it parses a :: destination as a daemon TCP + * destination (host -> server_host, module -> config->module, path -> + * receive_root_directory) and otherwise falls back to the existing SSH + * host:path handling. */ +bool config_is_daemon_dest(const char* s); +/* Returns 1 when the destination was daemon syntax and was parsed, 0 when it + * is not daemon syntax (nothing changed), -1 on an invalid daemon destination + * (a message is logged and config is left untouched). */ +int config_parse_daemon_dest(Config* config); +/* Returns 1/0/-1 mirroring config_parse_daemon_dest when the destination is + * daemon syntax; otherwise runs the existing SSH host:path parse and returns + * 0. */ +int config_parse_transport_dest(Config* config); + +/* True when the negotiated delete timing performs the extra-file deletion + * BEFORE the transfer data (--delete-before / --delete-during). The flag is + * a pure function of the config and is used identically on the sender (to pick + * the manifest-first frame order) and the receiver (to delete when the early + * manifest arrives). When false the deletion is committed only after the whole + * transfer succeeded (--delete / --delete-after / --delete-delay). */ +bool config_delete_timing_early(const Config* config); +/* Delete-timing sanity: with deletion enabled at most one timing flag may be + * set (none = the default delete-after commit timing); without deletion no + * timing flag may be set (each timing flag implies --delete). */ +bool config_has_valid_delete_timing(const Config* config); +/* True when at least one --compare-dest/--copy-dest/--link-dest was set. */ +bool config_has_basis(const Config* config); +/* Append one basis-dir entry. Returns 0 on success, -1 on allocation failure. */ +int config_basis_append(Config* config, BasisDestType type, const char* path); +/* Validate a client-provided basis-dir path (relative, confined, non-empty). */ +bool config_basis_path_valid(const char* path); + +/* Parse and validate a --sockopts=OPTIONS comma-separated "OPT=VAL" list into a + * malloc'd array of at most *out_count entries. Returns 0 on success (the + * caller takes ownership of *out), or -1 on the first invalid option name or + * value. Pure/static-analysis friendly: performs no socket calls, so it is + * directly unit-testable. */ +int config_sockopts_parse(const char* spec, SockOptEntry** out, int* out_count); + #endif diff --git a/src/shared/credentials.c b/src/shared/credentials.c new file mode 100644 index 0000000..f1a0c98 --- /dev/null +++ b/src/shared/credentials.c @@ -0,0 +1,1322 @@ +#include "credentials.h" +#include "log.h" +#include "utils.h" +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/* One store entry: a username and its salted PBKDF2 verifier. The plaintext + * password never appears here (and never on the daemon host); the verifier is + * not replayable because the proof is bound to a per-connection nonce. */ +typedef struct CredentialEntry { + char* user; + uint8_t salt[CREDENTIAL_SALT_LEN]; + uint32_t iters; + uint8_t stored_key[CREDENTIAL_KEY_LEN]; + uint8_t server_key[CREDENTIAL_KEY_LEN]; +} CredentialEntry; + +struct CredentialStore { + CredentialEntry* entries; + int count; + int capacity; + /* Store-wide uniform PBKDF2 iteration count. Every entry must agree on it + * (the parser refuses a store whose entries disagree), so a miss can be + * challenged with the same count as a hit and the count itself never leaks + * membership. Unused (0) for an empty store. */ + uint32_t iters; + /* Store-wide secret loaded from (or created in) the exact-mode-0600 + * `.dummykey` sidecar, so it also survives a daemon restart. The + * dummy salt handed out for an unknown/off-list user is + * HMAC-SHA256(dummy_key, username)[:SALT_LEN], so repeated probes of the same + * username always see an identical challenge while different usernames differ + * -- with no fresh-random tell, and cross-restart stability hides the + * restart-gated enumeration oracle. */ + uint8_t dummy_key[CREDENTIAL_KEY_LEN]; +}; + +/* Exact marker prefix of the new store verifier field. */ +#define CREDENTIAL_STORE_PREFIX "$fastsync$1$pbkdf2-sha256$" +#define CREDENTIAL_AUTH_PREFIX "FastSync-Auth-v1" +/* Exact-mode-0600 sidecar holding the persistent store-wide dummy key, placed + * next to the credential store (`.dummykey`). */ +#define CREDENTIAL_DUMMY_KEY_SUFFIX ".dummykey" + +/* Fixed dummy keys used when a user is unknown or off the module's list. They + * can never authenticate because acceptance additionally requires found=true. */ +static const uint8_t k_dummy_stored_key[CREDENTIAL_KEY_LEN] = {0}; +static const uint8_t k_dummy_server_key[CREDENTIAL_KEY_LEN] = {0}; + +static void set_error(char* err, size_t err_size, const char* fmt, ...) { + if (!err || err_size == 0) + return; + va_list args; + va_start(args, fmt); + vsnprintf(err, err_size, fmt, args); + va_end(args); +} + +static bool is_comment_char(char c) { + return c == '#' || c == ';'; +} + +/* Open a --password-file / --early-input after verifying the EXACT inode we + * will read: it must be owned by the effective user and grant no group/other + * permission bit (so 0600 and stricter modes such as 0400 are accepted), + * mirroring the TLS private-key check. This only rejects group/other bits, + * deliberately unlike the dummy-key sidecar which requires EXACT mode 0600. We + * open by + * path and then fstat the resulting fd (rather than stat()ing the path first + * and reopening it), so the permission decision is made on the same inode that + * is read and cannot be raced by swapping the path between check and open. + * The path may be a process-substitution pipe (`<(...)` -> /dev/fd/N), so + * regular files and FIFOs are accepted when the ownership/mode checks pass. + * + * Returns a FILE* the caller must fclose, or NULL with `err` filled. */ +static FILE* secret_file_open(const char* path, char* err, size_t err_size) { + int fd = open(path, O_RDONLY | O_CLOEXEC); + if (fd < 0) { + set_error(err, err_size, "cannot open secret file '%s': %s", path, strerror(errno)); + return NULL; + } + struct stat st; + if (fstat(fd, &st) != 0) { + set_error(err, err_size, "cannot stat secret file '%s': %s", path, strerror(errno)); + close(fd); + return NULL; + } + bool is_readable_kind = S_ISREG(st.st_mode) || S_ISFIFO(st.st_mode); + if (!is_readable_kind || st.st_uid != geteuid() || (st.st_mode & (S_IRWXG | S_IRWXO)) != 0) { + set_error(err, err_size, + "refusing to read secret file '%s': it must be owned by the current user and " + "owner-only (0600), not accessible to group/other", + path); + close(fd); + return NULL; + } + FILE* fp = fdopen(fd, "r"); + if (!fp) { + set_error(err, err_size, "cannot read secret file '%s': %s", path, strerror(errno)); + close(fd); + return NULL; + } + return fp; +} + +/* Trim leading/trailing ASCII space and tab in place; returns the new start. */ +static char* trim_space(char* s) { + while (*s == ' ' || *s == '\t') + s++; + size_t len = strlen(s); + while (len > 0 && (s[len - 1] == ' ' || s[len - 1] == '\t')) + s[--len] = '\0'; + return s; +} + +/* A username is a single token: non-empty, bounded, and free of whitespace and + * control characters. The same rule is applied to store users, client-file + * users and the module `auth users` gate so an exact strcmp can never be + * confused by invisible characters. */ +static bool username_wellformed(const char* user) { + if (!user || *user == '\0') + return false; + size_t len = strlen(user); + if (len > CREDENTIAL_MAX_USER_LEN) + return false; + for (size_t i = 0; i < len; i++) { + unsigned char c = (unsigned char)user[i]; + if (c <= 0x20 || c == 0x7f) + return false; + } + return true; +} + +bool credentials_username_valid(const char* user) { + return username_wellformed(user); +} + +static int hex_value(char c) { + if (c >= '0' && c <= '9') + return c - '0'; + if (c >= 'a' && c <= 'f') + return c - 'a' + 10; + return -1; +} + +/* True for the OLD `user:SHA256HEX` secret form: exactly 64 lowercase hex + * digits. Such a line is refused loudly (and never accepted) so an operator + * cannot keep a replayable bearer digest in place after the protocol bump. */ +static bool secret_is_legacy_hex(const char* s) { + if (!s) + return false; + for (int i = 0; i < 64; i++) { + if (hex_value(s[i]) < 0) + return false; + } + return s[64] == '\0'; +} + +bool credentials_b64_encode(const uint8_t* in, size_t n, char* out, size_t out_sz) { + if (!in || !out) + return false; + if (n > (size_t)INT_MAX) + return false; + size_t encoded_len = 4 * ((n + 2) / 3); + if (out_sz < encoded_len + 1) + return false; + int written = EVP_EncodeBlock((unsigned char*)out, in, (int)n); + if (written < 0 || (size_t)written != encoded_len) + return false; + out[encoded_len] = '\0'; + return true; +} + +bool credentials_b64_decode(const char* in, uint8_t* out, size_t out_sz, size_t* out_len) { + if (!in || !out || !out_len) + return false; + size_t len = strlen(in); + /* Every value we decode is short (a 32-byte key is 44 chars); refusing long + * input keeps the scratch buffer fixed and bounds a hostile frame. */ + if (len == 0 || (len % 4) != 0 || len > 256) + return false; + size_t padded_len = (len / 4) * 3; + size_t decoded_len = padded_len; + if (in[len - 1] == '=') + decoded_len--; + if (len >= 2 && in[len - 2] == '=') + decoded_len--; + if (decoded_len > out_sz) + return false; + /* EVP_DecodeBlock writes the full (padded) quantum, so decode into a scratch + * buffer sized for it and copy only the real bytes out. The single `done` + * path burns the scratch on failure as well as success, so no partial secret + * survives an early return. */ + uint8_t scratch[192] = {0}; + bool ok = false; + int n = EVP_DecodeBlock(scratch, (const unsigned char*)in, (int)len); + if (n < 0 || (size_t)n != padded_len) + goto done; + memcpy(out, scratch, decoded_len); + *out_len = decoded_len; + ok = true; +done: + credentials_burn((char*)scratch, sizeof(scratch)); + return ok; +} + +bool credentials_random_bytes(uint8_t* out, size_t n) { + if (!out || n == 0 || n > (size_t)INT_MAX) + return false; + return RAND_bytes(out, (int)n) == 1; +} + +/* HMAC-SHA256 via the OpenSSL 3 EVP_MAC API (HMAC() is deprecated). */ +static bool hmac_sha256(const uint8_t* key, size_t key_len, const uint8_t* data, size_t data_len, + uint8_t out[CREDENTIAL_KEY_LEN]) { + EVP_MAC* mac = EVP_MAC_fetch(NULL, "HMAC", NULL); + if (!mac) + return false; + EVP_MAC_CTX* ctx = EVP_MAC_CTX_new(mac); + EVP_MAC_free(mac); + if (!ctx) + return false; + OSSL_PARAM params[2]; + params[0] = OSSL_PARAM_construct_utf8_string("digest", (char*)"SHA256", 0); + params[1] = OSSL_PARAM_construct_end(); + size_t out_len = 0; + bool ok = + EVP_MAC_init(ctx, key, key_len, params) == 1 && EVP_MAC_update(ctx, data, data_len) == 1 && + EVP_MAC_final(ctx, out, &out_len, CREDENTIAL_KEY_LEN) == 1 && out_len == CREDENTIAL_KEY_LEN; + EVP_MAC_CTX_free(ctx); + return ok; +} + +static bool sha256(const uint8_t* data, size_t len, uint8_t out[CREDENTIAL_KEY_LEN]) { + unsigned int out_len = 0; + if (EVP_Digest(data, len, out, &out_len, EVP_sha256(), NULL) != 1) + return false; + return out_len == CREDENTIAL_KEY_LEN; +} + +bool credentials_compute_keys(const char* password, const uint8_t salt[CREDENTIAL_SALT_LEN], + uint32_t iters, uint8_t client_key[CREDENTIAL_KEY_LEN], + uint8_t stored_key[CREDENTIAL_KEY_LEN], + uint8_t server_key[CREDENTIAL_KEY_LEN]) { + if (!password || !salt) + return false; + /* Enforce the full [MIN,MAX] policy here so no caller can derive a verifier + * with a work factor outside the validated store range. */ + if (iters < CREDENTIAL_MIN_ITERS || iters > CREDENTIAL_MAX_ITERS) + return false; + size_t password_len = strlen(password); + if (password_len > CREDENTIAL_MAX_PASSWORD_LEN || password_len > (size_t)INT_MAX) + return false; + uint8_t k[CREDENTIAL_KEY_LEN]; + if (PKCS5_PBKDF2_HMAC(password, (int)password_len, salt, CREDENTIAL_SALT_LEN, (int)iters, + EVP_sha256(), CREDENTIAL_KEY_LEN, k) != 1) { + credentials_burn((char*)k, sizeof(k)); + return false; + } + uint8_t derived_client[CREDENTIAL_KEY_LEN]; + uint8_t derived_server[CREDENTIAL_KEY_LEN]; + bool ok = hmac_sha256(k, sizeof(k), (const uint8_t*)"Client Key", 10, derived_client) && + hmac_sha256(k, sizeof(k), (const uint8_t*)"Server Key", 10, derived_server); + if (ok && stored_key) + ok = sha256(derived_client, sizeof(derived_client), stored_key); + if (ok && client_key) + memcpy(client_key, derived_client, CREDENTIAL_KEY_LEN); + if (ok && server_key) + memcpy(server_key, derived_server, CREDENTIAL_KEY_LEN); + credentials_burn((char*)k, sizeof(k)); + credentials_burn((char*)derived_client, sizeof(derived_client)); + credentials_burn((char*)derived_server, sizeof(derived_server)); + return ok; +} + +static void write_be32(uint8_t* out, uint32_t value) { + out[0] = (uint8_t)(value >> 24); + out[1] = (uint8_t)(value >> 16); + out[2] = (uint8_t)(value >> 8); + out[3] = (uint8_t)value; +} + +bool credentials_build_auth_message(const char* user, const uint8_t* snonce, const uint8_t* cnonce, + uint8_t* out, size_t out_sz, size_t* out_len) { + if (!user || !snonce || !cnonce || !out || !out_len) + return false; + size_t user_len = strlen(user); + if (user_len > CREDENTIAL_MAX_USER_LEN) + return false; + size_t total = 16 + 4 + user_len + 4 + CREDENTIAL_NONCE_LEN + 4 + CREDENTIAL_NONCE_LEN; + if (out_sz < total) + return false; + size_t off = 0; + memcpy(out + off, CREDENTIAL_AUTH_PREFIX, 16); + off += 16; + write_be32(out + off, (uint32_t)user_len); + off += 4; + memcpy(out + off, user, user_len); + off += user_len; + write_be32(out + off, CREDENTIAL_NONCE_LEN); + off += 4; + memcpy(out + off, snonce, CREDENTIAL_NONCE_LEN); + off += CREDENTIAL_NONCE_LEN; + write_be32(out + off, CREDENTIAL_NONCE_LEN); + off += 4; + memcpy(out + off, cnonce, CREDENTIAL_NONCE_LEN); + off += CREDENTIAL_NONCE_LEN; + *out_len = off; + return true; +} + +bool credentials_client_proof(const uint8_t client_key[CREDENTIAL_KEY_LEN], + const uint8_t stored_key[CREDENTIAL_KEY_LEN], + const uint8_t server_key[CREDENTIAL_KEY_LEN], const uint8_t* auth_msg, + size_t msg_len, uint8_t proof[CREDENTIAL_KEY_LEN], + uint8_t server_sig[CREDENTIAL_KEY_LEN]) { + if (!client_key || !stored_key || !server_key || !auth_msg || !proof || !server_sig) + return false; + uint8_t client_sig[CREDENTIAL_KEY_LEN]; + bool ok = hmac_sha256(stored_key, CREDENTIAL_KEY_LEN, auth_msg, msg_len, client_sig); + if (ok) { + for (size_t i = 0; i < CREDENTIAL_KEY_LEN; i++) + proof[i] = client_key[i] ^ client_sig[i]; + ok = hmac_sha256(server_key, CREDENTIAL_KEY_LEN, auth_msg, msg_len, server_sig); + } + credentials_burn((char*)client_sig, sizeof(client_sig)); + return ok; +} + +bool credentials_verify_response(const CredentialVerifier* v, const char* user, + const uint8_t* snonce, const uint8_t* cnonce, + const uint8_t proof[CREDENTIAL_KEY_LEN], + uint8_t server_sig_out[CREDENTIAL_KEY_LEN]) { + if (!v || !user || !snonce || !cnonce || !proof || !server_sig_out) + return false; + uint8_t auth_msg[CREDENTIAL_AUTH_MESSAGE_MAX]; + size_t msg_len = 0; + if (!credentials_build_auth_message(user, snonce, cnonce, auth_msg, sizeof(auth_msg), &msg_len)) + return false; + uint8_t client_sig[CREDENTIAL_KEY_LEN]; + uint8_t client_key[CREDENTIAL_KEY_LEN]; + uint8_t recovered[CREDENTIAL_KEY_LEN]; + uint8_t server_sig[CREDENTIAL_KEY_LEN]; + bool computed = hmac_sha256(v->stored_key, CREDENTIAL_KEY_LEN, auth_msg, msg_len, client_sig); + if (computed) { + for (size_t i = 0; i < CREDENTIAL_KEY_LEN; i++) + client_key[i] = proof[i] ^ client_sig[i]; + computed = sha256(client_key, CREDENTIAL_KEY_LEN, recovered); + } + if (computed) + computed = hmac_sha256(v->server_key, CREDENTIAL_KEY_LEN, auth_msg, msg_len, server_sig); + if (computed) + memcpy(server_sig_out, server_sig, CREDENTIAL_KEY_LEN); + /* Always run the constant-time key compare (even when `found` is false) and + * fold the accept decision with bitwise AND so no short-circuit reveals + * whether the user was found. A tampered nonce changes the AuthMessage and + * so the recovered key. */ + bool key_match = false; + if (computed) + key_match = credentials_secure_equal((const char*)recovered, (const char*)v->stored_key, + CREDENTIAL_KEY_LEN); + bool accept = computed & v->found & key_match; + credentials_burn((char*)auth_msg, sizeof(auth_msg)); + credentials_burn((char*)client_sig, sizeof(client_sig)); + credentials_burn((char*)client_key, sizeof(client_key)); + credentials_burn((char*)recovered, sizeof(recovered)); + credentials_burn((char*)server_sig, sizeof(server_sig)); + return accept; +} + +static bool entries_equal(const CredentialEntry* a, const CredentialEntry* b) { + return a->iters == b->iters && + credentials_secure_equal((const char*)a->salt, (const char*)b->salt, + CREDENTIAL_SALT_LEN) && + credentials_secure_equal((const char*)a->stored_key, (const char*)b->stored_key, + CREDENTIAL_KEY_LEN) && + credentials_secure_equal((const char*)a->server_key, (const char*)b->server_key, + CREDENTIAL_KEY_LEN); +} + +static bool append_entry(CredentialStore* store, const char* user, const uint8_t* salt, + uint32_t iters, const uint8_t* stored_key, const uint8_t* server_key) { + if (store->count == store->capacity) { + int new_capacity = store->capacity == 0 ? 8 : store->capacity * 2; + CredentialEntry* grown = + realloc(store->entries, (size_t)new_capacity * sizeof(CredentialEntry)); + if (!grown) + return false; + store->entries = grown; + store->capacity = new_capacity; + } + CredentialEntry* entry = &store->entries[store->count]; + memset(entry, 0, sizeof(*entry)); + entry->user = str_dup(user); + if (!entry->user) + return false; + memcpy(entry->salt, salt, CREDENTIAL_SALT_LEN); + entry->iters = iters; + memcpy(entry->stored_key, stored_key, CREDENTIAL_KEY_LEN); + memcpy(entry->server_key, server_key, CREDENTIAL_KEY_LEN); + store->count++; + return true; +} + +static int find_user(const CredentialStore* store, const char* user) { + for (int i = 0; i < store->count; i++) { + if (strcmp(store->entries[i].user, user) == 0) + return i; + } + return -1; +} + +/* Parse the new `$fastsync$1$pbkdf2-sha256$...` verifier field in place. */ +static bool parse_verifier_secret(char* secret, CredentialEntry* entry, const char* path, + int line_no, const char* user, char* err, size_t err_size) { + if (secret_is_legacy_hex(secret)) { + set_error(err, err_size, + "credential file '%s' line %d: legacy unsalted SHA-256 secret for user '%s' is not " + "accepted (protocol 2.19.0 uses a salted PBKDF2 verifier); regenerate the store " + "with --hash-credentials", + path, line_no, user); + return false; + } + const char* prefix = CREDENTIAL_STORE_PREFIX; + size_t prefix_len = strlen(prefix); + if (strncmp(secret, prefix, prefix_len) != 0) { + set_error(err, err_size, + "credential file '%s' line %d: expected a '%s...' verifier for user '%s' (regenerate " + "a legacy line with --hash-credentials)", + path, line_no, prefix, user); + return false; + } + char* cursor = secret + prefix_len; + const char* iters_str = cursor; + char* sep = strchr(cursor, '$'); + if (!sep) + goto malformed; + *sep = '\0'; + const char* salt_str = sep + 1; + sep = strchr(salt_str, '$'); + if (!sep) + goto malformed; + *sep = '\0'; + const char* stored_str = sep + 1; + sep = strchr(stored_str, '$'); + if (!sep) + goto malformed; + *sep = '\0'; + const char* server_str = sep + 1; + if (*iters_str == '\0' || *salt_str == '\0' || *stored_str == '\0' || *server_str == '\0') + goto malformed; + + char* end = NULL; + unsigned long parsed = strtoul(iters_str, &end, 10); + if (!end || *end != '\0' || parsed < CREDENTIAL_MIN_ITERS || parsed > CREDENTIAL_MAX_ITERS) + goto malformed; + entry->iters = (uint32_t)parsed; + + size_t decoded = 0; + if (!credentials_b64_decode(salt_str, entry->salt, CREDENTIAL_SALT_LEN, &decoded) || + decoded != CREDENTIAL_SALT_LEN) + goto malformed; + if (!credentials_b64_decode(stored_str, entry->stored_key, CREDENTIAL_KEY_LEN, &decoded) || + decoded != CREDENTIAL_KEY_LEN) + goto malformed; + if (!credentials_b64_decode(server_str, entry->server_key, CREDENTIAL_KEY_LEN, &decoded) || + decoded != CREDENTIAL_KEY_LEN) + goto malformed; + return true; + +malformed: + set_error(err, err_size, + "credential file '%s' line %d: malformed verifier for user '%s' (expected " + "'%s$$$')", + path, line_no, user, prefix); + return false; +} + +/* Parse one credential store file into a fresh store. Duplicate usernames + * WITHIN one file are an error (ambiguous). A NULL path yields an empty + * store. */ +static CredentialStore* load_store_file(const char* path, char* err, size_t err_size) { + CredentialStore* store = calloc(1, sizeof(CredentialStore)); + if (!store) { + set_error(err, err_size, "out of memory allocating credential store"); + return NULL; + } + if (!path) + return store; + + FILE* fp = secret_file_open(path, err, err_size); + if (!fp) { + credentials_free(store); + return NULL; + } + + int line_no = 0; + char line[CREDENTIAL_MAX_LINE + 2]; + bool ok = true; + + while (fgets(line, sizeof(line), fp)) { + line_no++; + size_t len = strlen(line); + if (len == CREDENTIAL_MAX_LINE + 1 && line[len - 1] != '\n' && !feof(fp)) { + set_error(err, err_size, "credential file '%s' line %d exceeds the %d-byte limit", path, + line_no, CREDENTIAL_MAX_LINE); + ok = false; + break; + } + if (len > 0 && line[len - 1] == '\n') + line[--len] = '\0'; + if (len > 0 && line[len - 1] == '\r') + line[--len] = '\0'; + + char* cursor = line; + while (*cursor == ' ' || *cursor == '\t') + cursor++; + if (*cursor == '\0' || is_comment_char(*cursor)) + continue; /* blank or comment */ + + char* colon = strchr(cursor, ':'); + if (!colon) { + set_error(err, err_size, + "credential file '%s' line %d: expected 'user:$fastsync$...' (no ':' found)", path, + line_no); + ok = false; + break; + } + *colon = '\0'; + const char* user = trim_space(cursor); + char* secret = trim_space(colon + 1); + if (!username_wellformed(user)) { + set_error(err, err_size, + "credential file '%s' line %d: invalid username (must be 1-%d " + "non-whitespace characters)", + path, line_no, CREDENTIAL_MAX_USER_LEN); + ok = false; + break; + } + CredentialEntry parsed; + memset(&parsed, 0, sizeof(parsed)); + if (!parse_verifier_secret(secret, &parsed, path, line_no, user, err, err_size)) { + ok = false; + break; + } + /* Every entry must agree on the iteration count, so a miss can be answered + * with the store-wide count without leaking membership. */ + if (store->count == 0) { + store->iters = parsed.iters; + } else if (store->iters != parsed.iters) { + set_error(err, err_size, + "credential file '%s' line %d: iteration count %u disagrees with the store-wide %u " + "(the store must be uniform)", + path, line_no, parsed.iters, store->iters); + ok = false; + break; + } + if (find_user(store, user) >= 0) { + set_error(err, err_size, "credential file '%s' line %d: duplicate entry for user '%.*s'", + path, line_no, (int)strlen(user), user); + ok = false; + break; + } + if (!append_entry(store, user, parsed.salt, parsed.iters, parsed.stored_key, + parsed.server_key)) { + set_error(err, err_size, "out of memory reading credential file '%s'", path); + ok = false; + break; + } + } + + if (ok && ferror(fp)) { + set_error(err, err_size, "error reading credential file '%s': %s", path, strerror(errno)); + ok = false; + } + fclose(fp); + credentials_burn(line, sizeof(line)); + if (!ok) { + credentials_free(store); + return NULL; + } + return store; +} + +/* Validate and read an already-open `.dummykey` sidecar. Fails closed on + * anything that is not an exact-mode-0600 regular file of exactly + * CREDENTIAL_KEY_LEN bytes, so a loosened, swapped or truncated file can never + * silently change the dummy challenge. */ +static bool read_dummy_key_fd(int fd, const char* path, uint8_t out[CREDENTIAL_KEY_LEN], char* err, + size_t err_size) { + struct stat st; + if (fstat(fd, &st) != 0) { + set_error(err, err_size, "cannot stat dummy key file '%s': %s", path, strerror(errno)); + return false; + } + if (!S_ISREG(st.st_mode) || st.st_uid != geteuid() || (st.st_mode & 07777) != 0600 || + st.st_size != (off_t)CREDENTIAL_KEY_LEN) { + set_error(err, err_size, + "refusing to read dummy key file '%s': it must be an owned regular file with exact " + "mode 0600 and exactly %d bytes", + path, CREDENTIAL_KEY_LEN); + return false; + } + size_t got = 0; + while (got < CREDENTIAL_KEY_LEN) { + ssize_t n = read(fd, out + got, CREDENTIAL_KEY_LEN - got); + if (n < 0) { + if (errno == EINTR) + continue; + set_error(err, err_size, "cannot read dummy key file '%s': %s", path, strerror(errno)); + return false; + } + if (n == 0) + break; + got += (size_t)n; + } + if (got != CREDENTIAL_KEY_LEN) { + set_error(err, err_size, "dummy key file '%s' is truncated", path); + return false; + } + return true; +} + +/* fsync the directory containing `path` (best effort). After publishing the + * sidecar with link(2), syncing the directory makes the new name durable so a + * crash cannot leave a restart without the key it just started using. */ +static void fsync_containing_dir(const char* path) { + char* dir = str_dup(path); + if (!dir) + return; + char* slash = strrchr(dir, '/'); + if (!slash) { + free(dir); + dir = str_dup("."); + if (!dir) + return; + } else if (slash == dir) { + slash[1] = '\0'; /* keep the leading '/' */ + } else { + *slash = '\0'; + } + int dfd = open(dir, O_RDONLY | O_DIRECTORY | O_CLOEXEC); + free(dir); + if (dfd < 0) + return; + fsync(dfd); + close(dfd); +} + +/* Load the persistent dummy key for `store_path` from its `.dummykey` + * sidecar, creating it (exact mode 0600, 32 random bytes) if absent. A NULL + * store_path (empty store) yields a fresh ephemeral key. Reading an existing + * sidecar fails CLOSED on any validation error; only the CREATE path degrades + * to an ephemeral key (with a warning) when the filesystem cannot hold the + * sidecar (e.g. read-only mount), so a daemon still starts. + * + * Creation is ATOMIC: the key is written to a private same-directory temp file + * and hard-linked into place, so a concurrent starter (or reader) never observes + * a partial/zero sidecar that would fail the load closed. Returns false only + * when the CSPRNG itself fails (or a present-but-invalid sidecar is found). */ +static bool load_or_create_dummy_key(const char* store_path, uint8_t out[CREDENTIAL_KEY_LEN], + char* err, size_t err_size) { + if (!store_path) { + if (!credentials_random_bytes(out, CREDENTIAL_KEY_LEN)) { + set_error(err, err_size, "failed to generate the credential store dummy key"); + return false; + } + return true; + } + + size_t path_len = strlen(store_path); + size_t suffix_len = sizeof(CREDENTIAL_DUMMY_KEY_SUFFIX); /* includes the NUL */ + if (path_len > SIZE_MAX - suffix_len) { + set_error(err, err_size, "credential store path is too long to build a dummy key path"); + return false; + } + char* sidecar = malloc(path_len + suffix_len); + if (!sidecar) { + set_error(err, err_size, "out of memory building the dummy key path"); + return false; + } + int n = snprintf(sidecar, path_len + suffix_len, "%s%s", store_path, CREDENTIAL_DUMMY_KEY_SUFFIX); + if (n < 0 || (size_t)n >= path_len + suffix_len) { + set_error(err, err_size, "credential store path is too long to build a dummy key path"); + free(sidecar); + return false; + } + + /* Readers reject a planted symlink (O_NOFOLLOW) and never block on a planted + * FIFO (O_NONBLOCK; fstat rejects the non-regular file before any data read). + * Any open error other than ENOENT fails closed. */ + int fd = open(sidecar, O_RDONLY | O_NOFOLLOW | O_NONBLOCK | O_CLOEXEC); + if (fd >= 0) { + bool ok = read_dummy_key_fd(fd, sidecar, out, err, err_size); + close(fd); + free(sidecar); + return ok; + } + if (errno != ENOENT) { + /* The sidecar exists but cannot be opened for reading (EACCES, or ELOOP + * from a symlink): fail closed rather than substituting a different key. */ + set_error(err, err_size, "cannot open dummy key file '%s': %s", sidecar, strerror(errno)); + free(sidecar); + return false; + } + + /* Publish atomically: write a private same-directory temp file, fsync it, + * then hard-link it into place. A concurrent reader therefore only ever + * sees a complete 32-byte sidecar (or none), never a partial/zero file. + * + * The temp name carries both the pid and a fresh random suffix, so it is not + * predictable. If the name nevertheless already exists (a SIGKILL/crash + * leftover, pid reuse, or a planted file) the stale temp is removed and the + * O_EXCL create is retried once, so it can never silently defeat persistence + * for this pid. */ + uint8_t fresh[CREDENTIAL_KEY_LEN]; + if (!credentials_random_bytes(fresh, CREDENTIAL_KEY_LEN)) { + set_error(err, err_size, "failed to generate the credential store dummy key"); + free(sidecar); + return false; + } + + uint8_t name_rand[8]; + if (!credentials_random_bytes(name_rand, sizeof(name_rand))) { + set_error(err, err_size, "failed to generate the dummy key temp name"); + free(sidecar); + credentials_burn((char*)fresh, sizeof(fresh)); + return false; + } + char name_hex[sizeof(name_rand) * 2 + 1]; + static const char hex_digits[] = "0123456789abcdef"; + for (size_t i = 0; i < sizeof(name_rand); i++) { + name_hex[2 * i] = hex_digits[name_rand[i] >> 4]; + name_hex[2 * i + 1] = hex_digits[name_rand[i] & 0x0f]; + } + name_hex[sizeof(name_hex) - 1] = '\0'; + + char tmp_suffix[64]; + int pn = snprintf(tmp_suffix, sizeof(tmp_suffix), ".tmp.%ld.%s", (long)getpid(), name_hex); + if (pn < 0 || (size_t)pn >= sizeof(tmp_suffix)) { + set_error(err, err_size, "failed to build the dummy key temp path"); + free(sidecar); + credentials_burn((char*)fresh, sizeof(fresh)); + return false; + } + size_t sidecar_len = (size_t)n; + size_t tmp_len = sidecar_len + (size_t)pn; + char* tmp = malloc(tmp_len + 1); + if (!tmp) { + set_error(err, err_size, "out of memory building the dummy key temp path"); + free(sidecar); + credentials_burn((char*)fresh, sizeof(fresh)); + return false; + } + snprintf(tmp, tmp_len + 1, "%s%s", sidecar, tmp_suffix); + + /* Bounded create: at most one unlink+retry on EEXIST. The retry keeps + * O_EXCL, so only a stale name is reclaimed and a live peer's temp is never + * truncated. */ + int create_errno = 0; + for (int attempt = 0; attempt < 2; attempt++) { + fd = open(tmp, O_WRONLY | O_CREAT | O_EXCL | O_CLOEXEC, 0600); + if (fd >= 0) + break; + create_errno = errno; + if (create_errno != EEXIST || attempt == 1) + break; + unlink(tmp); + } + /* umask can clear owner bits from the 0600 create mode while the reader + * requires an exact 0600, so force the mode on the fd before publishing; a + * failure here is treated like any other create failure (warning + ephemeral + * key) so the published sidecar is always exactly 0600. */ + if (fd >= 0 && fchmod(fd, 0600) != 0) { + create_errno = errno; + close(fd); + unlink(tmp); + fd = -1; + } + if (fd < 0) { + /* Creation failed (read-only filesystem, missing directory, fchmod, ...). + * Warn and fall back to an ephemeral key: unknown-user challenges stay + * deterministic within this daemon lifetime but will change on restart. */ + char* escaped = output_escape(tmp, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, + "cannot create dummy key file %s: %s; using a transient dummy key so unknown-user " + "challenges will change across restarts", + escaped ? escaped : tmp, strerror(create_errno)); + free(escaped); + memcpy(out, fresh, CREDENTIAL_KEY_LEN); + free(tmp); + free(sidecar); + credentials_burn((char*)fresh, sizeof(fresh)); + return true; + } + + size_t written = 0; + bool write_ok = true; + int write_errno = 0; + while (written < CREDENTIAL_KEY_LEN) { + ssize_t w = write(fd, fresh + written, CREDENTIAL_KEY_LEN - written); + if (w < 0) { + if (errno == EINTR) + continue; + write_ok = false; + write_errno = errno; + break; + } + if (w == 0) { + /* A zero-length write is not a system error; errno is stale here, so + * report a clear short-write instead of a bogus strerror(errno). */ + write_ok = false; + write_errno = 0; + break; + } + written += (size_t)w; + } + if (write_ok && fsync(fd) != 0) { + write_ok = false; + write_errno = errno; + } + close(fd); + if (!write_ok) { + /* Do not leave a truncated temp file behind; fall back to an ephemeral key + * instead of failing closed on the next restart. */ + unlink(tmp); + char* escaped = output_escape(tmp, log_get_8_bit_output()); + const char* why = write_errno != 0 ? strerror(write_errno) : "short write"; + log_message(LOG_LEVEL_WARNING, + "cannot write dummy key file %s: %s; using a transient dummy key so unknown-user " + "challenges will change across restarts", + escaped ? escaped : tmp, why); + free(escaped); + memcpy(out, fresh, CREDENTIAL_KEY_LEN); + free(tmp); + free(sidecar); + credentials_burn((char*)fresh, sizeof(fresh)); + return true; + } + + if (link(tmp, sidecar) != 0) { + int link_errno = errno; + if (link_errno == EEXIST) { + /* A concurrent starter published first; adopt its key. Read it back + * through the same hardened path (no symlink, no block, exact mode). */ + int rfd = open(sidecar, O_RDONLY | O_NOFOLLOW | O_NONBLOCK | O_CLOEXEC); + if (rfd < 0) { + set_error(err, err_size, "cannot open dummy key file '%s': %s", sidecar, strerror(errno)); + unlink(tmp); + free(tmp); + free(sidecar); + credentials_burn((char*)fresh, sizeof(fresh)); + return false; + } + bool ok = read_dummy_key_fd(rfd, sidecar, out, err, err_size); + close(rfd); + unlink(tmp); + free(tmp); + free(sidecar); + credentials_burn((char*)fresh, sizeof(fresh)); + return ok; + } + /* Linking failed for another reason (e.g. no hard-link support on this + * filesystem). Warn and fall back to an ephemeral key. */ + unlink(tmp); + char* escaped = output_escape(sidecar, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, + "cannot publish dummy key file %s: %s; using a transient dummy key so unknown-user " + "challenges will change across restarts", + escaped ? escaped : sidecar, strerror(link_errno)); + free(escaped); + memcpy(out, fresh, CREDENTIAL_KEY_LEN); + free(tmp); + free(sidecar); + credentials_burn((char*)fresh, sizeof(fresh)); + return true; + } + + /* Published: make the new directory entry durable, then drop the private + * temp name (the sidecar keeps the inode alive). */ + fsync_containing_dir(sidecar); + unlink(tmp); + memcpy(out, fresh, CREDENTIAL_KEY_LEN); + free(tmp); + free(sidecar); + credentials_burn((char*)fresh, sizeof(fresh)); + return true; +} + +CredentialStore* credentials_load(const char* password_file, const char* early_input_file, + char* err, size_t err_size) { + if (err && err_size) + err[0] = '\0'; + CredentialStore* store = load_store_file(password_file, err, err_size); + if (!store) + return NULL; + /* Load (or create) the store-wide dummy key once for the final (possibly + * merged) store. It makes an unknown-user challenge deterministic AND + * stable across daemon restarts, so a restart cannot be used as a + * username-enumeration oracle. It is persisted in an exact-mode-0600 sidecar + * next to the credential store; a NULL store path (empty store) keeps it + * ephemeral. Fail the load if the CSPRNG is unavailable rather than + * degrading the anti-enumeration property. */ + const char* store_path = password_file ? password_file : early_input_file; + if (!load_or_create_dummy_key(store_path, store->dummy_key, err, err_size)) { + credentials_free(store); + return NULL; + } + if (!early_input_file) + return store; + + CredentialStore* early = load_store_file(early_input_file, err, err_size); + if (!early) { + credentials_free(store); + return NULL; + } + /* A layered store must stay uniform too. */ + if (store->count > 0 && early->count > 0 && store->iters != early->iters) { + set_error(err, err_size, + "credential file '%s' and early-input file '%s' disagree on the iteration count " + "(%u vs %u); the store must be uniform", + password_file, early_input_file, store->iters, early->iters); + credentials_free(early); + credentials_free(store); + return NULL; + } + if (store->count == 0 && early->count > 0) + store->iters = early->iters; + /* Layer early input over the password file: an identical verifier dedupes, a + * differing verifier for the same user is ambiguous and fails closed. */ + for (int i = 0; i < early->count; i++) { + int existing = find_user(store, early->entries[i].user); + if (existing >= 0) { + if (!entries_equal(&store->entries[existing], &early->entries[i])) { + set_error(err, err_size, + "credential file '%s' and early-input file '%s' disagree on the verifier for " + "user '%s'", + password_file, early_input_file, early->entries[i].user); + credentials_free(early); + credentials_free(store); + return NULL; + } + continue; /* identical; nothing to merge */ + } + if (!append_entry(store, early->entries[i].user, early->entries[i].salt, + early->entries[i].iters, early->entries[i].stored_key, + early->entries[i].server_key)) { + set_error(err, err_size, "out of memory merging early-input credentials"); + credentials_free(early); + credentials_free(store); + return NULL; + } + } + credentials_free(early); + return store; +} + +void credentials_free(CredentialStore* store) { + if (!store) + return; + for (int i = 0; i < store->count; i++) { + /* Wipe the derived keys before releasing the entry (A7-4). */ + credentials_burn((char*)store->entries[i].salt, CREDENTIAL_SALT_LEN); + credentials_burn((char*)store->entries[i].stored_key, CREDENTIAL_KEY_LEN); + credentials_burn((char*)store->entries[i].server_key, CREDENTIAL_KEY_LEN); + free(store->entries[i].user); + } + /* The store-wide dummy key is secret (it shapes the miss challenge), so wipe + * it before releasing the store. */ + credentials_burn((char*)store->dummy_key, sizeof(store->dummy_key)); + free(store->entries); + free(store); +} + +int credentials_store_size(const CredentialStore* store) { + return store ? store->count : 0; +} + +bool credentials_store_has(const CredentialStore* store, const char* user) { + return store && find_user(store, user) >= 0; +} + +bool credentials_secure_equal(const char* a, const char* b, size_t len) { + unsigned char diff = 0; + for (size_t i = 0; i < len; i++) + diff |= (unsigned char)a[i] ^ (unsigned char)b[i]; + return diff == 0; +} + +/* Constant-time equality over two usernames. Compares a fixed + * CREDENTIAL_MAX_USER_LEN-byte window (padding with zeros past each string's + * own length) and folds the length difference into the accumulator, so no byte + * returns early. This closes the byte-wise username-enumeration timing oracle + * that a plain strcmp (which short-circuits on the first differing byte) + * would otherwise expose. Over-long inputs are refused (length differs), which + * is a non-secret branch: usernames are bounded in every caller anyway. */ +static bool username_secure_equal(const char* a, const char* b) { + size_t alen = strlen(a); + size_t blen = strlen(b); + if (alen > CREDENTIAL_MAX_USER_LEN || blen > CREDENTIAL_MAX_USER_LEN) + return false; + size_t diff = alen ^ blen; + for (size_t i = 0; i < CREDENTIAL_MAX_USER_LEN; i++) { + unsigned char ac = i < alen ? (unsigned char)a[i] : 0u; + unsigned char bc = i < blen ? (unsigned char)b[i] : 0u; + diff |= (size_t)(ac ^ bc); + } + return diff == 0; +} + +bool credentials_get_verifier(const CredentialStore* store, const char* user, + const char* const* module_users, int n, CredentialVerifier* out) { + if (!out) + return false; + memset(out, 0, sizeof(*out)); + const char* uname = user ? user : ""; + /* The dummy verifier is shaped exactly like a hit: the store-wide uniform + * iteration count (default for an empty store) and fixed dummy keys. */ + out->iters = (store && store->count > 0) ? store->iters : CREDENTIAL_DEFAULT_ITERS; + memcpy(out->stored_key, k_dummy_stored_key, CREDENTIAL_KEY_LEN); + memcpy(out->server_key, k_dummy_server_key, CREDENTIAL_KEY_LEN); + out->found = false; + /* Deterministic per-username dummy salt: HMAC-SHA256(dummy_key, username) + * truncated to the salt length. Two probes of the same unknown username see + * an identical challenge; distinct usernames differ. A NULL store (never + * reached in production) falls back to the all-zero static key. */ + const uint8_t* dummy_key = store ? store->dummy_key : k_dummy_stored_key; + uint8_t mac[CREDENTIAL_KEY_LEN]; + if (!hmac_sha256(dummy_key, CREDENTIAL_KEY_LEN, (const uint8_t*)uname, strlen(uname), mac)) { + credentials_burn((char*)mac, sizeof(mac)); + return false; + } + memcpy(out->salt, mac, CREDENTIAL_SALT_LEN); + credentials_burn((char*)mac, sizeof(mac)); + /* Module-list membership: constant-time full scan, no early break, so the + * list is not a username-enumeration oracle. */ + bool on_list = false; + for (int i = 0; i < n; i++) { + const char* listed = (module_users && user) ? module_users[i] : NULL; + on_list |= listed ? username_secure_equal(listed, user) : false; + } + /* Store lookup is an unconditional constant-time full scan, executed even for + * an off-list user so a probe that is not on the module list still pays the + * same O(store) cost as one that is; skipping it would reopen an off-list + * timing channel. The real verifier is selected only when the user is both + * on the list and matched in the store. */ + const CredentialEntry* match = NULL; + for (int i = 0; store && user && i < store->count; i++) { + if (username_secure_equal(store->entries[i].user, user)) + match = &store->entries[i]; + } + if (on_list && match) { + memcpy(out->salt, match->salt, CREDENTIAL_SALT_LEN); + out->iters = match->iters; + memcpy(out->stored_key, match->stored_key, CREDENTIAL_KEY_LEN); + memcpy(out->server_key, match->server_key, CREDENTIAL_KEY_LEN); + out->found = true; + } + return true; +} + +bool credentials_hash_store_line(const char* user, const char* password, uint32_t iters, char* out, + size_t out_sz, char* err, size_t err_size) { + if (err && err_size) + err[0] = '\0'; + if (!username_wellformed(user)) { + set_error(err, err_size, "invalid username (1-%d non-whitespace characters)", + CREDENTIAL_MAX_USER_LEN); + return false; + } + if (!password || !out || out_sz == 0) { + set_error(err, err_size, "missing password or output buffer"); + return false; + } + if (strlen(password) > CREDENTIAL_MAX_PASSWORD_LEN) { + set_error(err, err_size, "password exceeds %d characters", CREDENTIAL_MAX_PASSWORD_LEN); + return false; + } + if (iters < CREDENTIAL_MIN_ITERS || iters > CREDENTIAL_MAX_ITERS) { + set_error(err, err_size, "iterations %u out of range [%u,%u]", iters, CREDENTIAL_MIN_ITERS, + CREDENTIAL_MAX_ITERS); + return false; + } + uint8_t salt[CREDENTIAL_SALT_LEN]; + uint8_t client_key[CREDENTIAL_KEY_LEN]; + uint8_t stored_key[CREDENTIAL_KEY_LEN]; + uint8_t server_key[CREDENTIAL_KEY_LEN]; + char salt_b64[25]; + char stored_b64[45]; + char server_b64[45]; + bool ok = + credentials_random_bytes(salt, sizeof(salt)) && + credentials_compute_keys(password, salt, iters, client_key, stored_key, server_key) && + credentials_b64_encode(salt, sizeof(salt), salt_b64, sizeof(salt_b64)) && + credentials_b64_encode(stored_key, sizeof(stored_key), stored_b64, sizeof(stored_b64)) && + credentials_b64_encode(server_key, sizeof(server_key), server_b64, sizeof(server_b64)); + int written = -1; + if (ok) { + written = snprintf(out, out_sz, "%s:%s%u$%s$%s$%s", user, CREDENTIAL_STORE_PREFIX, iters, + salt_b64, stored_b64, server_b64); + } + credentials_burn((char*)client_key, sizeof(client_key)); + credentials_burn((char*)stored_key, sizeof(stored_key)); + credentials_burn((char*)server_key, sizeof(server_key)); + credentials_burn((char*)salt, sizeof(salt)); + /* The base64 encodings of the salt/keys are secret material too (A7-4). */ + credentials_burn(salt_b64, sizeof(salt_b64)); + credentials_burn(stored_b64, sizeof(stored_b64)); + credentials_burn(server_b64, sizeof(server_b64)); + if (!ok) { + credentials_burn(out, out_sz); + return false; + } + if (written < 0 || (size_t)written >= out_sz) { + set_error(err, err_size, "output buffer too small for the credential line"); + credentials_burn(out, out_sz); + return false; + } + return true; +} + +int credentials_hash_file(const char* path, uint32_t iters, FILE* out, char* err, size_t err_size) { + if (err && err_size) + err[0] = '\0'; + if (!path || !out) { + set_error(err, err_size, "missing plaintext file or output stream"); + return -1; + } + if (iters < CREDENTIAL_MIN_ITERS || iters > CREDENTIAL_MAX_ITERS) { + set_error(err, err_size, "iterations %u out of range [%u,%u]", iters, CREDENTIAL_MIN_ITERS, + CREDENTIAL_MAX_ITERS); + return -1; + } + FILE* fp = secret_file_open(path, err, err_size); + if (!fp) + return -1; + int line_no = 0; + int result = 0; + char line[CREDENTIAL_MAX_LINE + 2]; + while (fgets(line, sizeof(line), fp)) { + line_no++; + size_t len = strlen(line); + if (len == CREDENTIAL_MAX_LINE + 1 && line[len - 1] != '\n' && !feof(fp)) { + set_error(err, err_size, "plaintext file '%s' line %d exceeds the %d-byte limit", path, + line_no, CREDENTIAL_MAX_LINE); + result = -1; + break; + } + while (len > 0 && (line[len - 1] == '\n' || line[len - 1] == '\r')) + line[--len] = '\0'; + char* cursor = line; + while (*cursor == ' ' || *cursor == '\t') + cursor++; + if (*cursor == '\0' || is_comment_char(*cursor)) + continue; + char* colon = strchr(cursor, ':'); + if (!colon) { + set_error(err, err_size, "plaintext file '%s' line %d: expected 'user:password'", path, + line_no); + result = -1; + break; + } + *colon = '\0'; + const char* user = trim_space(cursor); + const char* password = colon + 1; + if (!username_wellformed(user)) { + set_error(err, err_size, "plaintext file '%s' line %d: invalid username", path, line_no); + result = -1; + break; + } + if (*password == '\0') { + set_error(err, err_size, "plaintext file '%s' line %d: empty password", path, line_no); + result = -1; + break; + } + char store_line[CREDENTIAL_MAX_LINE]; + if (!credentials_hash_store_line(user, password, iters, store_line, sizeof(store_line), err, + err_size)) { + credentials_burn(store_line, sizeof(store_line)); + result = -1; + break; + } + if (fprintf(out, "%s\n", store_line) < 0) { + set_error(err, err_size, "cannot write hashed credentials: %s", strerror(errno)); + credentials_burn(store_line, sizeof(store_line)); + result = -1; + break; + } + credentials_burn(store_line, sizeof(store_line)); + } + if (result == 0 && ferror(fp)) { + set_error(err, err_size, "error reading plaintext file '%s': %s", path, strerror(errno)); + result = -1; + } + credentials_burn(line, sizeof(line)); + fclose(fp); + return result; +} + +int credentials_read_secret_file(const char* path, char** user_out, char** password_out, char* err, + size_t err_size) { + if (user_out) + *user_out = NULL; + if (password_out) + *password_out = NULL; + if (err && err_size) + err[0] = '\0'; + if (!path) { + set_error(err, err_size, "no --password-file path"); + return -1; + } + FILE* fp = secret_file_open(path, err, err_size); + if (!fp) + return -1; + + int line_no = 0; + char line[CREDENTIAL_MAX_LINE + 2]; + int result = -1; + + while (fgets(line, sizeof(line), fp)) { + line_no++; + size_t len = strlen(line); + if (len == CREDENTIAL_MAX_LINE + 1 && line[len - 1] != '\n' && !feof(fp)) { + set_error(err, err_size, "password file '%s' line %d exceeds the %d-byte limit", path, + line_no, CREDENTIAL_MAX_LINE); + goto done; + } + if (len > 0 && line[len - 1] == '\n') + line[--len] = '\0'; + if (len > 0 && line[len - 1] == '\r') + line[--len] = '\0'; + + char* cursor = line; + while (*cursor == ' ' || *cursor == '\t') + cursor++; + if (*cursor == '\0' || is_comment_char(*cursor)) + continue; /* skip blank/comment lines; the first real line is the secret */ + + char* colon = strchr(cursor, ':'); + if (!colon) { + set_error(err, err_size, + "password file '%s' line %d: expected 'user:password' (no ':' found)", path, + line_no); + goto done; + } + *colon = '\0'; + const char* user = trim_space(cursor); + /* Preserve the password's exact bytes: only the line's trailing CR/LF was + * already stripped above. Trimming leading/trailing space here would make + * a password that legitimately begins or ends with whitespace unusable. */ + const char* password = colon + 1; + if (!username_wellformed(user)) { + set_error(err, err_size, + "password file '%s' line %d: invalid username (must be 1-%d " + "non-whitespace characters)", + path, line_no, CREDENTIAL_MAX_USER_LEN); + goto done; + } + if (*password == '\0') { + set_error(err, err_size, "password file '%s' line %d: empty password", path, line_no); + goto done; + } + if (strlen(password) > CREDENTIAL_MAX_PASSWORD_LEN) { + set_error(err, err_size, "password file '%s' line %d: password exceeds %d characters", path, + line_no, CREDENTIAL_MAX_PASSWORD_LEN); + goto done; + } + char* user_dup = str_dup(user); + char* password_dup = str_dup(password); + if (!user_dup || !password_dup) { + free(user_dup); + free(password_dup); + set_error(err, err_size, "out of memory reading password file '%s'", path); + goto done; + } + if (user_out) + *user_out = user_dup; + else + free(user_dup); + if (password_out) + *password_out = password_dup; + else + free(password_dup); + result = 0; + goto done; + } + + if (ferror(fp)) { + set_error(err, err_size, "error reading password file '%s': %s", path, strerror(errno)); + goto done; + } + /* Reached end of file with no meaningful line: the file is empty (or only + * comments), which the client policy rejects. */ + set_error(err, err_size, "password file '%s' contains no 'user:password' line", path); + +done: + /* Wipe the stack line (which may hold the literal password) before return. + * user/password were str_dup'd into their outputs on success, so the stack + * copy is the only remaining plaintext. */ + credentials_burn(line, sizeof(line)); + fclose(fp); + return result; +} + +void credentials_burn(char* secret, size_t len) { + if (!secret) + return; + volatile char* p = (volatile char*)secret; + for (size_t i = 0; i < len; i++) + p[i] = '\0'; +} diff --git a/src/shared/credentials.h b/src/shared/credentials.h new file mode 100644 index 0000000..377fb02 --- /dev/null +++ b/src/shared/credentials.h @@ -0,0 +1,203 @@ +#ifndef CREDENTIALS_H +#define CREDENTIALS_H + +#include +#include +#include +#include + +/* Daemon password authentication (A7 remediation, protocol 2.19.0). + * + * FastSync authenticates a daemon connection with a SCRAM-SHA-256-style + * challenge/response handshake. The daemon stores only a salted PBKDF2 + * verifier (never the password, and never a value that can be replayed as a + * bearer credential): the client proves knowledge of the password against a + * per-connection server nonce, and the server proves the same shared secret + * back. See credentials.c for the exact derivation. + * + * Server credential store format (--password-file and --early-input): one line + * per entry, + * user:$fastsync$1$pbkdf2-sha256$$$$ + * with standard base64, a 16-byte salt and 32-byte keys, and iters in + * [CREDENTIAL_MIN_ITERS, CREDENTIAL_MAX_ITERS]. Blank lines and lines whose + * first non-space character is '#' or ';' are comments. The parser is STRICT: + * a malformed line fails the whole load so a typo can never silently change who + * may log in. A line holding the legacy (unsalted SHA-256 hex) secret is + * hard-rejected with an actionable "legacy" error; there is no auto-upgrade. + * Use `fastsync-server --hash-credentials` to generate new-format lines. + * + * Alongside the store, credentials_load maintains an exact-mode-0600 + * `.dummykey` sidecar holding the store-wide random dummy key. It + * is auto-created on first load and MUST be preserved across restarts: it makes + * the dummy challenge for an unknown user stable for the life of the store, so + * a daemon restart cannot be used as a username-enumeration oracle. A sidecar + * that is not an exact-mode-0600 regular file of exactly 32 bytes fails the load + * (fail closed); creation forces exact 0600 with fchmod (so a restrictive umask + * cannot leave the sidecar unreadable), and only a create/write/fsync/link or + * fchmod failure degrades to a transient per-run key with a warning. NOTE: the + * sidecar requires EXACT 0600, whereas the store / password files only reject + * group/other bits (a deliberate difference). + * + * Client --password-file format: the FIRST meaningful (non-comment, non-blank) + * line is `user:password`, holding the literal password. The client keeps it + * only for the duration of the handshake and wipes it at teardown; the file + * should be mode 0600 and readable only by its owner. */ + +/* Longest accepted credential-file line (excluding the trailing newline). */ +#define CREDENTIAL_MAX_LINE 4096 +/* Upper bound on a username in a credential file and on the wire. Kept well + * below MAX_STRING_SIZE so a wire username can never exhaust anything. */ +#define CREDENTIAL_MAX_USER_LEN 256 +/* Upper bound on a client-file password (before derivation). */ +#define CREDENTIAL_MAX_PASSWORD_LEN 1024 + +/* SCRAM-SHA-256 parameters. Salt and client nonce sizes are fixed by the + * shared-auth-message framing; keys are always 32 bytes (SHA-256). */ +#define CREDENTIAL_SALT_LEN 16 +#define CREDENTIAL_NONCE_LEN 32 +#define CREDENTIAL_KEY_LEN 32 +#define CREDENTIAL_DEFAULT_ITERS 600000u +#define CREDENTIAL_MIN_ITERS 100000u +#define CREDENTIAL_MAX_ITERS 10000000u +/* Buffer size for the full AuthMessage (prefix + three length-prefixed fields). + * Worst case: 16 + 4 + 256 + 4 + 32 + 4 + 32. */ +#define CREDENTIAL_AUTH_MESSAGE_MAX \ + (16 + 4 + CREDENTIAL_MAX_USER_LEN + 4 + CREDENTIAL_NONCE_LEN + 4 + CREDENTIAL_NONCE_LEN) + +typedef struct CredentialStore CredentialStore; + +/* One resolved verifier. `found` is false for an unknown user or a user not on + * a module's auth list; the remaining fields then hold a deterministic dummy + * salt (HMAC of the store-wide dummy key over the username), the store-wide + * uniform iteration count (default for an empty store) and fixed dummy keys, so + * the server can run the same challenge/response math with no enumeration or + * timing oracle. */ +typedef struct { + uint8_t salt[CREDENTIAL_SALT_LEN]; + uint32_t iters; + uint8_t stored_key[CREDENTIAL_KEY_LEN]; + uint8_t server_key[CREDENTIAL_KEY_LEN]; + bool found; +} CredentialVerifier; + +/* Load the daemon credential store. + * + * password_file and early_input_file are both NULL-or-path, matching the + * server's --password-file and --early-input options. A file that cannot be + * opened or that fails the strict grammar is a hard error (err filled, NULL + * returned) -- the daemon fails CLOSED rather than serving an auth-required + * module with a partial store. Both files may be NULL, which yields an empty + * store (every auth-required module then refuses connections). Every entry in + * the resulting store must agree on the iteration count; entries that disagree + * (within one file or across the two layered sources) are rejected. When both + * are given, the --early-input file is layered over --password-file: a duplicate + * username whose verifier matches is deduplicated; one whose verifier differs + * is an error (the two sources disagree), never a silent pick. + * + * The returned store is heap-owned; free it with credentials_free. */ +CredentialStore* credentials_load(const char* password_file, const char* early_input_file, + char* err, size_t err_size); + +/* Wipe every stored key/salt and free the store. */ +void credentials_free(CredentialStore* store); + +/* True when `user` is a single bounded token free of whitespace/control bytes + * (the rule applied to store users, client-file users and the module list). */ +bool credentials_username_valid(const char* user); + +/* Standard base64. encode writes NUL-terminated output to out (size out_sz). + * decode writes the raw bytes to out (capacity out_sz) and stores the length; + * the input must be a well-formed padded base64 string. Both return false on + * NULL arguments, a bad character/length, or insufficient output space. */ +bool credentials_b64_encode(const uint8_t* in, size_t n, char* out, size_t out_sz); +bool credentials_b64_decode(const char* in, uint8_t* out, size_t out_sz, size_t* out_len); + +/* Fill out[0..n) from the CSPRNG (RAND_bytes). Returns false on failure. */ +bool credentials_random_bytes(uint8_t* out, size_t n); + +/* Resolve `user` against the store AND the module's auth-user list. The list + * scan is a constant-time full-length comparison with no early break. On a + * miss, *out is filled with a dummy verifier (a deterministic per-username salt + * derived from the store's dummy key, the store-wide uniform iteration count, + * fixed dummy keys, found=false). Returns false on invalid arguments or an + * HMAC/crypto primitive failure. */ +bool credentials_get_verifier(const CredentialStore* store, const char* user, + const char* const* module_users, int n, CredentialVerifier* out); + +/* Derive the SCRAM keys from a plaintext password: + * K = PBKDF2-HMAC-SHA256(password, salt, iters, 32) + * ClientKey = HMAC-SHA256(K, "Client Key"); StoredKey = SHA256(ClientKey) + * ServerKey = HMAC-SHA256(K, "Server Key") + * Any of client_key/stored_key/server_key may be NULL when not needed. + * `iters` must lie in [CREDENTIAL_MIN_ITERS, CREDENTIAL_MAX_ITERS]. */ +bool credentials_compute_keys(const char* password, const uint8_t salt[CREDENTIAL_SALT_LEN], + uint32_t iters, uint8_t client_key[CREDENTIAL_KEY_LEN], + uint8_t stored_key[CREDENTIAL_KEY_LEN], + uint8_t server_key[CREDENTIAL_KEY_LEN]); + +/* Serialize the shared AuthMessage: + * "FastSync-Auth-v1" || be32(len(user)) || user + * || be32(32) || server_nonce + * || be32(32) || client_nonce + * out must hold at least CREDENTIAL_AUTH_MESSAGE_MAX bytes. *out_len receives + * the number of bytes written. */ +bool credentials_build_auth_message(const char* user, const uint8_t* snonce, const uint8_t* cnonce, + uint8_t* out, size_t out_sz, size_t* out_len); + +/* Client side: ClientProof = ClientKey XOR HMAC(StoredKey, AuthMessage), and + * the expected ServerSignature = HMAC(ServerKey, AuthMessage). */ +bool credentials_client_proof(const uint8_t client_key[CREDENTIAL_KEY_LEN], + const uint8_t stored_key[CREDENTIAL_KEY_LEN], + const uint8_t server_key[CREDENTIAL_KEY_LEN], const uint8_t* auth_msg, + size_t msg_len, uint8_t proof[CREDENTIAL_KEY_LEN], + uint8_t server_sig[CREDENTIAL_KEY_LEN]); + +/* Server side: recompute ClientSig' = HMAC(StoredKey, AuthMessage) and + * ClientKey' = proof XOR ClientSig', then accept iff v->found AND + * SHA256(ClientKey') equals StoredKey (constant-time over the 32-byte keys). + * Always computes server_sig_out = HMAC(ServerKey, AuthMessage). Returns the + * accept decision. */ +bool credentials_verify_response(const CredentialVerifier* v, const char* user, + const uint8_t* snonce, const uint8_t* cnonce, + const uint8_t proof[CREDENTIAL_KEY_LEN], + uint8_t server_sig_out[CREDENTIAL_KEY_LEN]); + +/* Derive a new-format store line for `user`/`password` and write it (without a + * trailing newline) into out. A random 16-byte salt is used. On failure err is + * filled. Used by --hash-credentials and by tests. */ +bool credentials_hash_store_line(const char* user, const char* password, uint32_t iters, char* out, + size_t out_sz, char* err, size_t err_size); + +/* Read `user:password` lines from `path` (the same no-group/other-bits check as + * the other secret files) and write one new-format store line per entry to + * `out`. + * Blank/comment lines are skipped; a malformed line fails the whole run. + * Returns 0 on success, -1 on error (err filled). Used by + * `--hash-credentials`. */ +int credentials_hash_file(const char* path, uint32_t iters, FILE* out, char* err, size_t err_size); + +/* Read the CLIENT-side secret file: the first meaningful line is + * `user:password` (the literal password). *user_out and *password_out are + * freshly allocated on success (password is plaintext -- the caller derives the + * proof and then burns/frees it); both are NULL on error. Returns 0 on + * success, -1 on failure (err filled: the path is named, never the credential + * itself). Only the line's trailing CR/LF are stripped: the password's bytes + * are otherwise preserved exactly, so a password with leading/trailing + * whitespace (after the ':') is kept usable. The username is trimmed of + * surrounding space/tabs. */ +int credentials_read_secret_file(const char* path, char** user_out, char** password_out, char* err, + size_t err_size); + +/* Constant-time equality over exactly len bytes. */ +bool credentials_secure_equal(const char* a, const char* b, size_t len); + +/* Overwrite secret[0..len) with zeros (best-effort wipe). */ +void credentials_burn(char* secret, size_t len); + +/* Number of entries currently in the store (tests/introspection). */ +int credentials_store_size(const CredentialStore* store); + +/* Whether the store contains an entry for `user` (tests/introspection). */ +bool credentials_store_has(const CredentialStore* store, const char* user); + +#endif diff --git a/src/shared/daemon_conf.c b/src/shared/daemon_conf.c new file mode 100644 index 0000000..e806549 --- /dev/null +++ b/src/shared/daemon_conf.c @@ -0,0 +1,454 @@ +#include "daemon_conf.h" +#include "utils.h" +#include +#include +#include +#include +#include +#include +#include + +/* ------------------------------------------------------------------ */ +/* helpers */ +/* ------------------------------------------------------------------ */ + +static void set_error(char* err, size_t err_size, const char* fmt, ...) { + if (!err || err_size == 0) + return; + va_list args; + va_start(args, fmt); + vsnprintf(err, err_size, fmt, args); + va_end(args); +} + +/* Trim leading and trailing ASCII space/tab in place; returns the new start. */ +static char* trim_ws(char* s) { + while (*s == ' ' || *s == '\t') + s++; + size_t len = strlen(s); + while (len > 0 && (s[len - 1] == ' ' || s[len - 1] == '\t')) + s[--len] = '\0'; + return s; +} + +/* Case-insensitive equality of a parsed key against a canonical key name. */ +static bool key_equals(const char* key, const char* canonical) { + return strcasecmp(key, canonical) == 0; +} + +static bool parse_bool_value(const char* value, bool* out) { + if (strcasecmp(value, "yes") == 0 || strcasecmp(value, "true") == 0 || strcmp(value, "1") == 0) { + *out = true; + return true; + } + if (strcasecmp(value, "no") == 0 || strcasecmp(value, "false") == 0 || strcmp(value, "0") == 0) { + *out = false; + return true; + } + return false; +} + +bool daemon_module_name_valid(const char* name) { + if (!name || *name == '\0') + return false; + size_t len = strlen(name); + if (len > DAEMON_MAX_MODULE_NAME) + return false; + for (size_t i = 0; i < len; i++) { + unsigned char c = (unsigned char)name[i]; + bool alnum = (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || (c >= '0' && c <= '9'); + if (!alnum && c != '.' && c != '_' && c != '-') + return false; + } + return true; +} + +DaemonConf* daemon_conf_create(void) { + DaemonConf* conf = calloc(1, sizeof(DaemonConf)); + if (!conf) + return NULL; + conf->global.port = DAEMON_CONF_DEFAULT_PORT; + return conf; +} + +void daemon_conf_free(DaemonConf* conf) { + if (!conf) + return; + free(conf->global.motd_file); + free(conf->global.address); + for (int i = 0; i < conf->module_count; i++) { + DaemonModule* m = &conf->modules[i]; + free(m->name); + free(m->path); + for (int j = 0; j < m->auth_user_count; j++) + free(m->auth_users[j]); + free(m->auth_users); + } + free(conf->modules); + free(conf); +} + +const DaemonModule* daemon_conf_find_module(const DaemonConf* conf, const char* name) { + if (!conf || !name) + return NULL; + for (int i = 0; i < conf->module_count; i++) { + if (strcmp(conf->modules[i].name, name) == 0) + return &conf->modules[i]; + } + return NULL; +} + +/* Replace *slot with a str_dup of value; returns false on allocation failure. */ +static bool store_string(char** slot, const char* value) { + char* dup = str_dup(value); + if (!dup) + return false; + free(*slot); + *slot = dup; + return true; +} + +static bool store_port(int* slot, const char* value, char* err, size_t err_size) { + char* end; + errno = 0; + long p = strtol(value, &end, 10); + if (errno != 0 || *end != '\0' || *value == '\0' || p <= 0 || p > 65535) { + set_error(err, err_size, "invalid port '%s' (must be 1-65535)", value); + return false; + } + *slot = (int)p; + return true; +} + +/* Apply a global scalar key/value. Keys are case-insensitive. Returns false + * (err filled) on an unknown key or an invalid value. */ +static bool apply_global_key(DaemonConf* conf, char* key, const char* value, char* err, + size_t err_size) { + if (key_equals(key, "port")) + return store_port(&conf->global.port, value, err, err_size); + if (key_equals(key, "motd file")) { + if (!store_string(&conf->global.motd_file, value)) { + set_error(err, err_size, "out of memory parsing 'motd file'"); + return false; + } + return true; + } + if (key_equals(key, "address")) { + if (!store_string(&conf->global.address, value)) { + set_error(err, err_size, "out of memory parsing 'address'"); + return false; + } + return true; + } + set_error(err, err_size, "unknown global key '%s'", key); + return false; +} + +/* Apply a module key/value to the currently-open module. Returns false (err + * filled) on an unknown module key or an invalid value. */ +static bool apply_module_key(DaemonModule* module, char* key, char* value, char* err, + size_t err_size) { + if (key_equals(key, "path")) { + if (*value == '\0') { + set_error(err, err_size, "module '%s': 'path' must not be empty", module->name); + return false; + } + if (!store_string(&module->path, value)) { + set_error(err, err_size, "out of memory parsing 'path' for module '%s'", module->name); + return false; + } + return true; + } + if (key_equals(key, "read only")) { + bool parsed; + if (!parse_bool_value(value, &parsed)) { + set_error(err, err_size, + "module '%s': 'read only' must be yes/no (or true/false/1/0), got '%s'", + module->name, value); + return false; + } + module->read_only = parsed; + return true; + } + if (key_equals(key, "client owner")) { + bool parsed; + if (!parse_bool_value(value, &parsed)) { + set_error(err, err_size, + "module '%s': 'client owner' must be yes/no (or true/false/1/0), got '%s'", + module->name, value); + return false; + } + module->client_owner = parsed; + return true; + } + if (key_equals(key, "auth users")) { + char* list = str_dup(value); + if (!list) { + set_error(err, err_size, "out of memory parsing 'auth users' for module '%s'", module->name); + return false; + } + char* save = NULL; + for (char* token = strtok_r(list, ",", &save); token; token = strtok_r(NULL, ",", &save)) { + const char* user = trim_ws(token); + if (*user == '\0') + continue; + char** grown = + realloc(module->auth_users, (size_t)(module->auth_user_count + 1) * sizeof(char*)); + if (!grown) { + free(list); + set_error(err, err_size, "out of memory parsing 'auth users' for module '%s'", + module->name); + return false; + } + module->auth_users = grown; + char* dup = str_dup(user); + if (!dup) { + free(list); + set_error(err, err_size, "out of memory parsing 'auth users' for module '%s'", + module->name); + return false; + } + module->auth_users[module->auth_user_count++] = dup; + } + free(list); + return true; + } + set_error(err, err_size, "unknown key '%s' in module '%s'", key, module->name); + return false; +} + +static bool module_open_valid(const DaemonModule* module, char* err, size_t err_size) { + if (module->path == NULL) { + set_error(err, err_size, "module '%s' has no 'path'", module->name); + return false; + } + return true; +} + +/* Validate a [section] header line body (text between the brackets) and set + * *name to the module name. Returns false on a malformed header. */ +static bool parse_section_name(char* body, const char** name_out, char* err, size_t err_size) { + char* name = trim_ws(body); + if (!daemon_module_name_valid(name)) { + set_error(err, err_size, "invalid module name '%s' (must be 1-%d chars of [A-Za-z0-9._-])", + name, DAEMON_MAX_MODULE_NAME); + return false; + } + *name_out = name; + return true; +} + +/* Open (or switch to) a module section. Closes any previously open module + * (validating it has a path) and appends the new one. */ +static int open_module(DaemonConf* conf, int* current_module, const char* name, char* err, + size_t err_size) { + if (*current_module >= 0) { + if (!module_open_valid(&conf->modules[*current_module], err, err_size)) + return -1; + } + if (daemon_conf_find_module(conf, name)) { + set_error(err, err_size, "duplicate module '%s'", name); + return -1; + } + DaemonModule* grown = + realloc(conf->modules, (size_t)(conf->module_count + 1) * sizeof(DaemonModule)); + if (!grown) { + set_error(err, err_size, "out of memory adding module '%s'", name); + return -1; + } + conf->modules = grown; + memset(&conf->modules[conf->module_count], 0, sizeof(DaemonModule)); + conf->modules[conf->module_count].name = str_dup(name); + if (!conf->modules[conf->module_count].name) { + set_error(err, err_size, "out of memory adding module '%s'", name); + return -1; + } + conf->module_count++; + *current_module = conf->module_count - 1; + return 0; +} + +/* Split a "key = value" line (value pointer returned in *value, pointing into + * line). Returns false when there is no '='. */ +static bool split_key_value(char* line, char** key, char** value) { + char* eq = strchr(line, '='); + if (!eq) + return false; + *eq = '\0'; + *key = trim_ws(line); + *value = trim_ws(eq + 1); + return true; +} + +/* Strip one layer of surrounding double quotes from a trimmed value. A value + * that starts with '"' but does not end with '"' is an error. */ +static bool unquote_value(char* value, char* err, size_t err_size) { + size_t len = strlen(value); + if (len == 0 || value[0] != '"') + return true; + if (len < 2 || value[len - 1] != '"') { + set_error(err, err_size, "unterminated quoted value"); + return false; + } + memmove(value, value + 1, len - 2); + value[len - 2] = '\0'; + return true; +} + +DaemonConf* daemon_conf_load(const char* path, char* err, size_t err_size) { + if (err && err_size) + err[0] = '\0'; + if (!path) { + set_error(err, err_size, "no daemon config path"); + return NULL; + } + FILE* fp = fopen(path, "r"); + if (!fp) { + set_error(err, err_size, "cannot open daemon config '%s': %s", path, strerror(errno)); + return NULL; + } + + DaemonConf* conf = daemon_conf_create(); + if (!conf) { + fclose(fp); + set_error(err, err_size, "out of memory allocating daemon config"); + return NULL; + } + + int current_module = -1; + int line_no = 0; + char line[DAEMON_CONF_MAX_LINE + 2]; + bool ok = true; + + while (ok && fgets(line, sizeof(line), fp)) { + line_no++; + size_t len = strlen(line); + if (len == DAEMON_CONF_MAX_LINE + 1 && line[len - 1] != '\n') { + /* The read stopped at the buffer edge without a newline and there is + * more file to come: the line exceeds the bound. */ + if (!feof(fp)) { + set_error(err, err_size, "line %d exceeds the %d-byte limit", line_no, + DAEMON_CONF_MAX_LINE); + ok = false; + break; + } + } + if (len > 0 && line[len - 1] == '\n') + line[--len] = '\0'; + if (len > 0 && line[len - 1] == '\r') + line[--len] = '\0'; + + char* cursor = line; + while (*cursor == ' ' || *cursor == '\t') + cursor++; + if (*cursor == '\0' || *cursor == '#' || *cursor == ';') + continue; /* blank or comment line */ + + if (*cursor == '[') { + char* close = strchr(cursor, ']'); + if (!close) { + set_error(err, err_size, "line %d: unterminated module header", line_no); + ok = false; + break; + } + *close = '\0'; + char* trailing = close + 1; + const char* rest = trim_ws(trailing); + if (*rest != '\0') { + set_error(err, err_size, "line %d: unexpected text after module header", line_no); + ok = false; + break; + } + const char* name = NULL; + if (!parse_section_name(cursor + 1, &name, err, err_size)) { + ok = false; + break; + } + if (open_module(conf, ¤t_module, name, err, err_size) != 0) { + ok = false; + break; + } + continue; + } + + char* key; + char* value; + if (!split_key_value(cursor, &key, &value)) { + set_error(err, err_size, "line %d: expected 'key = value'", line_no); + ok = false; + break; + } + if (*key == '\0') { + set_error(err, err_size, "line %d: empty key", line_no); + ok = false; + break; + } + if (!unquote_value(value, err, err_size)) { + ok = false; + break; + } + if (current_module >= 0) { + if (!apply_module_key(&conf->modules[current_module], key, value, err, err_size)) { + ok = false; + break; + } + } else { + if (!apply_global_key(conf, key, value, err, err_size)) { + ok = false; + break; + } + } + } + + if (ok && ferror(fp)) { + set_error(err, err_size, "error reading daemon config '%s': %s", path, strerror(errno)); + ok = false; + } + fclose(fp); + + if (ok && current_module >= 0 && + !module_open_valid(&conf->modules[current_module], err, err_size)) { + ok = false; + } + if (!ok) { + daemon_conf_free(conf); + return NULL; + } + return conf; +} + +int daemon_conf_apply_dparam(DaemonConf* conf, const char* assignment, char* err, size_t err_size) { + if (err && err_size) + err[0] = '\0'; + if (!conf || !assignment || *assignment == '\0') { + set_error(err, err_size, "--dparam requires a KEY=VALUE override"); + return -1; + } + char* copy = str_dup(assignment); + if (!copy) { + set_error(err, err_size, "out of memory parsing --dparam"); + return -1; + } + char* eq = strchr(copy, '='); + if (!eq) { + free(copy); + set_error(err, err_size, "--dparam '%s' has no '=' (expected KEY=VALUE)", assignment); + return -1; + } + *eq = '\0'; + char* key = trim_ws(copy); + const char* value = trim_ws(eq + 1); + if (*key == '\0') { + free(copy); + set_error(err, err_size, "--dparam '%s' has an empty key", assignment); + return -1; + } + if (*value == '\0') { + free(copy); + set_error(err, err_size, "--dparam '%s' has an empty value", assignment); + return -1; + } + bool ok = apply_global_key(conf, key, value, err, err_size); + free(copy); + return ok ? 0 : -1; +} diff --git a/src/shared/daemon_conf.h b/src/shared/daemon_conf.h new file mode 100644 index 0000000..0ecfe24 --- /dev/null +++ b/src/shared/daemon_conf.h @@ -0,0 +1,107 @@ +#ifndef DAEMON_CONF_H +#define DAEMON_CONF_H + +#include +#include + +/* FastSync-native daemon configuration (a FastSync analog of rsyncd.conf). + * + * This is the config the fastsync-server --daemon listener consumes. It is + * line-based with an implicit global section followed by zero or more + * [module] sections. The full grammar is documented in RSYNC_COMPAT.md + * ("Daemon Mode") and summarized below; the parser lives entirely in + * daemon_conf.c so it can be unit tested without any socket code. + * + * The parser is STRICT: an unknown key, a malformed line, a value that does + * not parse, a module without a `path`, or a line longer than + * DAEMON_CONF_MAX_LINE all fail the whole load with a clear, line-numbered + * error instead of being silently ignored. This keeps a typo from silently + * changing what a module serves. + */ + +/* A daemon module's configured root is used exactly like the standalone + * server's --destination-root: the daemon confines every connection that + * selects this module to this path (file_open_secure_parent / + * has_path_traversal / path_is_within all keep the existing confinement, just + * per-module). There is never any client-chosen root: a module path always + * stays confined. A daemon REFUSES every client-chosen ownership / super-user + * request by default -- --numeric-ids, --chown, --usermap/--groupmap, + * --fake-super, --copy-as and an explicit --super -- because there is no + * per-module opt-in unless the operator adds one. An operator opts a single + * module in with `client owner = yes` (DaemonModule.client_owner), which allows + * that client to choose ownership within that module's root (the standalone/SSH + * server honors such requests for its single operator-authorized root). The + * operator-level --no-super veto additionally forces super-user activities off + * for every daemon connection, even an opted-in module. See server_module_gate + * in server.c and RSYNC_COMPAT.md. + * + * `auth_users` is honored by Wave B daemon authentication: a module that + * declares auth users accepts a connection only when the presented username is + * on this list AND verifies against the daemon's credential store + * (--password-file / --early-input). An auth-required module with no usable + * store refuses (fail closed) rather than falling open; see server.c. Auth is + * never bypassed by ignoring the list. */ +typedef struct DaemonModule { + char* name; /* module name, as the client requests it */ + char* path; /* module root (daemon-side authorized root) */ + bool read_only; /* `read only = yes/no`; default no */ + bool client_owner; /* `client owner = yes/no`; default no. Per-module opt-in + that lets this module's clients choose ownership + (--numeric-ids/--chown/--usermap/--groupmap/--fake-super/ + --copy-as) and request explicit --super super-user + activities. Without it the daemon refuses all of them. */ + char** auth_users; /* `auth users = a,b`; Wave B credential list */ + int auth_user_count; +} DaemonModule; + +/* Global (pre-module) scalar keys. `motd file` is parsed and stored but has + * no wire effect yet (MOTD display is Wave C). */ +typedef struct DaemonConfGlobals { + int port; /* `port`, default DAEMON_CONF_DEFAULT_PORT (873) */ + char* motd_file; /* `motd file`, may be NULL */ + char* address; /* `address` (optional bind address), may be NULL */ +} DaemonConfGlobals; + +typedef struct DaemonConf { + DaemonConfGlobals global; + DaemonModule* modules; + int module_count; +} DaemonConf; + +#define DAEMON_CONF_DEFAULT_PORT 873 +/* Longest accepted config line (excluding the trailing newline). Longer lines + * are rejected rather than buffered unboundedly. */ +#define DAEMON_CONF_MAX_LINE 4096 +/* Upper bound on a module name. Kept far below MAX_STRING_SIZE so a wire + * module name can never exhaust anything by being long. */ +#define DAEMON_MAX_MODULE_NAME 200 + +/* Allocate an empty daemon config with defaulted globals (port 873, no + * modules, no motd/address). Never fails for an allocation failure; callers + * must still NULL-check. */ +DaemonConf* daemon_conf_create(void); + +/* Parse `path` into a freshly allocated DaemonConf. Returns NULL on any error + * and fills `err` (err_size bytes) with a clear, line-numbered message. The + * returned object is heap-owned; free it with daemon_conf_free. */ +DaemonConf* daemon_conf_load(const char* path, char* err, size_t err_size); + +void daemon_conf_free(DaemonConf* conf); + +/* Case-sensitive exact module lookup by name. Returns the module or NULL. + * Module names are matched exactly (rsync semantics). */ +const DaemonModule* daemon_conf_find_module(const DaemonConf* conf, const char* name); + +/* Module-name syntax check: non-empty, at most DAEMON_MAX_MODULE_NAME chars, + * and only [A-Za-z0-9._-]. Used by the config parser, the client's + * host::module/path destination parser, and (implicitly) by the daemon lookup + * (a name that fails this can never match a parsed module). */ +bool daemon_module_name_valid(const char* name); + +/* Parse one --dparam=KEY=VALUE (or "--dparam KEY=VALUE") override string and + * apply it to the global scalars only. Keys are case-insensitive and limited + * to the global scalar keys defined by the grammar (port, motd file, address). + * Returns 0 on success, -1 on error (err filled). */ +int daemon_conf_apply_dparam(DaemonConf* conf, const char* assignment, char* err, size_t err_size); + +#endif diff --git a/src/shared/data.c b/src/shared/data.c index 82e6198..55af503 100644 --- a/src/shared/data.c +++ b/src/shared/data.c @@ -1,11 +1,12 @@ #include "data.h" #include "log.h" +#include "protocol.h" #include Data* data_create_empty(size_t data_size) { /* malloc(0) is UB; allocate at least 1 byte but preserve requested size */ size_t alloc_size = data_size > 0 ? data_size : 1; - void* data = malloc(alloc_size); + void* data = protocol_alloc(alloc_size); if (data == NULL) { log_message(LOG_LEVEL_ERROR, "Could not allocate memory for empty data"); return NULL; @@ -14,18 +15,19 @@ Data* data_create_empty(size_t data_size) { } Data* data_create_reserve(size_t size) { - Data* d = malloc(sizeof(Data)); + Data* d = protocol_alloc(sizeof(Data)); if (d == NULL) { log_message(LOG_LEVEL_ERROR, "Could not allocate memory for data"); return NULL; } d->data = NULL; d->size = size; + d->protocol_charge = 0; return d; } Data* data_create(void* data, size_t data_size) { - Data* new_data = malloc(sizeof(Data)); + Data* new_data = protocol_alloc(sizeof(Data)); if (new_data == NULL) { log_message(LOG_LEVEL_ERROR, "Could not allocate memory for data"); free(data); @@ -33,12 +35,15 @@ Data* data_create(void* data, size_t data_size) { } new_data->data = data; new_data->size = data_size; + new_data->protocol_charge = 0; return new_data; } void data_destroy(Data* data) { if (data == NULL) return; + if (data->protocol_charge != 0) + protocol_release_memory(data->protocol_charge); free(data->data); free(data); } diff --git a/src/shared/data.h b/src/shared/data.h index f03fabf..b65ae29 100644 --- a/src/shared/data.h +++ b/src/shared/data.h @@ -6,11 +6,14 @@ typedef struct { void* data; size_t size; + /* Non-zero only for a buffer charged to the protocol connection budget. */ + size_t protocol_charge; } Data; Data* data_create_empty(size_t data_size); Data* data_create_reserve(size_t size); Data* data_create(void* data, size_t data_size); void data_destroy(Data* data); +void protocol_release_memory(size_t charge); #endif diff --git a/src/shared/delay_updates.c b/src/shared/delay_updates.c new file mode 100644 index 0000000..c867023 --- /dev/null +++ b/src/shared/delay_updates.c @@ -0,0 +1,338 @@ +#include "delay_updates.h" + +#include "config.h" +#include "file.h" +#include "log.h" +#include "utils.h" +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +DelayUpdatesContext* delay_updates_context_create(const char* root_directory) { + if (!root_directory) + return NULL; + DelayUpdatesContext* context = calloc(1, sizeof(DelayUpdatesContext)); + if (!context) + return NULL; + context->root_directory = str_dup(root_directory); + if (!context->root_directory) { + free(context); + return NULL; + } + context->staging_root = path_cat(root_directory, DELAY_UPDATES_STAGING_DIR); + if (!context->staging_root) { + free(context->root_directory); + free(context); + return NULL; + } + context->entries = NULL; + context->count = 0; + context->capacity = 0; + context->prepared = false; + context->lock_fd = -1; + if (mtx_init(&context->mutex, mtx_plain) != thrd_success) { + free(context->staging_root); + free(context->root_directory); + free(context); + return NULL; + } + return context; +} + +void delay_updates_context_destroy(DelayUpdatesContext* context) { + if (!context) + return; + mtx_destroy(&context->mutex); + if (context->lock_fd >= 0) + close(context->lock_fd); + context->lock_fd = -1; + free(context->staging_root); + free(context->root_directory); + for (size_t i = 0; i < context->count; i++) { + free(context->entries[i].staged_path); + free(context->entries[i].final_path); + free(context->entries[i].file_path); + } + free(context->entries); + free(context); +} + +bool delay_updates_staging_name_conflict(const char* dir) { + if (!dir || !*dir) + return false; + size_t length = strlen(dir); + while (length > 0 && dir[length - 1] == '/') + length--; + size_t reserved_length = strlen(DELAY_UPDATES_STAGING_DIR); + if (length != reserved_length) + return false; + return strncmp(dir, DELAY_UPDATES_STAGING_DIR, length) == 0; +} + +/* Recursively delete every entry inside an open directory (never following + symlinks). The directory itself is left in place. Mirrors the fd-relative + walk used by the delete code so a symlink planted inside the staging tree + can never redirect removal outside of it. */ +static bool delay_wipe_dir_fd(int dirfd) { + int scanfd = dup(dirfd); + if (scanfd < 0) + return false; + DIR* dir = fdopendir(scanfd); + if (!dir) { + close(scanfd); + return false; + } + bool operation_ok = true; + const struct dirent* entry; + while ((entry = readdir(dir)) != NULL) { + if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) + continue; + struct stat st; + if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) { + if (errno != ENOENT) + operation_ok = false; + continue; + } + if (S_ISDIR(st.st_mode)) { + int childfd = openat(dirfd, entry->d_name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + bool child_removed = false; + if (childfd >= 0) { + child_removed = delay_wipe_dir_fd(childfd); + close(childfd); + } else if (errno != ENOENT) { + operation_ok = false; + } + if (child_removed && unlinkat(dirfd, entry->d_name, AT_REMOVEDIR) != 0 && errno != ENOENT) + operation_ok = false; + } else { + if (unlinkat(dirfd, entry->d_name, 0) != 0 && errno != ENOENT) + operation_ok = false; + } + } + closedir(dir); + return operation_ok; +} + +bool delay_updates_prepare(DelayUpdatesContext* context) { + if (!context) + return false; + if (context->prepared) + return true; + int fd = file_open_private_dir(context->staging_root); + if (fd < 0) { + int saved_errno = errno; + char* escaped = output_escape(context->staging_root, false); + log_message(LOG_LEVEL_ERROR, "could not create --delay-updates staging directory '%s': %s", + escaped ? escaped : "", strerror(saved_errno)); + free(escaped); + return false; + } + /* Hold an exclusive advisory lock on the staging directory for the whole + transfer. The staging directory name is fixed, so two simultaneous + delayed transfers to the same destination root would otherwise share it + and destroy each other's staged files. The lock makes the second session + fail cleanly instead of corrupting the first. The lock is released when + the context (and its file descriptor) is destroyed. */ + if (flock(fd, LOCK_EX | LOCK_NB) != 0) { + int saved_errno = errno; + close(fd); + if (saved_errno == EWOULDBLOCK || saved_errno == EAGAIN) { + char* escaped = output_escape(context->staging_root, false); + log_message(LOG_LEVEL_ERROR, + "another --delay-updates transfer to '%s' is already in progress; refusing to " + "share the staging directory", + escaped ? escaped : ""); + free(escaped); + } else { + log_message(LOG_LEVEL_ERROR, "could not lock --delay-updates staging directory '%s': %s", + context->staging_root, strerror(saved_errno)); + } + return false; + } + context->lock_fd = fd; + /* Only now, with exclusive ownership, wipe leftovers from an interrupted + earlier transfer; this can never race with a live session. */ + bool ok = delay_wipe_dir_fd(fd); + if (!ok) { + log_message(LOG_LEVEL_ERROR, "could not clear stale --delay-updates staging files under '%s'", + context->staging_root); + close(context->lock_fd); + context->lock_fd = -1; + return false; + } + context->prepared = true; + return true; +} + +bool delay_updates_record(DelayUpdatesContext* context, const char* staged_path, + const char* final_path, const char* file_path) { + if (!context || !staged_path || !final_path || !file_path) + return false; + char* staged_copy = str_dup(staged_path); + char* final_copy = str_dup(final_path); + char* file_copy = str_dup(file_path); + if (!staged_copy || !final_copy || !file_copy) { + free(staged_copy); + free(final_copy); + free(file_copy); + return false; + } + mtx_lock(&context->mutex); + bool ok = true; + if (context->count == context->capacity) { + size_t new_capacity = context->capacity == 0 ? 64 : context->capacity * 2; + if (new_capacity < context->capacity) { + ok = false; + } else { + StagedFileEntry* grown = realloc(context->entries, new_capacity * sizeof(StagedFileEntry)); + if (!grown) { + ok = false; + } else { + context->entries = grown; + context->capacity = new_capacity; + } + } + } + if (ok) { + context->entries[context->count].staged_path = staged_copy; + context->entries[context->count].final_path = final_copy; + context->entries[context->count].file_path = file_copy; + context->count++; + } + mtx_unlock(&context->mutex); + if (!ok) { + free(staged_copy); + free(final_copy); + free(file_copy); + } + return ok; +} + +/* Move an existing final destination file aside before the staged replacement + is installed. Deferred from stage time so the final destination is not + modified until publication. Mirrors the immediate-mode backup logic. */ +static bool delay_publish_backup(const DelayUpdatesContext* context, const Config* config, + const StagedFileEntry* entry) { + bool backup_enabled = config && config->backup && !config->ignore_existing; + if (!backup_enabled) + return true; + const char* backup_suffix = (config && config->suffix) ? config->suffix : "~"; + struct stat backup_stat; + if (!file_stat_secure(entry->final_path, &backup_stat)) + return true; /* nothing to back up */ + + char* backup_path = NULL; + if (config->backup_dir) { + char* confined_backup = path_cat(context->root_directory, config->backup_dir); + if (!confined_backup) + return false; + backup_path = path_cat(confined_backup, entry->file_path); + free(confined_backup); + } else { + size_t path_len = strlen(entry->final_path); + size_t suffix_len = strlen(backup_suffix); + if (path_len > SIZE_MAX - suffix_len - 1) + return false; + backup_path = malloc(path_len + suffix_len + 1); + if (backup_path) { + memcpy(backup_path, entry->final_path, path_len); + memcpy(backup_path + path_len, backup_suffix, suffix_len + 1); + } + } + if (!backup_path) + return false; + char* parent_copy = str_dup(backup_path); + if (!parent_copy || !file_ensure_directory_secure(dirname(parent_copy))) { + free(parent_copy); + free(backup_path); + return false; + } + free(parent_copy); + bool ok = file_rename_secure(entry->final_path, backup_path); + free(backup_path); + return ok; +} + +static bool delay_publish_entry(DelayUpdatesContext* context, const Config* config, + const StagedFileEntry* entry) { + if (!delay_publish_backup(context, config, entry)) + return false; + if (!file_rename_secure(entry->staged_path, entry->final_path)) { + if (errno == EXDEV) { + char* escaped = output_escape(entry->final_path, false); + log_message(LOG_LEVEL_ERROR, + "staging directory is on a different filesystem than the destination; cannot " + "atomically install file (EXDEV): %s", + escaped ? escaped : ""); + free(escaped); + } else { + char* escaped = output_escape(entry->final_path, false); + log_message(LOG_LEVEL_ERROR, "could not install staged file '%s': %s", + escaped ? escaped : "", strerror(errno)); + free(escaped); + } + return false; + } + return true; +} + +/* Remove the staging tree (contents plus the directory itself). Returns true + when nothing is left behind (including the case where it never existed). */ +static bool delay_updates_remove_staging_tree(DelayUpdatesContext* context) { + int fd = open(context->staging_root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + if (fd < 0) + return errno == ENOENT; + bool ok = delay_wipe_dir_fd(fd); + if (close(fd) != 0) + ok = false; + if (ok && rmdir(context->staging_root) != 0 && errno != ENOENT) + ok = false; + return ok; +} + +bool delay_updates_publish(DelayUpdatesContext* context, const Config* config) { + if (!context) + return false; + mtx_lock(&context->mutex); + bool ok = true; + for (size_t i = 0; i < context->count; i++) { + if (!delay_publish_entry(context, config, &context->entries[i])) { + ok = false; + break; + } + } + mtx_unlock(&context->mutex); + + /* Renaming files out of the staging tree leaves the mirrored directories + behind, and a mid-publish failure leaves the remaining staged files. + Remove whatever is left so a later run starts from a clean staging area + and no staged content can linger after a failed publish. If that cleanup + fails, tell the operator: a stale staging directory would otherwise + silently accumulate and make the next transfer's prepare-wipe fail. */ + if (!delay_updates_remove_staging_tree(context)) { + log_message(LOG_LEVEL_WARNING, + "could not fully remove --delay-updates staging directory '%s' after publish; a " + "later --delay-updates transfer to this destination will try to clear it", + context->staging_root); + } + return ok; +} + +void delay_updates_cleanup(DelayUpdatesContext* context) { + if (!context) + return; + /* Only a context that gained exclusive ownership may touch the shared + staging directory. If prepare never succeeded (e.g. lock contention with + another live session) the directory belongs to that other session and must + be left alone. */ + if (!context->prepared) + return; + delay_updates_remove_staging_tree(context); +} diff --git a/src/shared/delay_updates.h b/src/shared/delay_updates.h new file mode 100644 index 0000000..225c19b --- /dev/null +++ b/src/shared/delay_updates.h @@ -0,0 +1,68 @@ +#ifndef DELAY_UPDATES_H +#define DELAY_UPDATES_H + +#include +#include +#include + +/* Forward-declared in config.h; full type needed by file_save_to_disk. */ +typedef struct Config Config; + +/* One staged file awaiting publication. */ +typedef struct { + char* staged_path; /* full path inside the staging tree */ + char* final_path; /* full final destination path */ + char* file_path; /* the file path as received on the wire */ +} StagedFileEntry; + +/* Receiver-side --delay-updates staging registry. All successfully written + files land under a private staging directory inside the receive root and are + atomically renamed into their final destination only at the very end of the + transfer. A single PipelineContextReceiver has exactly one writer thread, + but the registry is still mutex-protected so the same object can be safely + shared with the publish/cleanup phase that runs after the threads join. */ +typedef struct DelayUpdatesContext { + char* root_directory; /* receive root the staging dir lives under */ + char* staging_root; /* root_directory/ */ + mtx_t mutex; + StagedFileEntry* entries; + size_t count; + size_t capacity; + bool prepared; /* staging dir created, wiped, and exclusively locked */ + int lock_fd; /* advisory exclusive flock held on the staging dir, or -1 */ +} DelayUpdatesContext; + +/* Name of the private staging subdirectory created under the receive root. */ +#define DELAY_UPDATES_STAGING_DIR ".fastsync-stage" + +/* True when `dir` (ignoring a trailing "/") is the reserved staging directory + name. Used to reject a --backup-dir that would collide with the internal + staging area. */ +bool delay_updates_staging_name_conflict(const char* dir); + +/* Create an empty staging context rooted below root_directory. Does not touch + the filesystem yet. */ +DelayUpdatesContext* delay_updates_context_create(const char* root_directory); +void delay_updates_context_destroy(DelayUpdatesContext* context); + +/* Create the private 0700 staging directory (on first call) and wipe any + leftovers from a previously interrupted delayed transfer. Idempotent. */ +bool delay_updates_prepare(DelayUpdatesContext* context); + +/* Record a fully-written staged file for later publication. Copies all three + paths. Returns false on allocation failure. */ +bool delay_updates_record(DelayUpdatesContext* context, const char* staged_path, + const char* final_path, const char* file_path); + +/* Atomically rename every staged file into its final destination. Deferred + --backup handling runs immediately before each rename. On any failure the + remaining staged files are removed (best effort); already-published files + are not rolled back. Afterwards the staging tree is removed so a successful + or failed publish leaves no staging leftovers. */ +bool delay_updates_publish(DelayUpdatesContext* context, const Config* config); + +/* Best-effort removal of every staged file and the staging directory itself. + Safe to call when nothing was staged or after a successful publish. */ +void delay_updates_cleanup(DelayUpdatesContext* context); + +#endif diff --git a/src/shared/delta.c b/src/shared/delta.c index 0305fdc..dc5ea8a 100644 --- a/src/shared/delta.c +++ b/src/shared/delta.c @@ -1,6 +1,8 @@ #include "delta.h" #include "log.h" +#include "protocol.h" #include +#include #include #include @@ -27,21 +29,42 @@ uint32_t delta_xxhash32(const void* data, uint32_t len) { return XXH32(data, len, 0); } +uint32_t delta_xxhash32_seeded(const void* data, uint32_t len, uint32_t seed) { + return XXH32(data, len, seed); +} + +uint64_t delta_xxhash64(const void* data, size_t len) { + return XXH64(data, len, 0); +} + DeltaSignature* delta_signature_create(const void* old_file_data, uint64_t old_file_size, uint32_t block_size) { + return delta_signature_create_seeded(old_file_data, old_file_size, block_size, 0); +} + +DeltaSignature* delta_signature_create_seeded(const void* old_file_data, uint64_t old_file_size, + uint32_t block_size, uint32_t seed) { if (old_file_data == NULL || old_file_size == 0 || block_size == 0) return NULL; + if (old_file_size > DELTA_MAX_FILE_SIZE || block_size > DELTA_BLOCK_SIZE_MAX || + old_file_size > UINT32_MAX * (uint64_t)block_size) + return NULL; + uint32_t block_count = (uint32_t)((old_file_size + block_size - 1) / block_size); - DeltaSignature* sig = malloc(sizeof(DeltaSignature)); + DeltaSignature* sig = protocol_alloc(sizeof(DeltaSignature)); if (!sig) return NULL; sig->file_size = old_file_size; sig->block_size = block_size; sig->block_count = block_count; - sig->blocks = malloc(block_count * sizeof(DeltaBlockSig)); + if (block_count == 0) { + free(sig); + return NULL; + } + sig->blocks = protocol_alloc((size_t)block_count * sizeof(DeltaBlockSig)); if (!sig->blocks) { free(sig); return NULL; @@ -53,7 +76,7 @@ DeltaSignature* delta_signature_create(const void* old_file_data, uint64_t old_f uint32_t len = (uint32_t)((old_file_size - offset < block_size) ? (old_file_size - offset) : block_size); sig->blocks[i].adler32 = delta_adler32(data + offset, len); - sig->blocks[i].xxhash = delta_xxhash32(data + offset, len); + sig->blocks[i].xxhash = delta_xxhash32_seeded(data + offset, len, seed); } return sig; @@ -63,10 +86,13 @@ Data* delta_signature_serialize(const DeltaSignature* sig) { if (!sig) return NULL; - uint64_t total = sizeof(uint64_t) + sizeof(uint32_t) + sizeof(uint32_t) + - (uint64_t)sig->block_count * (sizeof(uint32_t) + sizeof(uint32_t)); + uint64_t block_bytes = (uint64_t)sig->block_count * (sizeof(uint32_t) + sizeof(uint32_t)); + uint64_t total = sizeof(uint64_t) + sizeof(uint32_t) + sizeof(uint32_t) + block_bytes; + if (block_bytes > UINT64_MAX - (sizeof(uint64_t) + sizeof(uint32_t) + sizeof(uint32_t)) || + total > SIZE_MAX) + return NULL; - uint8_t* buf = malloc((size_t)total); + uint8_t* buf = protocol_alloc((size_t)total); if (!buf) return NULL; @@ -95,7 +121,7 @@ DeltaSignature* delta_signature_deserialize(const Data* data) { const uint8_t* buf = (const uint8_t*)data->data; size_t pos = 0; - DeltaSignature* sig = malloc(sizeof(DeltaSignature)); + DeltaSignature* sig = protocol_alloc(sizeof(DeltaSignature)); if (!sig) return NULL; @@ -114,6 +140,13 @@ DeltaSignature* delta_signature_deserialize(const Data* data) { return NULL; } + if (sig->block_size == 0 || sig->block_size > DELTA_BLOCK_SIZE_MAX || + sig->file_size > DELTA_MAX_FILE_SIZE || sig->file_size == 0 || + (sig->file_size + sig->block_size - 1) / sig->block_size != sig->block_count) { + free(sig); + return NULL; + } + uint64_t expected = sizeof(uint64_t) + sizeof(uint32_t) + sizeof(uint32_t) + (uint64_t)sig->block_count * (sizeof(uint32_t) + sizeof(uint32_t)); if (data->size < expected) { @@ -126,7 +159,7 @@ DeltaSignature* delta_signature_deserialize(const Data* data) { free(sig); return NULL; } - sig->blocks = malloc((size_t)blocks_size); + sig->blocks = protocol_alloc((size_t)blocks_size); if (!sig->blocks) { free(sig); return NULL; @@ -152,8 +185,10 @@ void delta_signature_destroy(DeltaSignature* sig) { static bool ensure_capacity(DeltaInstruction** instrs, uint32_t* capacity, uint32_t count) { if (count < *capacity) return true; + if (*capacity > MAX_DELTA_INSTRUCTIONS / 2) + return false; uint32_t new_cap = *capacity * 2; - DeltaInstruction* tmp = realloc(*instrs, new_cap * sizeof(DeltaInstruction)); + DeltaInstruction* tmp = protocol_realloc(*instrs, (size_t)new_cap * sizeof(DeltaInstruction)); if (!tmp) return false; *instrs = tmp; @@ -165,10 +200,12 @@ static bool flush_literal(DeltaInstruction** instrs, uint32_t* capacity, uint32_ const uint8_t* data, uint64_t start, uint64_t end) { if (start >= end) return true; + if (end - start > UINT32_MAX || *count >= MAX_DELTA_INSTRUCTIONS) + return false; uint32_t lit_len = (uint32_t)(end - start); if (!ensure_capacity(instrs, capacity, *count)) return false; - uint8_t* lit_data = malloc(lit_len); + uint8_t* lit_data = protocol_alloc(lit_len); if (!lit_data) return false; memcpy(lit_data, data + start, lit_len); @@ -179,19 +216,166 @@ static bool flush_literal(DeltaInstruction** instrs, uint32_t* capacity, uint32_ return true; } +static void free_instructions(DeltaInstruction* instrs, uint32_t count) { + if (!instrs) + return; + for (uint32_t i = 0; i < count; i++) + if (instrs[i].type == DELTA_INSTR_LITERAL) + free(instrs[i].literal.data); + free(instrs); +} + +/* Sentinel meaning "no signature block" in the lookup index chains. Block + * counts are bounded well below UINT32_MAX, so it doubles as a null link. */ +#define DELTA_NO_BLOCK UINT32_MAX + +/* Avalanche mix for the rolling checksum so blocks do not cluster in the + * bucket table when the weak checksum has little entropy (e.g. all-zero or + * patterned files). */ +static uint32_t delta_adler_mix(uint32_t h) { + h ^= h >> 16; + h *= 0x7feb352dU; + h ^= h >> 15; + h *= 0x846ca68bU; + h ^= h >> 16; + return h; +} + +/* Smallest power of two >= v. v must be non-zero. */ +static uint32_t delta_next_pow2(uint32_t v) { + v--; + v |= v >> 1; + v |= v >> 2; + v |= v >> 4; + v |= v >> 8; + v |= v >> 16; + return v + 1; +} + +/* Build a hash index over sig->blocks keyed by the (mixed) rolling checksum. + * All blocks sharing an Adler-32 value land in the same bucket; collisions + * are chained through a single contiguous allocation: + * + * [0, bucket_count) heads (first block per bucket) + * [bucket_count, 2*bucket_count) tails (last block per bucket) + * [2*bucket_count, ...) per-block chain links + * + * Blocks are inserted in ascending index order so every bucket chain is + * ordered exactly like the historical linear scan. Returns the base pointer + * (also the heads array) or NULL when no index could be allocated; callers + * then fall back to the linear scan. */ +static uint32_t* delta_build_index(const DeltaSignature* sig, uint32_t bucket_count) { + if (sig->block_count == 0 || bucket_count == 0) + return NULL; + + size_t entries = (size_t)2 * bucket_count + sig->block_count; + if (entries > SIZE_MAX / sizeof(uint32_t)) + return NULL; + + uint32_t* index = protocol_alloc(entries * sizeof(uint32_t)); + if (!index) + return NULL; + + uint32_t* heads = index; + uint32_t* tails = index + bucket_count; + uint32_t* next = index + 2 * bucket_count; + uint32_t mask = bucket_count - 1; + + memset(heads, 0xFF, (size_t)bucket_count * sizeof(uint32_t)); + memset(tails, 0xFF, (size_t)bucket_count * sizeof(uint32_t)); + + for (uint32_t j = 0; j < sig->block_count; j++) { + uint32_t b = delta_adler_mix(sig->blocks[j].adler32) & mask; + if (heads[b] == DELTA_NO_BLOCK) + heads[b] = j; + else + next[tails[b]] = j; + tails[b] = j; + next[j] = DELTA_NO_BLOCK; + } + return index; +} + +/* Locate the signature block matching the byte window at new_data[i]. + * + * Mirrors the original per-window behaviour exactly: only a full block_size + * window can match, candidates are accepted only when the weak (Adler-32) and + * strong (xxHash32) checksums both agree, and the lowest block index wins so + * the emitted op stream is byte-identical to the linear scan. When heads is + * non-NULL the candidate set is reached through the bucket index (expected + * O(1) per window); otherwise an exact linear scan is used. */ +static uint32_t delta_find_match(const uint8_t* window, uint32_t window_len, uint32_t adler, + bool full_window, const DeltaSignature* sig, const uint32_t* heads, + const uint32_t* next, uint32_t mask, uint32_t seed) { + if (!full_window || sig->block_count == 0) + return DELTA_NO_BLOCK; + + if (heads) { + uint32_t b = delta_adler_mix(adler) & mask; + uint32_t window_xxh = 0; + bool have_xxh = false; + for (uint32_t j = heads[b]; j != DELTA_NO_BLOCK; j = next[j]) { + if (sig->blocks[j].adler32 != adler) + continue; + if (!have_xxh) { + window_xxh = delta_xxhash32_seeded(window, window_len, seed); + have_xxh = true; + } + if (window_xxh == sig->blocks[j].xxhash) + return j; + } + return DELTA_NO_BLOCK; + } + + /* Fallback used when the index could not be allocated. */ + for (uint32_t j = 0; j < sig->block_count; j++) { + if (sig->blocks[j].adler32 == adler) { + uint32_t window_xxh = delta_xxhash32_seeded(window, window_len, seed); + if (window_xxh == sig->blocks[j].xxhash) + return j; + } + } + return DELTA_NO_BLOCK; +} + Delta* delta_compute(const void* new_file_data, uint64_t new_file_size, const DeltaSignature* sig, uint32_t block_size) { - if (!new_file_data || !sig || new_file_size == 0 || block_size == 0) + return delta_compute_seeded(new_file_data, new_file_size, sig, block_size, 0); +} + +Delta* delta_compute_seeded(const void* new_file_data, uint64_t new_file_size, + const DeltaSignature* sig, uint32_t block_size, uint32_t seed) { + if (!new_file_data || !sig || !sig->blocks || new_file_size == 0 || block_size == 0 || + block_size > DELTA_BLOCK_SIZE_MAX || sig->block_size != block_size) return NULL; const uint8_t* new_data = (const uint8_t*)new_file_data; uint32_t capacity = 64; uint32_t count = 0; - DeltaInstruction* instrs = malloc(capacity * sizeof(DeltaInstruction)); + DeltaInstruction* instrs = protocol_alloc((size_t)capacity * sizeof(DeltaInstruction)); if (!instrs) return NULL; + /* Build a one-time bucket index over the signature blocks keyed by the weak + * checksum. This turns the per-byte-window candidate lookup from an + * O(block_count) linear scan into an expected O(1) probe, which dominates + * the cost for large mostly-matching files (the diff steps one byte at a + * time through changed regions). On allocation failure the probe falls back + * to the original linear scan, so behaviour is unchanged under memory + * pressure. */ + uint32_t* index = NULL; + const uint32_t* chain_next = NULL; + uint32_t mask = 0; + if (sig->block_count > 0) { + uint32_t bucket_count = delta_next_pow2(sig->block_count); + index = delta_build_index(sig, bucket_count); + if (index) { + chain_next = index + 2 * bucket_count; + mask = bucket_count - 1; + } + } + uint64_t literal_start = 0; bool has_literal = false; @@ -226,34 +410,32 @@ Delta* delta_compute(const void* new_file_data, uint64_t new_file_size, const De } bool matched = false; - for (uint32_t j = 0; j < sig->block_count; j++) { - if (adler == sig->blocks[j].adler32 && full_window) { - uint32_t xxh = delta_xxhash32(new_data + i, window_len); - if (xxh == sig->blocks[j].xxhash) { - if (has_literal) { - if (!flush_literal(&instrs, &capacity, &count, new_data, literal_start, i)) { - free(instrs); - return NULL; - } - has_literal = false; - } - - if (!ensure_capacity(&instrs, &capacity, count)) { - free(instrs); - return NULL; - } - instrs[count].type = DELTA_INSTR_BLOCK_MATCH; - instrs[count].match.block_index = j; - instrs[count].match.block_offset = 0; - instrs[count].match.length = window_len; - count++; - - i += window_len; - rolling_valid = false; - matched = true; - break; + uint32_t match_block = delta_find_match(new_data + i, window_len, adler, full_window, sig, + index, chain_next, mask, seed); + if (match_block != DELTA_NO_BLOCK) { + if (has_literal) { + if (!flush_literal(&instrs, &capacity, &count, new_data, literal_start, i)) { + free_instructions(instrs, count); + free(index); + return NULL; } + has_literal = false; } + + if (!ensure_capacity(&instrs, &capacity, count)) { + free_instructions(instrs, count); + free(index); + return NULL; + } + instrs[count].type = DELTA_INSTR_BLOCK_MATCH; + instrs[count].match.block_index = match_block; + instrs[count].match.block_offset = 0; + instrs[count].match.length = window_len; + count++; + + i += window_len; + rolling_valid = false; + matched = true; } if (!matched) { @@ -265,20 +447,18 @@ Delta* delta_compute(const void* new_file_data, uint64_t new_file_size, const De } } + free(index); + if (has_literal) { if (!flush_literal(&instrs, &capacity, &count, new_data, literal_start, new_file_size)) { - free(instrs); + free_instructions(instrs, count); return NULL; } } - Delta* delta = malloc(sizeof(Delta)); + Delta* delta = protocol_alloc(sizeof(Delta)); if (!delta) { - for (uint32_t k = 0; k < count; k++) { - if (instrs[k].type == DELTA_INSTR_LITERAL) - free(instrs[k].literal.data); - } - free(instrs); + free_instructions(instrs, count); return NULL; } @@ -288,11 +468,24 @@ Delta* delta_compute(const void* new_file_data, uint64_t new_file_size, const De delta->delta_size = 0; for (uint32_t k = 0; k < count; k++) { + if (delta->delta_size == UINT64_MAX) { + delta_destroy(delta); + return NULL; + } delta->delta_size += 1; if (instrs[k].type == DELTA_INSTR_BLOCK_MATCH) { + if (delta->delta_size > UINT64_MAX - sizeof(uint32_t) * 3) { + delta_destroy(delta); + return NULL; + } delta->delta_size += sizeof(uint32_t) * 3; } else { - delta->delta_size += sizeof(uint32_t) + instrs[k].literal.length; + uint64_t extra = sizeof(uint32_t) + instrs[k].literal.length; + if (delta->delta_size > UINT64_MAX - extra) { + delta_destroy(delta); + return NULL; + } + delta->delta_size += extra; } } @@ -303,8 +496,13 @@ Data* delta_serialize(const Delta* delta) { if (!delta) return NULL; - uint64_t total = sizeof(uint64_t) + sizeof(uint32_t) + delta->delta_size; - uint8_t* buf = malloc((size_t)total); + if (delta->instruction_count > 0 && !delta->instructions) + return NULL; + uint64_t header_size = sizeof(uint64_t) + sizeof(uint32_t); + if (delta->delta_size > UINT64_MAX - header_size || header_size + delta->delta_size > SIZE_MAX) + return NULL; + uint64_t total = header_size + delta->delta_size; + uint8_t* buf = protocol_alloc((size_t)total); if (!buf) return NULL; @@ -344,7 +542,7 @@ Delta* delta_deserialize(const Data* data) { const uint8_t* buf = (const uint8_t*)data->data; size_t pos = 0; - Delta* delta = malloc(sizeof(Delta)); + Delta* delta = protocol_alloc(sizeof(Delta)); if (!delta) return NULL; @@ -361,8 +559,11 @@ Delta* delta_deserialize(const Data* data) { return NULL; } - delta->instructions = malloc(delta->instruction_count * sizeof(DeltaInstruction)); - if (!delta->instructions) { + delta->instructions = + delta->instruction_count == 0 + ? NULL + : protocol_alloc((size_t)delta->instruction_count * sizeof(DeltaInstruction)); + if (delta->instruction_count > 0 && !delta->instructions) { free(delta); return NULL; } @@ -371,11 +572,7 @@ Delta* delta_deserialize(const Data* data) { for (uint32_t i = 0; i < delta->instruction_count; i++) { if (pos >= data->size) { - for (uint32_t k = 0; k < i; k++) { - if (delta->instructions[k].type == DELTA_INSTR_LITERAL) - free(delta->instructions[k].literal.data); - } - free(delta->instructions); + free_instructions(delta->instructions, i); free(delta); return NULL; } @@ -387,8 +584,8 @@ Delta* delta_deserialize(const Data* data) { delta->delta_size += 1; if (type == DELTA_OP_BLOCK_MATCH) { - if (pos + sizeof(uint32_t) * 3 > data->size) { - free(delta->instructions); + if (data->size - pos < sizeof(uint32_t) * 3) { + free_instructions(delta->instructions, i); free(delta); return NULL; } @@ -401,12 +598,8 @@ Delta* delta_deserialize(const Data* data) { pos += sizeof(uint32_t); delta->delta_size += sizeof(uint32_t) * 3; } else if (type == DELTA_OP_LITERAL) { - if (pos + sizeof(uint32_t) > data->size) { - for (uint32_t k = 0; k < i; k++) { - if (delta->instructions[k].type == DELTA_INSTR_LITERAL) - free(delta->instructions[k].literal.data); - } - free(delta->instructions); + if (data->size - pos < sizeof(uint32_t)) { + free_instructions(delta->instructions, i); free(delta); return NULL; } @@ -415,18 +608,15 @@ Delta* delta_deserialize(const Data* data) { pos += sizeof(uint32_t); uint32_t lit_len = delta->instructions[i].literal.length; - if (pos + lit_len > data->size) { - for (uint32_t k = 0; k < i; k++) { - if (delta->instructions[k].type == DELTA_INSTR_LITERAL) - free(delta->instructions[k].literal.data); - } - free(delta->instructions); + if (lit_len > data->size - pos) { + free_instructions(delta->instructions, i); free(delta); return NULL; } - delta->instructions[i].literal.data = malloc(lit_len); + delta->instructions[i].literal.data = protocol_alloc(lit_len ? lit_len : 1); if (!delta->instructions[i].literal.data) { - free(delta->instructions); + log_message(LOG_LEVEL_ERROR, "Failed to allocate %u bytes for literal data", lit_len); + free_instructions(delta->instructions, i); free(delta); return NULL; } @@ -434,11 +624,7 @@ Delta* delta_deserialize(const Data* data) { pos += lit_len; delta->delta_size += sizeof(uint32_t) + lit_len; } else { - for (uint32_t k = 0; k < i; k++) { - if (delta->instructions[k].type == DELTA_INSTR_LITERAL) - free(delta->instructions[k].literal.data); - } - free(delta->instructions); + free_instructions(delta->instructions, i); free(delta); return NULL; } @@ -449,10 +635,12 @@ Delta* delta_deserialize(const Data* data) { void* delta_apply(const void* old_data, uint64_t old_size, const Delta* delta, uint32_t block_size) { - if (!old_data || !delta) + if (!old_data || !delta || (delta->new_file_size > 0 && delta->instructions == NULL) || + (delta->instruction_count > 0 && block_size == 0) || + delta->new_file_size > DELTA_MAX_FILE_SIZE || delta->new_file_size > SIZE_MAX) return NULL; - void* output = malloc((size_t)delta->new_file_size); + void* output = protocol_alloc(delta->new_file_size ? (size_t)delta->new_file_size : 1); if (!output) return NULL; @@ -463,19 +651,31 @@ void* delta_apply(const void* old_data, uint64_t old_size, const Delta* delta, for (uint32_t i = 0; i < delta->instruction_count; i++) { if (delta->instructions[i].type == DELTA_INSTR_BLOCK_MATCH) { uint64_t src_offset = (uint64_t)delta->instructions[i].match.block_index * block_size; + if (src_offset > UINT64_MAX - delta->instructions[i].match.block_offset) { + free(output); + return NULL; + } src_offset += delta->instructions[i].match.block_offset; uint32_t len = delta->instructions[i].match.length; - if (src_offset + len > old_size) { + if (src_offset > old_size || (uint64_t)len > old_size - src_offset || + out_pos > delta->new_file_size || (uint64_t)len > delta->new_file_size - out_pos) { free(output); return NULL; } memcpy(out + out_pos, old + src_offset, len); out_pos += len; - } else { + } else if (delta->instructions[i].type == DELTA_INSTR_LITERAL) { uint32_t len = delta->instructions[i].literal.length; + if (out_pos > delta->new_file_size || (uint64_t)len > delta->new_file_size - out_pos) { + free(output); + return NULL; + } memcpy(out + out_pos, delta->instructions[i].literal.data, len); out_pos += len; + } else { + free(output); + return NULL; } } @@ -511,7 +711,7 @@ bool delta_should_attempt(uint64_t old_size, uint64_t new_size, uint64_t max_fil } bool delta_is_worthwhile(const Delta* delta, uint64_t new_file_size) { - if (!delta || delta->instruction_count == 0) + if (!delta || delta->instruction_count == 0 || new_file_size == 0) return false; bool has_match = false; diff --git a/src/shared/delta.h b/src/shared/delta.h index 96cf0a6..d395a90 100644 --- a/src/shared/delta.h +++ b/src/shared/delta.h @@ -56,12 +56,22 @@ typedef struct { DeltaSignature* delta_signature_create(const void* old_file_data, uint64_t old_file_size, uint32_t block_size); +/* Seeded equivalent of delta_signature_create: the per-block strong (xxHash32) + * checksum uses `seed` (the low 32 bits of --checksum-seed). Passing seed 0 is + * identical to the unseeded function. */ +DeltaSignature* delta_signature_create_seeded(const void* old_file_data, uint64_t old_file_size, + uint32_t block_size, uint32_t seed); Data* delta_signature_serialize(const DeltaSignature* sig); DeltaSignature* delta_signature_deserialize(const Data* data); void delta_signature_destroy(DeltaSignature* sig); Delta* delta_compute(const void* new_file_data, uint64_t new_file_size, const DeltaSignature* sig, uint32_t block_size); +/* Seeded equivalent of delta_compute: the per-window strong (xxHash32) check + * uses `seed` (the low 32 bits of --checksum-seed). The receiver's signature + * must have been built with the same seed for matching. */ +Delta* delta_compute_seeded(const void* new_file_data, uint64_t new_file_size, + const DeltaSignature* sig, uint32_t block_size, uint32_t seed); Data* delta_serialize(const Delta* delta); Delta* delta_deserialize(const Data* data); void* delta_apply(const void* old_data, uint64_t old_size, const Delta* delta, uint32_t block_size); @@ -72,5 +82,7 @@ bool delta_is_worthwhile(const Delta* delta, uint64_t new_file_size); uint32_t delta_adler32(const void* data, uint32_t len); uint32_t delta_xxhash32(const void* data, uint32_t len); +uint32_t delta_xxhash32_seeded(const void* data, uint32_t len, uint32_t seed); +uint64_t delta_xxhash64(const void* data, size_t len); #endif diff --git a/src/shared/file.c b/src/shared/file.c index 62082ff..6220967 100644 --- a/src/shared/file.c +++ b/src/shared/file.c @@ -1,41 +1,109 @@ -#include +#ifndef _GNU_SOURCE +#define _GNU_SOURCE /* statx + STATX_BTIME for --crtimes birth-time capture */ +#endif #include +#include #include #include -#include +#include +#include #include #include #include -#include #include #include -#include "compression.h" -#include "delta.h" -#include "log.h" -#include "config.h" #include "data.h" +#include "delta.h" #include "file.h" +#include "file_store.h" +#include "identity.h" #include "log.h" #include "metadata.h" -#include "protocol.h" #include "utils.h" +#include "protocol.h" +#include "xattr.h" + +static bool write_all(int fd, const void* data, unsigned long long size) { + const unsigned char* p = data; + unsigned long long done = 0; + while (done < size) { + ssize_t n = write(fd, p + done, (size_t)(size - done)); + if (n < 0 && errno == EINTR) + continue; + if (n <= 0) + return false; + done += (unsigned long long)n; + } + return true; +} + +/* Preallocate `size` bytes on `fd` before any data is written (--preallocate). + * posix_fallocate reserves real disk blocks, so an out-of-space condition + * (ENOSPC/EDQUOT) surfaces up front instead of partway through a transfer; + * unavoidable fragmentation of a streamed file is also reduced. Some + * filesystems (e.g. tmpfs, ZFS) do not support it and return EOPNOTSUPP/ENOSYS, + * where we fall back to ftruncate, which still extends the logical size so the + * fail-fast/contiguity intent degrades gracefully but never fails. Genuine + * allocation failures are propagated as the error code (caller fails the write). + * posix_fallocate leaves the fd's file offset unchanged, so the subsequent + * write_all at offset 0 is unaffected. Returns 0 on success (including the + * fallback) or a nonzero error code. */ +static int preallocate_fd(int fd, unsigned long long size) { + if (size == 0) + return 0; + int rc = posix_fallocate(fd, 0, (off_t)size); + if (rc == EOPNOTSUPP || rc == ENOSYS) { + if (ftruncate(fd, (off_t)size) == 0) + return 0; + return errno; + } + return rc; +} + +/* Process-wide counter for scratch temp names. A --temp-dir scratch directory + is flat: different destinations that share a basename must never race onto + the same temp name. Deriving the trailing number from a global atomic + sequence keeps every temp name unique across the whole scratch directory + even when several threads write concurrently, so the O_EXCL creation loop + below almost never needs a retry. */ +static unsigned long long next_temp_sequence(void) { + static atomic_ullong sequence; + return atomic_fetch_add_explicit(&sequence, 1, memory_order_relaxed); +} + +bool file_checksum(File* file, ChecksumAlgo algo, uint64_t seed, uint8_t* out, size_t out_capacity, + size_t* out_len) { + if (!file || !out || !out_len || !file->data) + return false; + if (file->data->size == 0) { + return checksum_digest(algo, seed, "", 0, out, out_capacity, out_len); + } + if (!file->data->data && !file_load_data(file)) + return false; + return checksum_digest(algo, seed, file->data->data, file->data->size, out, out_capacity, + out_len); +} File* file_create(const char* path) { - File* file = (File*)malloc(sizeof(File)); + if (!path) + return NULL; + File* file = (File*)protocol_alloc(sizeof(File)); if (file == NULL) { - perror("ERROR: Could not allocate memory for file struct"); + log_perror("ERROR: Could not allocate memory for file struct"); return NULL; } - int path_len = strlen(path); - file->path = (char*)malloc(path_len + 1); + size_t path_len = strlen(path); + file->path = (char*)protocol_alloc(path_len + 1); if (file->path == NULL) { free(file); return NULL; } - strcpy(file->path, path); + memcpy(file->path, path, path_len); + file->path[path_len] = '\0'; + file->send_path = NULL; file->data = data_create_reserve(0); if (file->data == NULL) { free(file->path); @@ -44,6 +112,18 @@ File* file_create(const char* path) { } file->metadata = NULL; file->skip = false; + file->is_dir = false; + file->dir_time_only = false; + file->basis_link = NULL; + file->link_group = 0; + file->link_first = false; + file->hardlink_target = NULL; + file->is_symlink = false; + file->symlink_target = NULL; + file->is_special = false; + file->rdev_major = 0; + file->rdev_minor = 0; + file->xattrs = NULL; return file; } @@ -57,13 +137,24 @@ void file_destroy(void* item) { file->metadata = NULL; free(file->path); file->path = NULL; + free(file->send_path); + file->send_path = NULL; + free(file->basis_link); + file->basis_link = NULL; + free(file->hardlink_target); + file->hardlink_target = NULL; + free(file->symlink_target); + file->symlink_target = NULL; + xattr_list_free(file->xattrs); + file->xattrs = NULL; free(file); } -FileMetadata* file_metadata_create(const struct stat* stats) { - FileMetadata* m = malloc(sizeof(FileMetadata)); +FileMetadata* file_metadata_create(const char* path, const struct stat* stats, bool capture_atime, + bool capture_crtime) { + FileMetadata* m = protocol_alloc(sizeof(FileMetadata)); if (m == NULL) { - perror("ERROR: Could not allocate memory for file metadata"); + log_perror("ERROR: Could not allocate memory for file metadata"); return NULL; } m->mode = stats->st_mode; @@ -75,6 +166,37 @@ FileMetadata* file_metadata_create(const struct stat* stats) { #else m->mtime_nsec = 0; #endif + /* -U/--atimes: capture the access time from the same pre-read stat the + scanner already took, so the value is not clobbered by a later read for + transfer. The timestamp is populated (and atime_valid set) only on Linux, + where st_atim is populated; on other platforms the atime is left alone + rather than clobbered to the default 0/epoch by an unpopulated value. */ +#ifdef __linux__ + m->atime_valid = capture_atime; + m->atime_sec = stats->st_atim.tv_sec; + m->atime_nsec = stats->st_atim.tv_nsec; +#else + m->atime_valid = false; + m->atime_sec = 0; + m->atime_nsec = 0; +#endif + /* -N/--crtimes: birth time is not available via struct stat in general; on + Linux it needs statx STATX_BTIME. If unavailable it is captured as a + documented no-op (the flag stays accepted, crtime_valid stays false). */ + m->crtime_valid = false; + m->crtime_sec = 0; + m->crtime_nsec = 0; + if (capture_crtime) { +#ifdef STATX_BTIME + struct statx stx; + if (path != NULL && statx(AT_FDCWD, path, AT_STATX_SYNC_AS_STAT, STATX_BTIME, &stx) == 0 && + (stx.stx_mask & STATX_BTIME) != 0) { + m->crtime_valid = true; + m->crtime_sec = (time_t)stx.stx_btime.tv_sec; + m->crtime_nsec = (long)stx.stx_btime.tv_nsec; + } +#endif + } return m; } @@ -82,569 +204,1111 @@ void file_metadata_destroy(void* metadata) { free(metadata); } +/* --open-noatime: process-wide sender policy (client-only, never crosses the + * wire). When enabled, opening a source file for transfer uses O_NOATIME so + * the read does not bump the source's on-disk access time. It degrades safely + * to a normal open where O_NOATIME is unavailable (not defined) or refused + * (EPERM, because it needs CAP_FOWNER): the data path never silently changes, + * only the atime-bump is skipped. */ +static bool file_open_noatime = false; + +void file_set_open_noatime(bool enable) { + file_open_noatime = enable; +} + +bool file_get_open_noatime(void) { + return file_open_noatime; +} + +/* Open `path` read-only for transfer, honouring --open-noatime when set. */ +int file_open_for_read(const char* path) { + int flags = O_RDONLY; +#ifdef O_NOATIME + if (file_get_open_noatime()) + flags |= O_NOATIME; +#endif + int fd = open(path, flags); +#ifdef O_NOATIME + if (fd < 0 && (flags & O_NOATIME)) + fd = open(path, O_RDONLY); /* degrade safely on EPERM / unsupported fs */ +#endif + return fd; +} + bool file_load_data(File* file) { - if (file == NULL) + if (file == NULL || !file->data) return false; if (file->data->data == NULL) { - file->data->data = malloc(file->data->size); + if (file->data->size == 0) + return true; + file->data->data = protocol_alloc(file->data->size); if (file->data->data == NULL) { - perror("Could not allocate memory for file data"); + log_perror("Could not allocate memory for file data"); return false; } } size_t bytes_read = file_content_to_buffer(file); if (bytes_read != file->data->size) { log_message(LOG_LEVEL_ERROR, "Did not read expected amount of bytes from file"); + free(file->data->data); + file->data->data = NULL; + file->data->size = 0; return false; } return true; } -bool file_send_single_calls(File* file, int file_descriptor, bool use_metadata, - int compression_level, bool send_path) { - const Data* data_to_send = file->data; - Data* compressed_data = NULL; - if (compression_level > 0 && !compression_should_skip(file->path)) { - compressed_data = data_compress(file->data, compression_level); - if (compressed_data == NULL) { - log_message(LOG_LEVEL_ERROR, "Failed to compress file data"); - return false; - } - data_to_send = compressed_data; - } - if (send_path && !send_str(file_descriptor, file->path)) { - data_destroy(compressed_data); - return false; - } - if (use_metadata && !metadata_send(file_descriptor, file->metadata)) { - data_destroy(compressed_data); - return false; - } - if (!send_data(file_descriptor, data_to_send)) { - data_destroy(compressed_data); - return false; - } - data_destroy(compressed_data); - return true; -} - -bool file_save_to_disk(const char* root_directory, File* file, const Config* config) { - (void)config; - if (has_path_traversal(file->path)) { - log_message(LOG_LEVEL_ERROR, "Path traversal detected in file path: %s", file->path); - return false; - } - - // Resolve the destination root to its real path, preventing symlink-based escapes. - // If the root does not yet exist, try to create it so realpath can succeed. - char* resolved_root = realpath(root_directory, NULL); - if (resolved_root == NULL) { - if (mkdir_r(root_directory)) { - resolved_root = realpath(root_directory, NULL); - } - } - if (resolved_root == NULL) { - log_message(LOG_LEVEL_ERROR, "Failed to resolve destination root: %s", root_directory); - return false; - } - - char* disk_path = path_cat(resolved_root, file->path); - if (disk_path == NULL) { - free(resolved_root); - return false; - } - - // Ensure the target directory exists so the parent can be resolved for path safety. - char* dir_dup = str_dup(disk_path); - if (!dir_dup) { - free(resolved_root); - free(disk_path); - return false; - } - char* dir_str = dirname(dir_dup); - // Create the directory if needed (no-op if it already exists) so realpath can resolve it. - if (!mkdir_r(dir_str)) { - free(dir_dup); - free(resolved_root); - free(disk_path); - return false; - } - char* resolved_dir = realpath(dir_str, NULL); - free(dir_dup); - if (resolved_dir == NULL) { - log_message(LOG_LEVEL_ERROR, "Failed to resolve directory for: %s", disk_path); - free(resolved_root); - free(disk_path); - return false; - } - - // Verify that the resolved directory is inside the resolved root. - // Both are canonical absolute paths — this prevents symlink-based escapes. - size_t root_len = strlen(resolved_root); - if (strncmp(resolved_dir, resolved_root, root_len) != 0 || - (resolved_dir[root_len] != '\0' && resolved_dir[root_len] != '/')) { - log_message(LOG_LEVEL_ERROR, "Path escape detected: %s is outside %s", disk_path, - root_directory); - free(resolved_dir); - free(resolved_root); - free(disk_path); - return false; - } - free(resolved_dir); - free(resolved_root); - - bool ok = to_disk(disk_path, file->data->data, file->data->size); - if (ok) - file_restore_metadata(disk_path, file->metadata); - free(disk_path); - return ok; -} - -static void* old_data_from_path(const char* full_path, unsigned long long old_size) { - void* data = malloc((size_t)old_size); - if (!data) - return NULL; - FILE* fp = fopen(full_path, "rb"); - if (!fp) { - free(data); - return NULL; - } - size_t nread = fread(data, 1, (size_t)old_size, fp); - fclose(fp); - if (nread != (size_t)old_size) { - free(data); - return NULL; - } - return data; -} - -static File* receive_delta_file(int fd, const Config* config, const char* check_path, - void* old_data, unsigned long long old_size) { - if (!old_data) - return NULL; - - DeltaSignature* sig = delta_signature_create(old_data, old_size, config->delta_block_size); - if (!sig) { - free(old_data); - return NULL; - } - - Data* sig_data = delta_signature_serialize(sig); - if (!sig_data) { - delta_signature_destroy(sig); - free(old_data); - return NULL; - } - - bool sig_sent = send_status(fd, STATUS_DELTA_SIGNATURE) && send_data(fd, sig_data); - data_destroy(sig_data); - - if (!sig_sent) { - delta_signature_destroy(sig); - free(old_data); - return NULL; - } - - Status resp; - if (!receive_status(fd, &resp)) { - delta_signature_destroy(sig); - free(old_data); - return NULL; - } - - if (resp == STATUS_DELTA_DATA) { - Data* delta_data = receive_data(fd); - if (!delta_data) { - delta_signature_destroy(sig); - free(old_data); - send_status(fd, STATUS_ERROR); - return NULL; - } - - Data* raw_delta = delta_data; - if (config->use_compression) { - raw_delta = data_decompress(delta_data); - data_destroy(delta_data); - if (!raw_delta) { - free(old_data); - delta_signature_destroy(sig); - send_status(fd, STATUS_ERROR); - return NULL; - } - } - - Delta* delta = delta_deserialize(raw_delta); - data_destroy(raw_delta); - if (!delta) { - free(old_data); - delta_signature_destroy(sig); - send_status(fd, STATUS_ERROR); - return NULL; - } - - void* new_data = delta_apply(old_data, old_size, delta, config->delta_block_size); - uint64_t new_size = delta->new_file_size; - delta_destroy(delta); - - if (!new_data) { - free(old_data); - delta_signature_destroy(sig); - send_status(fd, STATUS_ERROR); - return NULL; - } - - File* file = file_create(check_path); - if (!file) { - free(new_data); - free(old_data); - delta_signature_destroy(sig); - send_status(fd, STATUS_ERROR); - return NULL; - } - - if (config->use_metadata) { - int meta_ok = 1; - file->metadata = metadata_receive(fd, &meta_ok); - if (!meta_ok) { - file_destroy(file); - free(new_data); - free(old_data); - delta_signature_destroy(sig); - send_status(fd, STATUS_ERROR); - return NULL; - } - } - - data_destroy(file->data); - file->data = data_create(new_data, (size_t)new_size); - - free(old_data); - delta_signature_destroy(sig); - return file; - } - - if (resp == STATUS_NEXT) { - delta_signature_destroy(sig); - free(old_data); - - File* file = file_create(check_path); - if (!file) { - send_status(fd, STATUS_ERROR); - return NULL; - } - - if (config->use_metadata) { - int meta_ok = 1; - file->metadata = metadata_receive(fd, &meta_ok); - if (!meta_ok) { - file_destroy(file); - send_status(fd, STATUS_ERROR); - return NULL; - } - } - - Data* file_data = receive_data(fd); - if (file_data == NULL) { - file_destroy(file); - send_status(fd, STATUS_ERROR); - return NULL; - } - - if (config->use_compression) { - Data* uncompressed = data_decompress(file_data); - data_destroy(file_data); - if (uncompressed == NULL) { - file_destroy(file); - send_status(fd, STATUS_ERROR); - return NULL; - } - file_data = uncompressed; - } - - data_destroy(file->data); - file->data = file_data; - return file; - } - - delta_signature_destroy(sig); - free(old_data); - return NULL; -} - -File* receive_incremental_check(int fd, const Config* config, bool* skipped) { - *skipped = false; - char* check_path = receive_str(fd); - if (check_path == NULL) { - send_status(fd, STATUS_ERROR); - return NULL; - } - - unsigned long long check_size; - long long check_mtime; - if (!receive_n_data(fd, &check_size, sizeof(check_size)) || - !receive_n_data(fd, &check_mtime, sizeof(check_mtime))) { - free(check_path); - send_status(fd, STATUS_ERROR); - return NULL; - } - - if (has_path_traversal(check_path)) { - log_message(LOG_LEVEL_ERROR, "Path traversal detected: %s", check_path); - free(check_path); - send_status(fd, STATUS_ERROR); - return NULL; - } - - char* full_path = path_cat(config->receive_root_directory, check_path); - struct stat st; - bool has_old_file = (full_path && lstat(full_path, &st) == 0); - unsigned long long old_size = has_old_file ? (unsigned long long)st.st_size : 0; - - bool match = has_old_file && (unsigned long long)st.st_size == check_size && - (long long)st.st_mtime == check_mtime; - - if (match) { - if (!send_status(fd, STATUS_OK)) { - free(full_path); - free(check_path); - return NULL; - } - free(full_path); - free(check_path); - *skipped = true; - return NULL; - } - - bool try_delta = config->use_delta && has_old_file && - delta_should_attempt(old_size, check_size, config->delta_max_file_size); - - if (try_delta) { - void* old_data = old_data_from_path(full_path, old_size); - File* delta_file = receive_delta_file(fd, config, check_path, old_data, old_size); - if (delta_file) { - free(full_path); - free(check_path); - return delta_file; - } - try_delta = false; - } - - if (!try_delta) { - if (!send_status(fd, STATUS_NEXT)) { - free(full_path); - free(check_path); - return NULL; - } - } - - File* file = file_create(check_path); - free(check_path); - free(full_path); - if (file == NULL) { - send_status(fd, STATUS_ERROR); - return NULL; - } - - if (config->use_metadata) { - int meta_ok = 1; - file->metadata = metadata_receive(fd, &meta_ok); - if (!meta_ok) { - file_destroy(file); - send_status(fd, STATUS_ERROR); - return NULL; - } - } - - Data* file_data = receive_data(fd); - if (file_data == NULL) { - file_destroy(file); - send_status(fd, STATUS_ERROR); - return NULL; - } - - if (config->use_compression) { - Data* uncompressed = data_decompress(file_data); - data_destroy(file_data); - if (uncompressed == NULL) { - file_destroy(file); - send_status(fd, STATUS_ERROR); - return NULL; - } - file_data = uncompressed; - } - - data_destroy(file->data); - file->data = file_data; - return file; -} - -bool to_disk(const char* path, const void* data, unsigned long long data_size) { - char* tmp_path = NULL; - char* directory = NULL; - - char* path_dup = str_dup(path); - if (!path_dup) - return false; - const char* dir_result = dirname(path_dup); - directory = str_dup(dir_result); - free(path_dup); - if (!directory) - return false; - - bool ok = true; - if (!mkdir_r(directory)) - goto done; - - size_t path_len = strlen(path); - tmp_path = malloc(path_len + 5); - if (!tmp_path) { - ok = false; - goto done; - } - memcpy(tmp_path, path, path_len); - memcpy(tmp_path + path_len, ".tmp", 5); - - FILE* file_pointer = fopen(tmp_path, "wb"); - if (file_pointer == NULL) { - perror("Could not open temporary file"); - ok = false; - goto done; - } - if (fwrite(data, 1, data_size, file_pointer) != data_size) { - perror("Failed to write all data to temporary file"); - fclose(file_pointer); - unlink(tmp_path); - ok = false; - goto done; - } - fclose(file_pointer); - - if (rename(tmp_path, path) != 0) { - perror("Failed to atomically rename temporary file"); - unlink(tmp_path); - ok = false; - goto done; - } - -done: - free(tmp_path); - free(directory); - return ok; -} - -bool file_send_sendfile(File* file, int file_descriptor, bool use_metadata, int compression_level, - bool send_path) { - // sendfile is incompatible with compression (kernel zero-copy). - // If compression is requested, fall back to the regular send path. - // NOTE: This is a safety net only — callers must ensure compression_level == 0 - // before calling file_send_sendfile. The fallback to file_send_single_calls - // preserves the send_path contract, but callers should not rely on it for - // correctness (the --sendfile flag is validated to be mutually exclusive with - // -c/--compress at the CLI layer). - if (compression_level > 0) - return file_send_single_calls(file, file_descriptor, use_metadata, compression_level, - send_path); - - if (send_path && !send_str(file_descriptor, file->path)) - return false; - if (use_metadata && !metadata_send(file_descriptor, file->metadata)) - return false; - - int fd = open(file->path, O_RDONLY); - if (fd == -1) { - perror("Could not open file for sendfile"); - return false; - } - - unsigned long long file_size = file->data->size; - if (!send_n_data(file_descriptor, &file_size, sizeof(unsigned long long))) { - close(fd); - return false; - } - - off_t offset = 0; - while ((unsigned long long)offset < file_size) { - ssize_t sent = sendfile(file_descriptor, fd, &offset, file_size - offset); - if (sent == -1) { - if (errno == EAGAIN || errno == EINTR) - continue; - perror("sendfile failed"); - close(fd); - return false; - } - } - - close(fd); - return true; -} - -File* file_receive(const Config* config, int file_descriptor) { - char* path = receive_str(file_descriptor); - if (path == NULL) - return NULL; - File* file = file_create(path); - free(path); - if (file == NULL) - return NULL; - if (config->use_metadata) { - int meta_ok = 1; - file->metadata = metadata_receive(file_descriptor, &meta_ok); - if (!meta_ok) { - file_destroy(file); - return NULL; - } - } - Data* file_data = receive_data(file_descriptor); - if (file_data == NULL) { - file_destroy(file); - return NULL; - } - if (config->use_compression) { - Data* file_data_uncompressed = data_decompress(file_data); - data_destroy(file_data); - if (file_data_uncompressed == NULL) { - file_destroy(file); - return NULL; - } - file_data = file_data_uncompressed; - } - data_destroy(file->data); - file->data = file_data; - return file; -} - size_t file_content_to_buffer(File* file) { - FILE* file_pointer = fopen(file->path, "rb"); + if (!file || !file->path || !file->data || (!file->data->data && file->data->size != 0)) + return 0; + int fd = file_open_for_read(file->path); + if (fd < 0) { + log_perror("Could not open the file!"); + return 0; + } + FILE* file_pointer = fdopen(fd, "rb"); if (file_pointer == NULL) { - perror("Could not open the file!"); + close(fd); + log_perror("Could not open the file!"); return 0; } size_t bytes_read = fread(file->data->data, 1, file->data->size, file_pointer); if (bytes_read != (size_t)file->data->size) { fclose(file_pointer); - perror("Read unexpected number of bytes from File!"); + log_perror("Read unexpected number of bytes from File!"); return 0; } fclose(file_pointer); return bytes_read; } -int receive_manifest(int fd, const Config* config, int* next_status) { - int count; - if (!receive_int(fd, &count)) +/* ---- Secure filesystem primitives ---- */ + +static int authorized_root_fd = -1; +static char* authorized_root_path; + +static bool path_is_within_root(const char* root, const char* path) { + size_t root_len = strlen(root); + return strncmp(root, path, root_len) == 0 && (path[root_len] == '\0' || path[root_len] == '/'); +} + +bool file_set_authorized_root(int fd, const char* canonical_path) { + char* path_copy = canonical_path ? str_dup(canonical_path) : NULL; + if (canonical_path && !path_copy) { + authorized_root_fd = -1; + free(authorized_root_path); + authorized_root_path = NULL; + return false; + } + authorized_root_fd = fd; + free(authorized_root_path); + authorized_root_path = path_copy; + return true; +} + +bool file_path_exists_secure(const char* path) { + if (!path) + return false; + char* leaf = NULL; + int parent_fd = file_open_secure_parent(path, &leaf, false); + if (parent_fd < 0) + return false; + struct stat st; + bool exists = fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) == 0; + close(parent_fd); + free(leaf); + return exists; +} + +bool file_stat_secure(const char* path, struct stat* st) { + if (!path || !st) + return false; + char* leaf = NULL; + int parent_fd = file_open_secure_parent(path, &leaf, false); + if (parent_fd < 0) + return false; + bool exists = fstatat(parent_fd, leaf, st, AT_SYMLINK_NOFOLLOW) == 0 && S_ISREG(st->st_mode); + close(parent_fd); + free(leaf); + return exists; +} + +static bool stat_is_newer(const struct stat* st, const FileMetadata* metadata) { + if (!st || !metadata) + return false; +#ifdef __linux__ + long mtime_nsec = st->st_mtim.tv_nsec; +#else + long mtime_nsec = 0; +#endif + return st->st_mtime > metadata->mtime_sec || + (st->st_mtime == metadata->mtime_sec && mtime_nsec > metadata->mtime_nsec); +} + +bool file_destination_is_newer_secure(const char* path, const FileMetadata* metadata) { + struct stat st; + return file_stat_secure(path, &st) && stat_is_newer(&st, metadata); +} + +/* --keep-dirlinks (-K) receiver process-wide policy: when set, a destination + * path component that is itself a symlink to an in-root directory is followed + * (used as that directory) instead of failing the O_NOFOLLOW walk. Only ever + * honoured when the resolved target is a directory that stays beneath the + * authorized root, so a malicious symlink can never redirect the write outside + * it. Client of record is the server's receiver. */ +static bool file_keep_dirlinks = false; + +void file_set_keep_dirlinks(bool enable) { + file_keep_dirlinks = enable; +} + +bool file_get_keep_dirlinks(void) { + return file_keep_dirlinks; +} + +/* --trust-sender (Phase 5) receiver process-wide policy: when set, the receiver + * trusts the sender's file list and skips its own redundant up-front re- + * validation (empty/".." path rejection, escaping-symlink-target containment). + * Kept OFF by default; the server's per-connection handler sets it once from the + * received config before any receiver/writer threads start (each connection is + * its own forked process, so this per-process value never bleeds across + * connections). */ +static bool file_trust_sender = false; + +void file_set_trust_sender(bool enable) { + file_trust_sender = enable; +} + +bool file_get_trust_sender(void) { + return file_trust_sender; +} + +/* True when `target` is a lexical symlink target that can never escape the + * receive root once created beneath it: relative (not absolute) and containing + * no ".." path component. Used by --munge-links' sender-side containment: an + * escaping target is never transmitted (the entry is skipped/contained). */ +bool file_symlink_target_contained(const char* target) { + if (!target || target[0] == '\0' || target[0] == '/') + return false; + const char* p = target; + while (*p) { + const char* slash = strchr(p, '/'); + size_t comp_len = slash ? (size_t)(slash - p) : strlen(p); + if (comp_len == 2 && p[0] == '.' && p[1] == '.') + return false; + if (!slash) + break; + p = slash + 1; + } + return true; +} + +/* Remove a leading symlink munge marker (if present); returns true when the + * marker was stripped. `target` is a mutable NUL-terminated buffer. */ +bool file_symlink_unmunge(char* target) { + if (!target) + return false; + static const char* const marker = SYMLINK_MUNGE_PREFIX; + size_t marker_len = strlen(marker); + if (strncmp(target, marker, marker_len) != 0) + return false; + size_t rest = strlen(target + marker_len) + 1; + memmove(target, target + marker_len, rest); + return true; +} + +/* Owned copy of `target` prefixed with SYMLINK_MUNGE_PREFIX (the sender-side + * --munge-links rewriting). Returns NULL on allocation failure. */ +char* file_symlink_munge(const char* target) { + if (!target) + return NULL; + static const char* const marker = SYMLINK_MUNGE_PREFIX; + size_t marker_len = strlen(marker); + size_t target_len = strlen(target); + char* out = malloc(marker_len + target_len + 1); + if (!out) + return NULL; + memcpy(out, marker, marker_len); + memcpy(out + marker_len, target, target_len + 1); + return out; +} + +/* Create a symlink at `path` pointing to `target`, confined below the + * authorized root: the parent directory is opened with an O_NOFOLLOW fd walk + * and the link is created with symlinkat so neither the destination chain nor + * the target is ever followed. The final component is never dereferenced: an + * existing non-directory entry at `path` is unlinked by name before the link is + * placed; an existing directory there is left untouched (returns false, so a + * caller can treat it as a collision). As a receiver-side trust-boundary + * invariant, `target` must be file_symlink_target_contained() (relative and + * ".."-free): an absolute or escaping target is rejected outright (returns + * false) so a malicious sender can never materialize a symlink that points + * outside the receive root. */ +bool file_symlink_at_secure(const char* path, const char* target) { + /* The link itself (`path`) is always kept below the authorized root. The + TARGET may point anywhere: normally only a contained (relative, ".."-free) + target is permitted so a malicious sender can never plant a symlink that + later dereferences outside the root. Under --trust-sender that target + containment check is relaxed (the receiver trusts the sender and copies the + link verbatim, matching rsync -l), but path/leaf confinement is never + disabled, so the link still cannot be placed outside the tree. */ + if (!path || !target || has_path_traversal(path)) + return false; + if (!file_trust_sender && !file_symlink_target_contained(target)) + return false; + char* leaf = NULL; + int parent_fd = file_open_secure_parent(path, &leaf, true); + if (parent_fd < 0) + return false; + bool ok = false; + struct stat st; + bool exists = fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) == 0; + if (exists && S_ISDIR(st.st_mode)) { + /* A directory already at this path cannot be replaced atomically with a + symlink without --force semantics; leave it and report the collision. */ + ok = false; + } else { + if (exists && unlinkat(parent_fd, leaf, 0) != 0 && errno != ENOENT) + goto out; + ok = symlinkat(target, parent_fd, leaf) == 0; + } +out: + close(parent_fd); + free(leaf); + return ok; +} + +/* Open the directory named by canonical absolute `resolved`, which the caller + * has already verified lies beneath `root` (the canonical authorized root). + * Each component is opened relative to the authorized-root fd with O_NOFOLLOW, + * so a directory swapped for a symlink after the realpath() check cannot + * redirect the open outside the root -- the walk simply fails. This replaces + * re-opening the absolute resolved path (TOCTOU). Returns an O_DIRECTORY fd, + * or -1 (the root itself and any error are refused). */ +static int open_dir_beneath_root(const char* resolved, const char* root) { + size_t root_len = strlen(root); + const char* rel = resolved + root_len; + while (*rel == '/') + rel++; + if (*rel == '\0') return -1; - ArrayList* manifest = array_list_create(free); - if (manifest) { - for (int i = 0; i < count; i++) { - char* s = receive_str(fd); - if (s) - array_list_add(manifest, s); + int fd = dup(authorized_root_fd); + if (fd < 0) + return -1; + char* copy = str_dup(rel); + if (!copy) { + close(fd); + return -1; + } + char* save = NULL; + for (char* component = strtok_r(copy, "/", &save); component; + component = strtok_r(NULL, "/", &save)) { + if (strcmp(component, ".") == 0) + continue; + /* A canonical realpath() output never contains "." or ".."; refuse ".." + defensively rather than let it climb toward the root. */ + int next = strcmp(component, "..") == 0 + ? -1 + : openat(fd, component, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + if (next < 0) { + close(fd); + free(copy); + return -1; + } + close(fd); + fd = next; + } + free(copy); + return fd; +} + +int file_open_secure_parent(const char* path, char** leaf_out, bool create_dirs) { + char* copy = str_dup(path); + if (!copy) + return -1; + char* parent = dirname(copy); + const char* slash = strrchr(path, '/'); + char* leaf = str_dup(slash ? slash + 1 : path); + if (!leaf) { + free(copy); + return -1; + } + int fd; + if (authorized_root_fd >= 0) { + if (!authorized_root_path || path[0] != '/' || + !path_is_within_root(authorized_root_path, path)) { + free(copy); + free(leaf); + return -1; + } + fd = dup(authorized_root_fd); + if (fd < 0) { + free(copy); + free(leaf); + return -1; + } + size_t root_len = strlen(authorized_root_path); + char* relative = str_dup(path + root_len); + if (!relative) { + free(copy); + free(leaf); + close(fd); + return -1; + } + free(copy); + copy = relative; + parent = dirname(copy); + } else { + fd = (parent[0] == '/') ? open("/", O_RDONLY | O_DIRECTORY | O_CLOEXEC) + : open(".", O_RDONLY | O_DIRECTORY | O_CLOEXEC); + } + if (fd < 0) { + free(copy); + free(leaf); + return -1; + } + char* save = NULL; + char* component = strtok_r(parent, "/", &save); + char rel_buf[PATH_MAX] = ""; + while (component) { + if (strcmp(component, "..") == 0) { + close(fd); + free(copy); + free(leaf); + return -1; + } + if (strcmp(component, ".") != 0) { + int next = openat(fd, component, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + if (next < 0 && create_dirs && errno == ENOENT) { + bool created = mkdirat(fd, component, 0755) == 0; + if (created || errno == EEXIST) { + /* P7 Wave E: --copy-as owns EVERY entry, including the intermediate + directories this walk creates implicitly. Its target ids are a + global policy, so they are available here without per-entry source + metadata. Only a directory this walk actually created is chowned + (a pre-existing destination directory is left alone, matching + rsync's transferred-entry scope); the helper is a no-op unless an + identity policy is active. */ + if (created && identity_copy_as_active() && + !identity_apply_ownership_link(fd, component, 0, 0)) { + /* A REQUIRED --copy-as ownership that cannot be applied to a + directory this walk just created must fail the entry rather than + leave that implicit parent owned by the receiver. Preserve the + failing errno across the cleanup so the caller logs the real + reason. */ + int saved_errno = errno; + close(fd); + free(copy); + free(leaf); + errno = saved_errno; + return -1; + } + next = openat(fd, component, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + } + } + /* --keep-dirlinks (-K): a path component that is an existing symlink to + an in-root directory is used as THAT directory rather than failing the + O_NOFOLLOW walk. Only honoured when the symlink resolves to a + directory that stays beneath the authorized root, so a malicious link + can never redirect the write outside it. */ + if (next < 0 && file_keep_dirlinks && authorized_root_path != NULL && + (errno == ELOOP || errno == ENOTDIR || errno == EACCES)) { + struct stat lst; + if (fstatat(fd, component, &lst, AT_SYMLINK_NOFOLLOW) == 0 && S_ISLNK(lst.st_mode)) { + char candidate[PATH_MAX]; + char root[PATH_MAX]; + if (realpath(authorized_root_path, root) && + snprintf(candidate, sizeof(candidate), "%s%s/%s", root, rel_buf, component) < + (int)sizeof(candidate)) { + char resolved[PATH_MAX]; + if (realpath(candidate, resolved) && strcmp(resolved, root) != 0 && + strncmp(root, resolved, strlen(root)) == 0 && + (resolved[strlen(root)] == '/' || resolved[strlen(root)] == '\0')) { + struct stat rst; + if (stat(resolved, &rst) == 0 && S_ISDIR(rst.st_mode)) { + /* Open the resolved directory through a relative no-follow walk + from the authorized-root fd instead of re-opening the + absolute `resolved` path: swapping an intermediate directory + for a symlink between realpath() and open() (TOCTOU) then + merely fails the walk rather than redirecting the fd outside + the root. */ + next = open_dir_beneath_root(resolved, root); + } + } + } + } + } + if (next < 0) { + close(fd); + free(copy); + free(leaf); + return -1; + } + close(fd); + fd = next; + /* Track the walked relative prefix so the -K candidate path can be + reconstructed. An overflow while building it means the whole path is + at the PATH_MAX edge, so fail hard rather than silently building a + wrong (truncated) candidate for a later -K follow. */ + size_t need = strlen(rel_buf) + strlen(component) + 2; + if (need <= sizeof(rel_buf)) { + strcat(rel_buf, "/"); + strcat(rel_buf, component); + } else if (file_keep_dirlinks) { + close(fd); + free(copy); + free(leaf); + return -1; + } + } + component = strtok_r(NULL, "/", &save); + } + free(copy); + *leaf_out = leaf; + return fd; +} + +/* Normalized copy of a directory path: leading '/' kept, trailing '/' removed + * ("/" and "//" both collapse to "/"). A trailing slash otherwise makes the + * last path component empty, so probing that empty leaf below its parent + * always fails. */ +static char* normalize_directory_path(const char* path) { + if (!path) + return NULL; + size_t len = strlen(path); + while (len > 1 && path[len - 1] == '/') + len--; + char* norm = malloc(len + 1); + if (!norm) + return NULL; + memcpy(norm, path, len); + norm[len] = '\0'; + return norm; +} + +bool file_ensure_directory_secure(const char* path) { + if (!path) + return false; + char* norm = normalize_directory_path(path); + if (!norm) + return false; + /* The authorized root is already an open directory, and the filesystem root + is always present: there is no final component left to create for them. */ + bool root_is_open = + authorized_root_fd >= 0 && authorized_root_path && strcmp(norm, authorized_root_path) == 0; + if (root_is_open || strcmp(norm, "/") == 0) { + free(norm); + return true; + } + char* leaf = NULL; + int parent_fd = file_open_secure_parent(norm, &leaf, true); + free(norm); + if (parent_fd < 0) + return false; + + int dir_fd = openat(parent_fd, leaf, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + bool created = false; + if (dir_fd < 0 && errno == ENOENT) { + if (mkdirat(parent_fd, leaf, 0755) == 0) { + created = true; + dir_fd = openat(parent_fd, leaf, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + } else if (errno == EEXIST) { + dir_fd = openat(parent_fd, leaf, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); } - fprintf(stderr, "Deleting files not in manifest...\n"); - delete_extras(config->receive_root_directory, manifest); - array_list_delete(manifest); } - if (!receive_status(fd, next_status)) + bool ok = dir_fd >= 0; + /* --copy-as owns a directory this call just created (the final component; + intermediate components were handled by file_open_secure_parent above). A + failed REQUIRED ownership fails the call rather than leaving the directory + owned by the receiver. */ + if (ok && created && identity_copy_as_active() && + !identity_apply_ownership_link(parent_fd, leaf, 0, 0)) + ok = false; + if (dir_fd >= 0) + close(dir_fd); + close(parent_fd); + free(leaf); + return ok; +} + +/* True when `path` resolves to an existing directory below the authorized root + * (never creating anything). Used by the server to decide whether a client's + * destination root already exists. A trailing slash on `path` and a destination + * equal to the authorized root itself are normalized/handled here so both + * previously-working destination forms keep working. */ +bool file_directory_exists_secure(const char* path) { + if (!path) + return false; + char* norm = normalize_directory_path(path); + if (!norm) + return false; + bool root_is_open = + authorized_root_fd >= 0 && authorized_root_path && strcmp(norm, authorized_root_path) == 0; + if (root_is_open || strcmp(norm, "/") == 0) { + free(norm); + return true; + } + char* leaf = NULL; + int parent_fd = file_open_secure_parent(norm, &leaf, false); + free(norm); + if (parent_fd < 0) + return false; + int dir_fd = openat(parent_fd, leaf, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + if (dir_fd < 0 && errno == ENOENT) + dir_fd = -1; + bool ok = dir_fd >= 0; + if (dir_fd >= 0) + close(dir_fd); + close(parent_fd); + free(leaf); + return ok; +} + +bool file_rename_secure(const char* old_path, const char* new_path) { + char *old_leaf = NULL, *new_leaf = NULL; + int old_parent = file_open_secure_parent(old_path, &old_leaf, false); + int new_parent = file_open_secure_parent(new_path, &new_leaf, true); + bool ok = old_parent >= 0 && new_parent >= 0 && + renameat(old_parent, old_leaf, new_parent, new_leaf) == 0; + if (old_parent >= 0) + close(old_parent); + if (new_parent >= 0) + close(new_parent); + free(old_leaf); + free(new_leaf); + return ok; +} + +/* Recursively delete every entry inside an open directory, never following a + symlink (an O_NOFOLLOW fd walk, so a symlink planted inside the tree can + never redirect removal outside of it). The directory itself is left in + place; returns false on any failure. */ +static bool wipe_dir_fd(int dirfd) { + int scanfd = dup(dirfd); + if (scanfd < 0) + return false; + DIR* dir = fdopendir(scanfd); + if (!dir) { + close(scanfd); + return false; + } + bool operation_ok = true; + const struct dirent* entry; + while ((entry = readdir(dir)) != NULL) { + if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) + continue; + struct stat st; + if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) { + if (errno != ENOENT) + operation_ok = false; + continue; + } + if (S_ISDIR(st.st_mode)) { + int childfd = openat(dirfd, entry->d_name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + bool child_removed = false; + if (childfd >= 0) { + child_removed = wipe_dir_fd(childfd); + close(childfd); + } else if (errno != ENOENT) { + operation_ok = false; + } + if (child_removed && unlinkat(dirfd, entry->d_name, AT_REMOVEDIR) != 0 && errno != ENOENT) + operation_ok = false; + } else { + /* Files and symlinks alike are removed by name, never followed. */ + if (unlinkat(dirfd, entry->d_name, 0) != 0 && errno != ENOENT) + operation_ok = false; + } + } + closedir(dir); + return operation_ok; +} + +/* Remove the whole directory tree at `path` (confined below the authorized + root, symlink-safe). --force uses this to clear a non-empty destination + directory that blocks an incoming regular file. Returns true when the path + no longer exists as a directory (a missing path or a non-directory at the + final component is a no-op success; the normal write path replaces files). */ +bool file_remove_tree_secure(const char* path) { + if (!path) + return false; + char* leaf = NULL; + int parent_fd = file_open_secure_parent(path, &leaf, false); + if (parent_fd < 0) + return false; + int dirfd = openat(parent_fd, leaf, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + if (dirfd < 0) { + bool absent = errno == ENOENT || errno == ENOTDIR || errno == ELOOP; + close(parent_fd); + free(leaf); + return absent; + } + bool ok = wipe_dir_fd(dirfd); + close(dirfd); + if (ok && unlinkat(parent_fd, leaf, AT_REMOVEDIR) != 0 && errno != ENOENT) + ok = false; + close(parent_fd); + free(leaf); + return ok; +} + +/* Open a private staging/scratch directory, creating it (and any missing path + components) on demand. dir_path is expected to already be confined below + the authorized root by the caller; file_open_secure_parent re-checks that + confinement and rejects `..` components, so a scratch directory can never be + created or opened outside the destination root. The directory itself is + created 0700 so other users cannot race on names inside it. Returns an + O_DIRECTORY|O_NOFOLLOW fd, or -1 on error. */ +int file_open_private_dir(const char* dir_path) { + if (!dir_path) return -1; - return 0; + char* leaf = NULL; + int parent_fd = file_open_secure_parent(dir_path, &leaf, true); + if (parent_fd < 0) + return -1; + int fd = openat(parent_fd, leaf, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + if (fd < 0 && errno == ENOENT) { + if (mkdirat(parent_fd, leaf, 0700) == 0 || errno == EEXIST) + fd = openat(parent_fd, leaf, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + } + close(parent_fd); + free(leaf); + return fd; +} + +/* After the content and mode/times are restored on the just-written file, apply + * the per-file xattrs (-X/-A) and, for --fake-super, park the source's + * uid/gid/mode/mtime in the reserved xattr. All fd-relative (confined to the + * destination file) and best-effort: a per-attribute or privilege failure is + * logged and skipped, never fatal. */ +static void restore_extra_fd(int fd, const FileMetadata* metadata, const FileXattrList* xattrs, + bool fake_super) { + xattr_apply_fd(fd, xattrs); + if (fake_super && metadata) { + fake_super_store_fd(fd, (uint32_t)metadata->uid, (uint32_t)metadata->gid, + (uint32_t)metadata->mode, metadata->mtime_sec, metadata->mtime_nsec); + /* Replay: re-apply the recorded uid/gid/mode/mtime fd-relative so a save + under --fake-super restores the attrs (when privileged) instead of only + recording them. Best-effort; fake_super_restore_fd silently skips a + non-root fchown EPERM/EACCES and never fatal. */ + fake_super_restore_fd(fd); + } +} + +static bool file_to_disk_secure_impl(const char* path, const void* data, + unsigned long long data_size, bool inplace, bool sparse, + bool preallocate, const FileMetadata* metadata, + bool preserve_executability, bool update, bool no_replace, + bool use_fsync, const char* temp_dir, + const FileXattrList* xattrs, bool fake_super, + bool keep_partial) { + char* leaf = NULL; + int dirfd = file_open_secure_parent(path, &leaf, true); + if (dirfd < 0) + return false; + int fd = -1; + bool ok = false; + if (inplace) { + /* --inplace writes directly into the destination; a scratch --temp-dir + does not apply and must never redirect these writes. */ + fd = openat(dirfd, leaf, O_WRONLY | O_CREAT | O_CLOEXEC | O_NOFOLLOW, 0644); + if (fd >= 0) { + struct stat destination_stat; + bool newer = false; + if (update && metadata && fstat(fd, &destination_stat) == 0 && + S_ISREG(destination_stat.st_mode)) { + newer = stat_is_newer(&destination_stat, metadata); + } + if (newer) { + ok = true; + } else { + /* Preallocate the expected payload size before writing so an + out-of-space condition fails cleanly up front (--preallocate). + --sparse takes precedence: posix_fallocate would allocate every + block, defeating the holes the sparse writer would create, so the + two never combine here (the ftruncate presize below stays). */ + int prealloc_rc = 0; + if (preallocate && !sparse && data_size > 0) { + prealloc_rc = preallocate_fd(fd, data_size); + if (prealloc_rc != 0) { + char* escaped_path = output_escape(path, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "preallocate failed for '%s' (%s); transfer aborted", + escaped_path ? escaped_path : "", strerror(prealloc_rc)); + free(escaped_path); + } + } + if (prealloc_rc == 0) { + /* posix_fallocate does not guarantee the fd's file offset is left + unchanged, so seek back to 0 before the data write. */ + lseek(fd, 0, SEEK_SET); + if (sparse && data_size > 0) + ok = ftruncate(fd, (off_t)data_size) == 0; + if (ok || !sparse || data_size == 0) + ok = sparse && data_size > 0 + ? file_store_write_sparse(fd, (const unsigned char*)data, data_size) + : write_all(fd, data, data_size); + if (ok) + ok = ftruncate(fd, (off_t)data_size) == 0; + /* Normalize the mode: apply the metadata-derived safe mode when the + sender supplied metadata (setuid/setgid/sticky are never honored); + otherwise fall back to a safe default so dangerous bits on an + existing destination cannot survive an overwrite. */ + if (ok) { + if (metadata) + ok = file_restore_metadata_fd(fd, metadata, preserve_executability); + else if (fchmod(fd, S_IRUSR | S_IWUSR | S_IRGRP | S_IROTH) != 0) + ok = false; + } + if (ok) + restore_extra_fd(fd, metadata, xattrs, fake_super); + if (ok && use_fsync) + ok = fsync(fd) == 0; + } + } + } + } else { + /* The --update newer-destination check runs first so a skipped file never + creates an empty scratch directory behind it. */ + /* True once the temp is being written: distinguishes a mid-write/metadata/ + install failure (partial data may exist, --partial may retain it) from a + pre-write validation failure (nothing to retain). */ + bool write_attempted = false; + if (update && metadata) { + /* This check protects the normal atomic path as far as possible. A + concurrent replacement can still occur before the final rename. */ + struct stat destination_stat; + if (fstatat(dirfd, leaf, &destination_stat, AT_SYMLINK_NOFOLLOW) == 0 && + S_ISREG(destination_stat.st_mode) && stat_is_newer(&destination_stat, metadata)) { + close(dirfd); + free(leaf); + return true; + } + } + /* Scratch directory for the temporary working copy. When NULL the temp + file is created in the destination directory, exactly as historically. */ + int scratch_dirfd = -1; + if (temp_dir) { + scratch_dirfd = file_open_private_dir(temp_dir); + if (scratch_dirfd < 0) { + int saved_errno = errno; + log_message(LOG_LEVEL_ERROR, "could not open --temp-dir scratch directory '%s': %s", + temp_dir, strerror(saved_errno)); + close(dirfd); + free(leaf); + return false; + } + } + /* Temp names can exceed NAME_MAX for basenames near the limit (leaf plus + the ".tmp.." decoration); heap-size the buffer instead of + truncating into a fixed array, which would silently collide in a flat + scratch directory. The sizing sentinel is the widest value of each + format. */ + int tmp_size; + if (scratch_dirfd >= 0) + tmp_size = snprintf(NULL, 0, ".%s.tmp.%ld.%llu", leaf, (long)getpid(), ~0ULL); + else + tmp_size = snprintf(NULL, 0, ".%s.tmp.%ld.%u", leaf, (long)getpid(), 999U); + if (tmp_size < 0) { + if (scratch_dirfd >= 0) + close(scratch_dirfd); + close(dirfd); + free(leaf); + return false; + } + char* tmp = malloc((size_t)tmp_size + 1); + if (!tmp) { + if (scratch_dirfd >= 0) + close(scratch_dirfd); + close(dirfd); + free(leaf); + return false; + } + for (unsigned int i = 0; i < 100; ++i) { + /* The temp name is created inside the scratch directory (when one is + configured) and, on success, atomically renamed into the destination + directory. In a shared scratch directory the atomic sequence number + keeps the name unique even for destinations with a common basename. */ + if (scratch_dirfd >= 0) + snprintf(tmp, (size_t)tmp_size + 1, ".%s.tmp.%ld.%llu", leaf, (long)getpid(), + next_temp_sequence()); + else + snprintf(tmp, (size_t)tmp_size + 1, ".%s.tmp.%ld.%u", leaf, (long)getpid(), i); + fd = openat(scratch_dirfd >= 0 ? scratch_dirfd : dirfd, tmp, + O_WRONLY | O_CREAT | O_EXCL | O_CLOEXEC | O_NOFOLLOW, 0600); + if (fd < 0) + continue; /* EEXIST (or a transient open error): try a fresh name. */ + int prealloc_rc = 0; + if (preallocate && !sparse && data_size > 0) { + prealloc_rc = preallocate_fd(fd, data_size); + if (prealloc_rc != 0) { + char* escaped_path = output_escape(path, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "preallocate failed for '%s' (%s); transfer aborted", + escaped_path ? escaped_path : "", strerror(prealloc_rc)); + free(escaped_path); + } + } + if (prealloc_rc == 0) { + lseek(fd, 0, SEEK_SET); + if (sparse && data_size > 0) + ok = ftruncate(fd, (off_t)data_size) == 0; + /* A real write attempt begins here (the ftruncate presize succeeded or + no presize applies): a later mid-write / metadata / fsync / install + failure may leave partial data that --partial retention can rename. */ + if (ok || (!sparse || data_size == 0)) { + write_attempted = true; + ok = sparse && data_size > 0 + ? file_store_write_sparse(fd, (const unsigned char*)data, data_size) + : write_all(fd, data, data_size); + } + if (ok && metadata) + ok = file_restore_metadata_fd(fd, metadata, preserve_executability); + if (ok) + restore_extra_fd(fd, metadata, xattrs, fake_super); + if (ok && use_fsync) + ok = fsync(fd) == 0; + } + if (close(fd) != 0) + ok = false; + fd = -1; + if (ok) { + if (no_replace) { + /* The probe and commit cannot be one operation. A concurrent + creator may win; EEXIST is then the requested skip. */ + if (linkat(scratch_dirfd >= 0 ? scratch_dirfd : dirfd, tmp, dirfd, leaf, 0) == 0 || + errno == EEXIST) { + if (unlinkat(scratch_dirfd >= 0 ? scratch_dirfd : dirfd, tmp, 0) != 0 && + errno != ENOENT) + ok = false; + } else { + if (scratch_dirfd >= 0 && errno == EXDEV) + log_message(LOG_LEVEL_ERROR, + "temp dir is on a different filesystem than the destination; cannot " + "link file into place (EXDEV); no fallback copy is attempted"); + ok = false; + } + } else if (renameat(scratch_dirfd >= 0 ? scratch_dirfd : dirfd, tmp, dirfd, leaf) != 0) { + if (scratch_dirfd >= 0 && errno == EXDEV) + log_message(LOG_LEVEL_ERROR, + "temp dir is on a different filesystem than the destination; cannot " + "atomically install file (EXDEV); no fallback copy is attempted"); + ok = false; + } + } + if (!ok) { + /* --partial retention (best-effort): on a failure that happened after + the temp held data (mid-write / metadata / fsync / install error), + keep the already-written temp at the final destination path instead + of unlinking it, so a later --append / --append-verify run can resume. + This only ever renames the already-written temp (never a corrupt + blend); the rename can fail (cross-device, permissions) and we then + fall through to the normal unlink cleanup. Never retains when + keep_partial is off, when nothing was actually written, or under + --ignore-existing/--existing (no_replace), where the destination is + not ours to overwrite. */ + if (!keep_partial || !write_attempted || no_replace || + renameat(scratch_dirfd >= 0 ? scratch_dirfd : dirfd, tmp, dirfd, leaf) != 0) + unlinkat(scratch_dirfd >= 0 ? scratch_dirfd : dirfd, tmp, 0); + } + /* Once the temp fd was created the outcome is permanent: a write, + metadata, fsync, close, linkat or renameat failure will not be fixed + by retrying under a fresh name, so stop here. Only the open-failure + path above retries a new name. */ + break; + } + free(tmp); + if (scratch_dirfd >= 0) + close(scratch_dirfd); + } + if (fd >= 0) + close(fd); + close(dirfd); + free(leaf); + return ok; +} + +bool file_to_disk_secure(const char* path, const void* data, unsigned long long data_size, + bool inplace, bool sparse, bool preallocate, const FileMetadata* metadata, + bool preserve_executability, const char* temp_dir) { + return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata, + preserve_executability, false, false, false, temp_dir, NULL, + false, false); +} + +bool file_to_disk_secure_update(const char* path, const void* data, unsigned long long data_size, + bool inplace, bool sparse, bool preallocate, + const FileMetadata* metadata, bool preserve_executability, + const char* temp_dir) { + return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata, + preserve_executability, true, false, false, temp_dir, NULL, false, + false); +} + +bool file_to_disk_secure_with_fsync(const char* path, const void* data, + unsigned long long data_size, bool inplace, bool sparse, + bool preallocate, const FileMetadata* metadata, + bool preserve_executability, bool use_fsync, + const char* temp_dir) { + return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata, + preserve_executability, false, false, use_fsync, temp_dir, NULL, + false, false); +} + +bool file_to_disk_secure_no_replace(const char* path, const void* data, + unsigned long long data_size, bool sparse, bool preallocate, + const FileMetadata* metadata, bool preserve_executability, + const char* temp_dir) { + return file_to_disk_secure_impl(path, data, data_size, false, sparse, preallocate, metadata, + preserve_executability, false, true, false, temp_dir, NULL, false, + false); +} + +/* Receiver write-path variant that also applies the per-file xattrs (-X/-A) + * and, under --fake-super, parks the source stat in the reserved xattr, on the + * just-written file descriptor before the final rename. `no_replace` / `update` + * mirror the plain wrappers; `keep_partial` enables --partial retention of a + * failed write's temp. See file_to_disk_secure_impl for the semantics. */ +bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long long data_size, + bool inplace, bool sparse, bool preallocate, + const FileMetadata* metadata, bool preserve_executability, + bool update, bool no_replace, bool use_fsync, + const FileXattrList* xattrs, bool fake_super, bool keep_partial, + const char* temp_dir) { + return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata, + preserve_executability, update, no_replace, use_fsync, temp_dir, + xattrs, fake_super, keep_partial); +} + +/* Atomic --link-dest install. The destination is replaced (via a temporary + * name and a final rename) with a hard link to `basis_path`. When a hard + * link cannot be created (the basis lives on a different filesystem, the + * filesystem refuses hard links, ...) the install falls back to writing a + * local copy from `data`/`data_size`, which the caller has already verified is + * byte-identical to the basis file. `metadata` is only applied on that copy + * fallback; a successful hard link keeps the basis inode's own attributes + * (applying metadata through the shared inode would mutate the basis file). + * Returns false only when both the link and the copy fallback fail. */ +/* --link-dest / -H hardlink install with a byte-copy fallback. `metadata` is + * applied only on the copy fallback; a successful hard link keeps the basis + * inode's own attributes (applying through the shared inode would mutate the + * basis). Likewise `xattrs`/`fake_super` are applied only on the copy + * fallback, so a fallback copy preserves the per-file attributes instead of + * silently dropping them. */ +static bool file_to_disk_secure_link_impl(const char* path, const char* basis_path, + const void* data, unsigned long long data_size, + bool preallocate, const FileMetadata* metadata, + bool preserve_executability, bool use_fsync, + const FileXattrList* xattrs, bool fake_super, + const char* temp_dir) { + if (!path || !basis_path) + return false; + char* leaf = NULL; + int dirfd = file_open_secure_parent(path, &leaf, true); + if (dirfd < 0) + return false; + + int scratch_dirfd = -1; + if (temp_dir) { + scratch_dirfd = file_open_private_dir(temp_dir); + if (scratch_dirfd < 0) { + int saved_errno = errno; + log_message(LOG_LEVEL_ERROR, "could not open --temp-dir scratch directory '%s': %s", temp_dir, + strerror(saved_errno)); + close(dirfd); + free(leaf); + return false; + } + } + + char* basis_leaf = NULL; + int basis_dirfd = file_open_secure_parent(basis_path, &basis_leaf, false); + bool linked = false; + if (basis_dirfd >= 0 && basis_leaf != NULL) { + int tmp_size = snprintf(NULL, 0, ".%s.tmp.%ld.%llu", leaf, (long)getpid(), ~0ULL); + char* tmp = NULL; + if (tmp_size >= 0) + tmp = malloc((size_t)tmp_size + 1); + if (!tmp) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed while hard-linking basis file"); + } else { + for (unsigned int i = 0; i < 100 && !linked; ++i) { + if (scratch_dirfd >= 0) + snprintf(tmp, (size_t)tmp_size + 1, ".%s.tmp.%ld.%llu", leaf, (long)getpid(), + next_temp_sequence()); + else + snprintf(tmp, (size_t)tmp_size + 1, ".%s.tmp.%ld.%u", leaf, (long)getpid(), i); + if (linkat(basis_dirfd, basis_leaf, scratch_dirfd >= 0 ? scratch_dirfd : dirfd, tmp, 0) == + 0) { + linked = true; + break; + } + if (errno != EEXIST) + break; /* EXDEV / EPERM / ...: give up and fall back to a copy */ + } + if (linked) { + int target_dirfd = scratch_dirfd >= 0 ? scratch_dirfd : dirfd; + if (use_fsync) { + int tfd = openat(target_dirfd, tmp, O_RDONLY | O_NOFOLLOW | O_CLOEXEC); + if (tfd < 0 || fsync(tfd) != 0) { + linked = false; + if (tfd >= 0) + close(tfd); + } else { + close(tfd); + } + } + if (linked && renameat(target_dirfd, tmp, dirfd, leaf) != 0) + linked = false; + if (!linked) + unlinkat(target_dirfd, tmp, 0); + } + free(tmp); + } + } + if (basis_dirfd >= 0) + close(basis_dirfd); + free(basis_leaf); + basis_leaf = NULL; + + if (!linked) { + if (scratch_dirfd >= 0) + close(scratch_dirfd); + close(dirfd); + free(leaf); + /* The basis file could not be linked in (missing, cross-device, refused + by the filesystem). Write a byte-identical local copy instead. */ + return file_to_disk_secure_attrs(path, data, data_size, false, false, preallocate, metadata, + preserve_executability, false, false, use_fsync, xattrs, + fake_super, false, temp_dir); + } + + if (scratch_dirfd >= 0) + close(scratch_dirfd); + close(dirfd); + free(leaf); + return true; +} + +bool file_to_disk_secure_link(const char* path, const char* basis_path, const void* data, + unsigned long long data_size, bool preallocate, + const FileMetadata* metadata, bool preserve_executability, + bool use_fsync, const char* temp_dir) { + return file_to_disk_secure_link_impl(path, basis_path, data, data_size, preallocate, metadata, + preserve_executability, use_fsync, NULL, false, temp_dir); +} + +bool file_to_disk_secure_link_attrs(const char* path, const char* basis_path, const void* data, + unsigned long long data_size, bool preallocate, + const FileMetadata* metadata, bool preserve_executability, + bool use_fsync, const FileXattrList* xattrs, bool fake_super, + const char* temp_dir) { + return file_to_disk_secure_link_impl(path, basis_path, data, data_size, preallocate, metadata, + preserve_executability, use_fsync, xattrs, fake_super, + temp_dir); +} + +bool file_write_to_disk(const char* path, const void* data, unsigned long long data_size, + bool inplace, bool sparse) { + if (!path || (!data && data_size != 0) || has_path_traversal(path)) + return false; + return file_to_disk_secure(path, data, data_size, inplace, sparse, false, NULL, false, NULL); } diff --git a/src/shared/file.h b/src/shared/file.h index e3e514b..554170b 100644 --- a/src/shared/file.h +++ b/src/shared/file.h @@ -1,42 +1,144 @@ #ifndef FILE_H #define FILE_H -#include "config.h" -#include "data.h" +#include "file_send.h" +#include "file_receive.h" +#include "file_types.h" +#include "checksum.h" #include +#include #include -typedef enum { FILE_TYPE_REGULAR, FILE_TYPE_SYMLINK, FILE_TYPE_DIR } FileType; - -typedef struct { - mode_t mode; - uid_t uid; - gid_t gid; - time_t mtime_sec; - long mtime_nsec; -} FileMetadata; - -typedef struct { - char* path; - Data* data; - FileMetadata* metadata; - bool skip; -} File; +/* File/FileMetadata lifecycle, local disk helpers, and secure filesystem + primitives shared by the send/receive pipelines. */ File* file_create(const char* path); void file_destroy(void* item); bool file_load_data(File* file); -File* file_receive(const Config* config, int file_descriptor); -bool file_send_single_calls(File* file, int file_descriptor, bool use_metadata, - int compression_level, bool send_path); -bool file_send_sendfile(File* file, int file_descriptor, bool use_metadata, int compression_level, - bool send_path); +/* Compute the whole-file content digest of `file` with the negotiated + * --checksum-choice algorithm and --checksum-seed. Writes the digest into + * `out` (capacity `out_capacity`) and its length into `*out_len`. Returns + * false on read/allocation failure or when the digest would not fit. */ +bool file_checksum(File* file, ChecksumAlgo algo, uint64_t seed, uint8_t* out, size_t out_capacity, + size_t* out_len); size_t file_content_to_buffer(File* file); -FileMetadata* file_metadata_create(const struct stat* stats); +FileMetadata* file_metadata_create(const char* path, const struct stat* stats, bool capture_atime, + bool capture_crtime); void file_metadata_destroy(void* metadata); -bool to_disk(const char* path, const void* data, unsigned long long data_size); -bool file_save_to_disk(const char* root_directory, File* file, const Config* config); -File* receive_incremental_check(int fd, const Config* config, bool* skipped); -int receive_manifest(int fd, const Config* config, int* next_status); +/* --open-noatime process-wide sender policy; see file.c. */ +void file_set_open_noatime(bool enable); +bool file_get_open_noatime(void); +/* Open `path` read-only for transfer, honouring --open-noatime when set. */ +int file_open_for_read(const char* path); +bool file_write_to_disk(const char* path, const void* data, unsigned long long data_size, + bool inplace, bool sparse); + +/* Symlink trust-boundary helpers (Phase 4, symlink wave). --munge-links + * sender-side marker: every transmitted symlink target is prefixed with this + * while the flag is on; the receiver strips it to restore the real target. */ +#define SYMLINK_MUNGE_PREFIX "#SYMLINK/" + +char* file_symlink_munge(const char* target); +/* True when a lexical target is relative and contains no ".." component, so it + * can never escape the receive root once created beneath it. */ +bool file_symlink_target_contained(const char* target); +/* Strip a leading SYMLINK_MUNGE_PREFIX from `target` (mutable, in place); + * returns true when a marker was removed. */ +bool file_symlink_unmunge(char* target); +/* Create a symlink at `path` -> `target`, confined below the authorized root + * (O_NOFOLLOW parent walk, symlinkat; the target is never followed). Returns + * false when a directory already occupies `path`. */ +bool file_symlink_at_secure(const char* path, const char* target); +/* --keep-dirlinks (-K) receiver process-wide policy: allow an in-root existing + * symlink-to-directory to be followed as a directory. */ +void file_set_keep_dirlinks(bool enable); +bool file_get_keep_dirlinks(void); + +/* --trust-sender receiver process-wide policy (Phase 5). When set, the + * receiver trusts that the sender already produced a clean file list and skips + * its own redundant up-front re-validation of incoming paths (the empty/".." + * rejection and the escaping-symlink-target containment). The low-level + * fd-relative confinement primitives below are deliberately NOT disabled by + * this flag, so a hostile sender still cannot escape the authorized root. */ +void file_set_trust_sender(bool enable); +bool file_get_trust_sender(void); + +/* A configured fd without a canonical identity deliberately rejects paths. */ +bool file_set_authorized_root(int fd, const char* canonical_path); + +/* Secure path/filesystem primitives (symlink-safe, O_NOFOLLOW, root-confined). */ +bool file_path_exists_secure(const char* path); +bool file_stat_secure(const char* path, struct stat* st); +bool file_destination_is_newer_secure(const char* path, const FileMetadata* metadata); +int file_open_secure_parent(const char* path, char** leaf_out, bool create_dirs); +bool file_ensure_directory_secure(const char* path); +bool file_directory_exists_secure(const char* path); +bool file_rename_secure(const char* old_path, const char* new_path); +/* Remove the whole directory tree at `path` (confined, symlink-safe). Used by + --force to clear a non-empty destination directory that blocks an incoming + regular file. See the .c for the exact success semantics. */ +bool file_remove_tree_secure(const char* path); +/* Open a private 0700 directory (creating it on demand) that must live below + the authorized root. Used for the --temp-dir scratch directory and the + --delay-updates staging directory. */ +int file_open_private_dir(const char* dir_path); + +/* The file_to_disk_secure* variants write a temporary copy in the destination + directory and atomically rename it over `path`. temp_dir is an absolute, + root-confined scratch directory (already validated by the caller): when it + is non-NULL the temporary copy is instead created there (with a name unique + across the whole scratch directory) and atomically renamed into the + destination directory once fully written and fsynced. A rename across + filesystems (EXDEV) fails the write with an error; the file is never + silently copied into place. Pass NULL for the historical same-directory + behavior. --inplace writes never use temp_dir. */ +bool file_to_disk_secure(const char* path, const void* data, unsigned long long data_size, + bool inplace, bool sparse, bool preallocate, const FileMetadata* metadata, + bool preserve_executability, const char* temp_dir); +bool file_to_disk_secure_with_fsync(const char* path, const void* data, + unsigned long long data_size, bool inplace, bool sparse, + bool preallocate, const FileMetadata* metadata, + bool preserve_executability, bool use_fsync, + const char* temp_dir); +/* With update enabled, an existing newer destination is left untouched. The + check is descriptor-based for inplace writes; atomic replacement still has + an unavoidable final rename race without filesystem locking. */ +bool file_to_disk_secure_update(const char* path, const void* data, unsigned long long data_size, + bool inplace, bool sparse, bool preallocate, + const FileMetadata* metadata, bool preserve_executability, + const char* temp_dir); +bool file_to_disk_secure_no_replace(const char* path, const void* data, + unsigned long long data_size, bool sparse, bool preallocate, + const FileMetadata* metadata, bool preserve_executability, + const char* temp_dir); +/* Receiver write-path variant that also applies per-file xattrs (-X/-A) and the + * --fake-super stat xattr fd-relative before the final rename. `update` / + * `no_replace` / `use_fsync` mirror the plain wrappers above; `keep_partial` + * enables --partial best-effort retention of a failed write's temp. */ +bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long long data_size, + bool inplace, bool sparse, bool preallocate, + const FileMetadata* metadata, bool preserve_executability, + bool update, bool no_replace, bool use_fsync, + const FileXattrList* xattrs, bool fake_super, bool keep_partial, + const char* temp_dir); +/* Atomic --link-dest install: replace `path` with a hard link to `basis_path` + (via a temp name + rename); fall back to a byte-identical local copy from + `data` when the link is impossible (EXDEV/EPERM/unsupported filesystem). + `metadata` is applied only on the copy fallback. `preallocate` applies to + that copy fallback only (a hard-linked file shares the basis inode and is + never re-allocated). */ +bool file_to_disk_secure_link(const char* path, const char* basis_path, const void* data, + unsigned long long data_size, bool preallocate, + const FileMetadata* metadata, bool preserve_executability, + bool use_fsync, const char* temp_dir); +/* Like file_to_disk_secure_link, but the byte-copy fallback also applies the + * per-file xattrs (-X/-A) and --fake-super stat xattr (fd-relative). On a + * successful hard link no attributes are applied (the shared inode already + * carries the basis's). */ +bool file_to_disk_secure_link_attrs(const char* path, const char* basis_path, const void* data, + unsigned long long data_size, bool preallocate, + const FileMetadata* metadata, bool preserve_executability, + bool use_fsync, const FileXattrList* xattrs, bool fake_super, + const char* temp_dir); #endif diff --git a/src/shared/file_list.c b/src/shared/file_list.c new file mode 100644 index 0000000..8e40c78 --- /dev/null +++ b/src/shared/file_list.c @@ -0,0 +1,191 @@ +#include "file_list.h" +#include "log.h" +#include "utils.h" +#include +#include +#include +#include + +typedef struct { + char** items; + int count; + int capacity; +} StringList; + +static void string_list_destroy(StringList* list) { + if (!list) + return; + for (int i = 0; i < list->count; i++) + free(list->items[i]); + free(list->items); +} + +static bool string_list_add(StringList* list, const char* text) { + if (list->count == list->capacity) { + int new_cap = list->capacity > 0 ? list->capacity * 2 : 16; + char** grown = realloc(list->items, (size_t)new_cap * sizeof(char*)); + if (!grown) + return false; + list->items = grown; + list->capacity = new_cap; + } + list->items[list->count] = str_dup(text); + if (!list->items[list->count]) + return false; + list->count++; + return true; +} + +/* Validate and normalize one entry. Returns: + * 1 -> added to `out` + * 0 -> blank entry, skip + * -1 -> invalid (message set in `err`) + * `strip_line_endings` trims a trailing CR/LF (line mode only); NUL mode keeps + * the entry bytes verbatim so names ending in CR/LF survive. */ +static int normalize_entry(const char* raw, size_t len, bool strip_line_endings, StringList* out, + char* err, size_t err_size) { + if (strip_line_endings) { + while (len > 0 && (raw[len - 1] == '\n' || raw[len - 1] == '\r')) + len--; + } + if (len == 0) + return 0; + if (raw[0] == '/') { + snprintf(err, err_size, "absolute path entries are not allowed: '%.*s'", (int)len, raw); + return -1; + } + /* Reject NUL bytes inside a token defensively (NUL-delimited mode splits on + * them, so this only guards against embedded garbage). */ + char* dup = malloc(len + 1); + if (!dup) { + snprintf(err, err_size, "memory allocation failed"); + return -1; + } + memcpy(dup, raw, len); + dup[len] = '\0'; + + /* Rebuild the path token-by-token: skip '.' and empty segments, reject '..'. */ + size_t out_len = 0; + for (const char* part = dup;;) { + const char* slash = strchr(part, '/'); + size_t part_len = slash ? (size_t)(slash - part) : strlen(part); + if (part_len == 1 && part[0] == '.') { + /* skip "." segment */ + } else if (part_len == 2 && part[0] == '.' && part[1] == '.') { + snprintf(err, err_size, "path traversal entry is not allowed: '%s'", dup); + free(dup); + return -1; + } else if (part_len > 0) { + if (out_len > 0) + dup[out_len++] = '/'; + memmove(dup + out_len, part, part_len); + out_len += part_len; + } + if (!slash) + break; + part = slash + 1; + } + dup[out_len] = '\0'; + + int result; + if (out_len == 0) { + /* "." / "./" lists the source root: the whole tree is transferred. */ + result = string_list_add(out, "") ? 1 : -1; + if (result < 0) + snprintf(err, err_size, "memory allocation failed"); + } else { + result = string_list_add(out, dup) ? 1 : -1; + if (result < 0) + snprintf(err, err_size, "memory allocation failed"); + } + free(dup); + return result; +} + +static FileListSet* string_list_to_set(StringList* raw, char* err, size_t err_size) { + FileListSet* set = malloc(sizeof(FileListSet)); + if (!set) { + snprintf(err, err_size, "memory allocation failed"); + return NULL; + } + set->count = raw->count; + set->entries = raw->items; + raw->items = NULL; + raw->count = 0; + return set; +} + +FileListSet* file_list_load(const char* path, bool null_separated, char* err, size_t err_size) { + if (err && err_size > 0) + err[0] = '\0'; + if (!path || !*path) { + snprintf(err, err_size, "no file given"); + return NULL; + } + FILE* fp = fopen(path, "r"); + if (!fp) { + char* escaped = output_escape(path, false); + snprintf(err, err_size, "could not open '%s': %s", escaped ? escaped : path, strerror(errno)); + free(escaped); + return NULL; + } + + StringList raw = {0}; + char* line = NULL; + size_t line_cap = 0; + ssize_t n; + bool ok = true; + char delim = null_separated ? '\0' : '\n'; + while (ok && (n = getdelim(&line, &line_cap, delim, fp)) != -1) { + int r = normalize_entry(line, (size_t)n, !null_separated, &raw, err, err_size); + if (r < 0) { + ok = false; + break; + } + } + free(line); + fclose(fp); + if (!ok) { + string_list_destroy(&raw); + return NULL; + } + FileListSet* set = string_list_to_set(&raw, err, err_size); + if (!set) + string_list_destroy(&raw); + return set; +} + +void file_list_destroy(FileListSet* set) { + if (!set) + return; + for (int i = 0; i < set->count; i++) + free(set->entries[i]); + free(set->entries); + free(set); +} + +static bool path_has_prefix(const char* path, const char* prefix) { + size_t plen = strlen(prefix); + if (strncmp(path, prefix, plen) != 0) + return false; + return path[plen] == '/' || path[plen] == '\0'; +} + +bool file_list_affects(const FileListSet* set, const char* rel) { + if (!set) + return true; + if (!rel) + return false; + for (int i = 0; i < set->count; i++) { + const char* entry = set->entries[i]; + if (entry[0] == '\0') + return true; /* whole tree listed */ + if (strcmp(rel, entry) == 0) + return true; /* the entry itself is listed */ + if (path_has_prefix(rel, entry)) + return true; /* rel lives under a listed directory */ + if (path_has_prefix(entry, rel)) + return true; /* rel is an ancestor directory of a listed entry */ + } + return false; +} diff --git a/src/shared/file_list.h b/src/shared/file_list.h new file mode 100644 index 0000000..0ced332 --- /dev/null +++ b/src/shared/file_list.h @@ -0,0 +1,35 @@ +#ifndef FILE_LIST_H +#define FILE_LIST_H + +#include +#include + +/* --files-from allow-set. The file lists source paths RELATIVE to the source + * root. A listed regular file is transferred; a listed directory transfers its + * whole subtree (FastSync's recursion is always on). Blank lines are ignored. + * + * Entries are normalized: leading "./" and duplicate "/" are removed, an entry + * of "." means the whole tree, absolute entries and ".." traversal are + * rejected at parse time. The set is immutable and shared read-only across + * scanner worker threads. + */ + +typedef struct { + char** entries; /* normalized rel paths; "" means the whole tree */ + int count; +} FileListSet; + +/* Load and validate a --files-from file. When `null_separated` (-0/--from0) + * entries are delimited by NUL instead of newlines. Returns NULL with a message + * in `err` on open/validation failure. An empty file yields an empty set + * (nothing is transferred). */ +FileListSet* file_list_load(const char* path, bool null_separated, char* err, size_t err_size); +void file_list_destroy(FileListSet* set); + +/* True when `rel` (path relative to the source root, "" == root) is a listed + * entry, lives under a listed directory, or is an ancestor directory of a + * listed entry. Used to prune scanning: directories are descended only when + * this returns true, files are transferred only when it returns true. */ +bool file_list_affects(const FileListSet* set, const char* rel); + +#endif diff --git a/src/shared/file_receive.c b/src/shared/file_receive.c new file mode 100644 index 0000000..d29bf77 --- /dev/null +++ b/src/shared/file_receive.c @@ -0,0 +1,2880 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "array_list.h" +#include "charset.h" +#include "chmod.h" +#include "compression.h" +#include "config.h" +#include "data.h" +#include "delay_updates.h" +#include "delta.h" +#include "file.h" +#include "identity.h" +#include "log.h" +#include "metadata.h" +#include "protocol.h" +#include "utils.h" +#include "xattr.h" + +#define MAX_SERVER_DELETE_COUNT 100000U +#define MAX_FILE_DATA_SIZE MAX_RECEIVE_WHOLE_FILE_SIZE + +bool file_save_to_disk(const char* root_directory, const File* file, const Config* config) { + return file_save_to_disk_full(root_directory, file, config) != FILE_SAVE_ERROR; +} + +/* --delay-updates receiver path: write the file into a private staging tree + below the receive root instead of its final destination, and remember it so + it can be atomically renamed into place only once the whole transfer has + succeeded. Existence/update policies (--existing/--ignore-existing/--update) + are decided against the FINAL destination path at stage time so the run + decides exactly what an immediate (non-delayed) run would decide; the staged + file is then never re-checked at publication. Backups are deferred to + publication so the final destination is untouched until the transfer ends. */ +static FileSaveResult file_stage_delayed_update(const char* root_directory, + const char* destination_path, const File* file, + Config* config) { + if (!config) + return FILE_SAVE_ERROR; + bool sparse = config->preserve_sparse; + bool preserve_executability = config->use_executability; + + if (config->existing && !file_path_exists_secure(destination_path)) + return FILE_SAVE_SKIPPED; + if (config->ignore_existing && file_path_exists_secure(destination_path)) + return FILE_SAVE_SKIPPED; + if (config->update && file_destination_is_newer_secure(destination_path, file->metadata)) + return FILE_SAVE_SKIPPED; + + FileMetadata adjusted_metadata; + const FileMetadata* metadata = file->metadata; + if (metadata && config->chmod_spec && *config->chmod_spec) { + adjusted_metadata = *metadata; + if (!chmod_apply(adjusted_metadata.mode, config->chmod_spec, &adjusted_metadata.mode)) + return FILE_SAVE_ERROR; + metadata = &adjusted_metadata; + } + + if (!config->delay_context) { + config->delay_context = delay_updates_context_create(root_directory); + if (!config->delay_context) + return FILE_SAVE_ERROR; + } + DelayUpdatesContext* context = config->delay_context; + if (!delay_updates_prepare(context)) + return FILE_SAVE_ERROR; + + char* staged_path = path_cat(context->staging_root, file->path); + if (!staged_path) + return FILE_SAVE_ERROR; + + /* The staged location is brand new (stale leftovers from a prior crash were + wiped by prepare), so the plain atomic temp+rename engine installs the + complete file there. --temp-dir scratch is deliberately not layered on + top of the delay-updates staging tree. A --link-dest basis file is hard + linked into the staging tree (so publication's rename keeps the link). */ + bool ok; + if (file->basis_link) { + ok = file_to_disk_secure_link(staged_path, file->basis_link, file->data->data, file->data->size, + config->preallocate, metadata, preserve_executability, + config->use_fsync, NULL); + } else { + ok = file_to_disk_secure_attrs(staged_path, file->data->data, file->data->size, false, sparse, + config->preallocate, metadata, preserve_executability, false, + false, config->use_fsync, file->xattrs, config->fake_super, + false, NULL); + } + if (!ok) { + free(staged_path); + return FILE_SAVE_ERROR; + } + + if (!delay_updates_record(context, staged_path, destination_path, file->path)) { + unlink(staged_path); + free(staged_path); + return FILE_SAVE_ERROR; + } + free(staged_path); + return FILE_SAVE_WRITTEN; +} + +/* Read the whole content of a confined regular file (used to fall back to a + byte-identical copy when a hard-link sibling's link() fails). Symlink-safe + (parent resolved via file_open_secure_parent + O_NOFOLLOW). A zero-length + file yields *out_size 0 and *out_buf NULL as a SUCCESS. Returns false only + on a real error/read failure, setting *source_absent to true when the reason + was that the path does not exist (ENOENT/ENOTDIR), so the caller can decide + between an abort and a graceful skip. */ +static bool hardlink_read_source(const char* path, void** out_buf, unsigned long long* out_size, + bool* source_absent) { + *out_buf = NULL; + *out_size = 0; + *source_absent = false; + if (!path) + return false; + char* leaf = NULL; + int parent_fd = file_open_secure_parent(path, &leaf, false); + if (parent_fd < 0) { + *source_absent = errno == ENOENT || errno == ENOTDIR; + return false; + } + int fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW); + int saved_errno = errno; + free(leaf); + close(parent_fd); + if (fd < 0) { + *source_absent = saved_errno == ENOENT || saved_errno == ENOTDIR; + return false; + } + struct stat st; + if (fstat(fd, &st) != 0 || !S_ISREG(st.st_mode)) { + close(fd); + return false; + } + unsigned long long size = (unsigned long long)st.st_size; + if (size > MAX_RECEIVE_WHOLE_FILE_SIZE || size > SIZE_MAX) { + close(fd); + return false; + } + if (size == 0) { + close(fd); + return true; + } + void* buf = protocol_alloc((size_t)size); + if (!buf) { + close(fd); + return false; + } + size_t got = 0; + while (got < (size_t)size) { + ssize_t n = read(fd, (char*)buf + got, (size_t)size - got); + if (n <= 0) { + free(buf); + close(fd); + return false; + } + got += (size_t)n; + } + close(fd); + *out_buf = buf; + *out_size = size; + return true; +} + +/* The group's first member's installed file is absent, but its destination + path was validated (a sibling is only ever processed after its group's first + member). When the sibling's OWN destination already exists it should be + left alone -- a clean skip -- rather than aborting the whole transfer (the + asymmetric --existing case: the first member was skipped because its + destination was missing, while the sibling already has one). Only when the + sibling's destination is missing too is this a genuine failure to + link/copy, which aborts. */ +static FileSaveResult hardlink_sibling_absent_first(const char* destination_path) { + if (destination_path && file_path_exists_secure(destination_path)) + return FILE_SAVE_SKIPPED; + return FILE_SAVE_ERROR; +} + +/* Install a --hard-links/-H sibling: the destination entry is atomically + replaced (temp + rename) with a hard link to the group's first member. The + first member is guaranteed already installed at `hardlink_target` under the + root because -H relies on the receiver's single-FIFO-writer pipeline (one + receive thread, one write thread, FIFO queue => wire order == write order) + plus the sender's forced sequential scan, so a sibling is always processed + after its group's first member. When link() fails (different filesystem, + filesystem refuses links) a byte-identical copy of the first member is + written instead, so the result is never partial or corrupt. With + --delay-updates the sibling is staged as a hard link to the first member's + STAGED file (publication's renames preserve the shared inode). The final + --existing/--ignore-existing/--update policies are decided against the final + destination like every normal write. */ +static FileSaveResult file_save_hardlink_sibling(const char* root_directory, const File* file, + const Config* config) { + Config* cfg = (Config*)config; + if (!root_directory || !file || !file->path || !file->hardlink_target) + return FILE_SAVE_ERROR; + char* destination_path = path_cat(root_directory, file->path); + if (!destination_path) + return FILE_SAVE_ERROR; + + if (cfg->existing && !file_path_exists_secure(destination_path)) { + free(destination_path); + return FILE_SAVE_SKIPPED; + } + if (cfg->ignore_existing && file_path_exists_secure(destination_path)) { + free(destination_path); + return FILE_SAVE_SKIPPED; + } + if (cfg->update && file_destination_is_newer_secure(destination_path, file->metadata)) { + free(destination_path); + return FILE_SAVE_SKIPPED; + } + + bool preallocate = cfg && cfg->preallocate; + bool preserve_executability = cfg && cfg->use_executability; + bool use_fsync = cfg && cfg->use_fsync; + + if (cfg->delay_updates) { + if (!cfg->delay_context) { + cfg->delay_context = delay_updates_context_create(root_directory); + if (!cfg->delay_context) { + free(destination_path); + return FILE_SAVE_ERROR; + } + } + if (!delay_updates_prepare(cfg->delay_context)) { + free(destination_path); + return FILE_SAVE_ERROR; + } + char* staged_first = path_cat(cfg->delay_context->staging_root, file->hardlink_target); + char* staged_sibling = path_cat(cfg->delay_context->staging_root, file->path); + if (!staged_first || !staged_sibling) { + free(staged_first); + free(staged_sibling); + free(destination_path); + return FILE_SAVE_ERROR; + } + void* content = NULL; + unsigned long long content_size = 0; + bool source_absent = false; + if (!hardlink_read_source(staged_first, &content, &content_size, &source_absent)) { + FileSaveResult absent_result = + source_absent ? hardlink_sibling_absent_first(destination_path) : FILE_SAVE_ERROR; + free(staged_first); + free(staged_sibling); + free(destination_path); + return absent_result; + } + FileXattrList* sibling_xattrs = cfg->use_xattrs ? xattr_capture_path(staged_first) : NULL; + bool ok = file_to_disk_secure_link_attrs( + staged_sibling, staged_first, content, content_size, preallocate, file->metadata, + preserve_executability, use_fsync, sibling_xattrs, cfg ? cfg->fake_super : false, NULL); + xattr_list_free(sibling_xattrs); + free(content); + if (ok) + ok = delay_updates_record(cfg->delay_context, staged_sibling, destination_path, file->path); + if (!ok) + unlink(staged_sibling); + free(staged_first); + free(staged_sibling); + free(destination_path); + return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; + } + + char* first_disk = path_cat(root_directory, file->hardlink_target); + if (!first_disk) { + free(destination_path); + return FILE_SAVE_ERROR; + } + void* content = NULL; + unsigned long long content_size = 0; + bool source_absent = false; + if (!hardlink_read_source(first_disk, &content, &content_size, &source_absent)) { + FileSaveResult absent_result = + source_absent ? hardlink_sibling_absent_first(destination_path) : FILE_SAVE_ERROR; + free(first_disk); + free(destination_path); + return absent_result; + } + const char* temp_dir = (cfg && cfg->temp_dir) ? cfg->temp_dir : NULL; + FileXattrList* sibling_xattrs = cfg->use_xattrs ? xattr_capture_path(first_disk) : NULL; + bool ok = file_to_disk_secure_link_attrs( + destination_path, first_disk, content, content_size, preallocate, file->metadata, + preserve_executability, use_fsync, sibling_xattrs, cfg ? cfg->fake_super : false, temp_dir); + xattr_list_free(sibling_xattrs); + free(content); + free(first_disk); + free(destination_path); + return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; +} + +/* Validate a transmitted special rdev against the node kind implied by `mode`'s + * S_IFMT bits. Char/block devices require a legal major/minor pair (non-negative, + * range-checked); a non-device special (FIFO/socket) must carry an empty rdev. + * Used identically on the wire path and at the secure recreation site so a + * malicious/bogus rdev can never drive a dangerous node. */ +bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode) { + bool is_device = S_ISCHR(mode) || S_ISBLK(mode); + if (is_device) + return major >= 0 && minor >= 0 && major <= 0xffff && minor <= 0x00ffffff; + /* A non-device entry must actually be a special (FIFO/socket) and carry no + rdev; a regular/dir mode is never a valid special node. */ + return (S_ISFIFO(mode) || S_ISSOCK(mode)) && major == 0 && minor == 0; +} + +/* ---- Device/special node RECREATION (--devices/--specials), receiver side ---- + * + * Privilege gating: making a real device node requires CAP_MKNOD (root); making + * a FIFO works unprivileged (mkfifo). When the receiver lacks the capability, + * mknodat() fails with EPERM and the entry is SKIPPED with a warning -- the + * whole transfer must NOT abort just because the environment cannot make the + * node. CI runs non-root, so device creation is expected to skip there and + * only a FIFO is honestly assertable unprivileged. + * + * Confinement: the parent directory is opened fd-relative below the receive + * root (file_open_secure_parent: O_NOFOLLOW, no "..", root-checked) and the + * node is created with mknodat()/mkfifoat(), so it can never be placed outside + * the confined root and never follows a symlink. + * + * rdev validation: a malicious/bogus rdev (negative, out-of-range) is rejected + * here as well as on the wire (file_receive_special / chunk_deserialize), and a + * non-device entry must carry an empty rdev. + */ +static FileSaveResult file_save_special_to_disk(const char* root_directory, const File* file, + const Config* config) { + /* The empty-path and structural checks stay unconditional; the redundant + ".." list-path re-check is skipped under --trust-sender exactly like the + receive layer (confinement is deferred to the secure parent walk below, + which is never disabled). */ + if (!root_directory || !file || !file->path || file->path[0] == '\0' || + (!file_get_trust_sender() && has_path_traversal(file->path)) || !file->metadata) + return FILE_SAVE_ERROR; + + mode_t mode = file->metadata->mode; + bool is_char = S_ISCHR(mode); + bool is_blk = S_ISBLK(mode); + bool is_fifo = S_ISFIFO(mode); + bool is_sock = S_ISSOCK(mode); + if (!is_char && !is_blk && !is_fifo && !is_sock) { + log_message(LOG_LEVEL_ERROR, "Special node has no device/FIFO/socket mode"); + return FILE_SAVE_ERROR; + } + if (is_sock) { + /* No standard filesystem call recreates a socket; best-effort unsupported. */ + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "socket not recreated: %s (unsupported; skipped)", + escaped_path ? escaped_path : ""); + free(escaped_path); + return FILE_SAVE_SKIPPED; + } + if (is_char || is_blk) { + if (!config || !config->preserve_devices) + return FILE_SAVE_SKIPPED; + /* --super / --no-super (P7 Wave E): char/block device-node creation is a + super-user activity. --no-super forbids it even for a root receiver; + AUTO and --super attempt it (an unprivileged attempt is refused by the + kernel and skipped). The helper is evaluated against THIS config's mode + so the policy does not depend on a prior identity_set_active(). Pure + FIFO creation is unprivileged and deliberately NOT gated here. */ + if (!privilege_super_mode_permitted(config->super_mode)) { + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, + "skipping %s: super-user device-node creation is not permitted on this receiver", + escaped_path ? escaped_path : ""); + free(escaped_path); + return FILE_SAVE_SKIPPED; + } + } else if (is_fifo) { + if (!config || !config->preserve_specials) + return FILE_SAVE_SKIPPED; + } + /* Defense-in-depth rdev/type validation (also done on the wire path). */ + if (!file_special_rdev_valid(file->rdev_major, file->rdev_minor, mode)) { + log_message(LOG_LEVEL_ERROR, "Rejected out-of-range device rdev %d:%d", file->rdev_major, + file->rdev_minor); + return FILE_SAVE_ERROR; + } + + char* destination = path_cat(root_directory, file->path); + if (!destination) + return FILE_SAVE_ERROR; + char* leaf = NULL; + int parent_fd = file_open_secure_parent(destination, &leaf, true); + if (parent_fd < 0) { + free(destination); + return FILE_SAVE_ERROR; + } + + /* --existing / --ignore-existing / --update decide against the node that + would be replaced, mirroring the regular-file path. */ + if (config->existing && !file_path_exists_secure(destination)) { + close(parent_fd); + free(leaf); + free(destination); + return FILE_SAVE_SKIPPED; + } + if (config->ignore_existing && file_path_exists_secure(destination)) { + close(parent_fd); + free(leaf); + free(destination); + return FILE_SAVE_SKIPPED; + } + if (config->update && file_destination_is_newer_secure(destination, file->metadata)) { + close(parent_fd); + free(leaf); + free(destination); + return FILE_SAVE_SKIPPED; + } + + dev_t rdev = 0; + mode_t create_mode; + if (is_char) { + create_mode = S_IFCHR; + rdev = makedev((unsigned)file->rdev_major, (unsigned)file->rdev_minor); + } else if (is_blk) { + create_mode = S_IFBLK; + rdev = makedev((unsigned)file->rdev_major, (unsigned)file->rdev_minor); + } else { + create_mode = S_IFIFO; + } + mode_t perms = mode & 0777; + + int rc = is_fifo ? mkfifoat(parent_fd, leaf, perms) + : mknodat(parent_fd, leaf, create_mode | perms, rdev); + if (rc != 0) { + if (errno == EEXIST) { + /* An entry already exists: only skip when it already is a matching node; + never replace an existing directory or unrelated entry with the node. */ + struct stat st; + if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) == 0 && + ((is_char && S_ISCHR(st.st_mode)) || (is_blk && S_ISBLK(st.st_mode)) || + (is_fifo && S_ISFIFO(st.st_mode)))) { + close(parent_fd); + free(leaf); + free(destination); + return FILE_SAVE_SKIPPED; + } + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "refusing to replace existing entry with %s: %s (skipped)", + is_fifo ? "FIFO" : "device", escaped_path ? escaped_path : ""); + free(escaped_path); + } else if (errno == EPERM || errno == EACCES) { + /* Missing CAP_MKNOD / parent write permission: the environment cannot + create the node, so skip instead of failing the whole run. */ + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, + "skipping %s: cannot create %s node (%s)\n" + " --devices/--specials node creation needs privilege (CAP_MKNOD)", + escaped_path ? escaped_path : "", is_fifo ? "FIFO" : "device", + strerror(errno)); + free(escaped_path); + } else { + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "failed to create %s %s: %s (skipped)", + is_fifo ? "FIFO" : "device", escaped_path ? escaped_path : "", + strerror(errno)); + free(escaped_path); + } + close(parent_fd); + free(leaf); + free(destination); + return FILE_SAVE_SKIPPED; + } + + /* Apply mtime on the fresh node (utimensat, no-follow). */ + struct timespec times[2] = { + {.tv_sec = 0, .tv_nsec = UTIME_OMIT}, + {.tv_sec = file->metadata->mtime_sec, .tv_nsec = file->metadata->mtime_nsec}}; + utimensat(parent_fd, leaf, times, AT_SYMLINK_NOFOLLOW); + /* P7 Wave E: apply the negotiated ownership to the node ITSELF. A FIFO is + created unprivileged, but --copy-as and explicit identity policies own + every entry (a char/block node path is already privilege-gated above). The + no-follow helper changes the node's own ownership without dereferencing it; + it is a no-op unless an identity policy is active. */ + bool owner_ok = true; + if (identity_active_enabled()) + owner_ok = identity_apply_ownership_link(parent_fd, leaf, (int32_t)file->metadata->uid, + (int32_t)file->metadata->gid); + close(parent_fd); + free(leaf); + free(destination); + /* A failed required --copy-as ownership marks the node as failed; every other + * identity policy stays best-effort. */ + return owner_ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; +} + +/* --write-devices (receiver): write the received data directly into an EXISTING + * device node on the destination instead of creating a regular file. The node + * must already exist and be a char/block device (the device itself is opened and + * followed); it is confined to the receive root via file_open_secure_parent. + * Dangerous by nature, so deliberately restricted: a missing/non-device + * destination, or a write failure, is SKIPPED with a warning rather than + * allowed. On environments without device access the run still succeeds (the + * entry is skipped), never aborts. */ +static FileSaveResult file_save_write_device(const char* root_directory, const File* file) { + if (!root_directory || !file || !file->path || file->path[0] == '\0' || + (!file_get_trust_sender() && has_path_traversal(file->path))) + return FILE_SAVE_ERROR; + if (!file->data) + return FILE_SAVE_ERROR; + char* destination = path_cat(root_directory, file->path); + if (!destination) + return FILE_SAVE_ERROR; + char* leaf = NULL; + int parent_fd = file_open_secure_parent(destination, &leaf, false); + if (parent_fd < 0) { + free(destination); + return FILE_SAVE_SKIPPED; + } + /* O_NONBLOCK: a pre-existing FIFO at the target would otherwise block the + receive thread forever on open(2). With it the open only succeeds for a + readerless FIFO with O_RDWR (which the device fstat gate rejects anyway) + or fails with ENXIO/EAGAIN, both treated as a normal skip below. */ + int fd = openat(parent_fd, leaf, O_WRONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); + int saved_errno = errno; + free(leaf); + close(parent_fd); + if (fd < 0) { + free(destination); + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + const char* shown_path = escaped_path ? escaped_path : ""; + if (saved_errno == ENXIO || saved_errno == EAGAIN) { + /* A FIFO with no reader / an unreadable special: skip like every other + unusable write-devices target instead of blocking or failing. */ + log_message(LOG_LEVEL_WARNING, "write-devices: %s not writable (%s); skipped", shown_path, + strerror(saved_errno)); + } else { + log_message(LOG_LEVEL_WARNING, "write-devices: cannot open %s (%s); skipped", shown_path, + strerror(saved_errno)); + } + free(escaped_path); + return FILE_SAVE_SKIPPED; + } + struct stat st; + if (fstat(fd, &st) != 0 || !(S_ISCHR(st.st_mode) || S_ISBLK(st.st_mode))) { + close(fd); + free(destination); + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "write-devices: %s is not a device node; skipped", + escaped_path ? escaped_path : ""); + free(escaped_path); + return FILE_SAVE_SKIPPED; + } + bool ok = true; + if (file->data->size > 0) { + size_t total = (size_t)file->data->size; + size_t written = 0; + while (written < total) { + ssize_t n = write(fd, (char*)file->data->data + written, total - written); + if (n <= 0) { + ok = false; + break; + } + written += (size_t)n; + } + } + close(fd); + free(destination); + return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_SKIPPED; +} + +FileSaveResult file_save_to_disk_full(const char* root_directory, const File* file, + const Config* config) { + /* Backups are incompatible with ignore-existing: moving the entry first + would make a concurrent no-replace commit overwrite its old name. */ + bool backup_enabled = config && config->backup && !config->ignore_existing; + bool inplace = config && config->inplace; + bool sparse = config && config->preserve_sparse; + bool preserve_executability = config && config->use_executability; + const char* backup_suffix = (config && config->suffix) ? config->suffix : "~"; + const char* backup_dir = (config && config->backup_dir) ? config->backup_dir : NULL; + const char* partial_dir = (config && config->partial_dir) ? config->partial_dir : NULL; + const char* temp_dir = (config && config->temp_dir) ? config->temp_dir : NULL; + bool use_partial_root = partial_dir && config && config->partial; + char *confined_backup = NULL, *confined_partial = NULL, *disk_path = NULL; + char* destination_path = NULL; + char *backup_path = NULL, *parent_copy = NULL; + + if (!file || !file->path || !file->data || (file->data->size != 0 && !file->data->data) || + (!file_get_trust_sender() && has_path_traversal(file->path)) || + (backup_enabled && + (!backup_suffix || backup_suffix[0] == '\0' || strchr(backup_suffix, '/') != NULL || + strcmp(backup_suffix, ".") == 0 || strcmp(backup_suffix, "..") == 0))) { + log_message(LOG_LEVEL_ERROR, "Invalid file or path received"); + return FILE_SAVE_ERROR; + } + + /* P7 Wave D #1: a STATUS_DIR_TIMES entry is RECORD-ONLY. The scanner + captures every traversed directory -- including empty ones whose parents + were never created by a child write and directories pruned by + -m/--prune-empty-dirs. Creating them here would resurrect empty + directories (an -a behavior change) and could abort the whole transfer on a + pre-existing regular file/symlink at the mirror path. Short-circuit before + any device/write-devices/directory branch and report it as skipped so the + sink still accumulates its metadata for the deferred DirTimeList + application, but create nothing. */ + if (file->dir_time_only) + return FILE_SAVE_SKIPPED; + + /* Device/special node (--devices/--specials): recreate the node instead of + writing content (privilege-gated, confined, rdev-validated). */ + if (file->is_special) + return file_save_special_to_disk(root_directory, file, config); + /* --write-devices: write straight into an existing device node. Writing + into a device is a super-user activity, so --no-super must suppress it just + like device-node creation; the default AUTO/--super attempt it (the wide + open below keeps its own confinement and best-effort skip semantics). */ + if (config && config->write_devices) { + if (!privilege_super_mode_permitted(config->super_mode)) { + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, + "write-devices: %s skipped: super-user activities are not permitted on this " + "receiver", + escaped_path ? escaped_path : "(null)"); + free(escaped_path); + return FILE_SAVE_SKIPPED; + } + return file_save_write_device(root_directory, file); + } + + /* Explicit directory entries (--dirs) carry an empty payload; the entry is + created as a directory under the receive root, applying the same secure + mkdir-parent semantics as regular writes. Directories are created + immediately (they are never staged by --delay-updates, matching rsync, + where directory creation is not delayed). */ + if (file->is_dir) { + if (file->path[0] == '\0' || (!file_get_trust_sender() && has_path_traversal(file->path))) { + log_message(LOG_LEVEL_ERROR, "Invalid directory path received"); + return FILE_SAVE_ERROR; + } + char* dir_path = path_cat(root_directory, file->path); + if (!dir_path) + return FILE_SAVE_ERROR; + bool ok = file_ensure_directory_secure(dir_path); + /* P7 Wave E: apply the negotiated ownership to the directory ITSELF (not + just the files inside it). --copy-as and every explicit identity policy + own every entry, so a directory must not keep the receiver's owner while + its children get the policy owner. Applied no-follow on the confined + parent fd after the mkdir; identity_apply_ownership_link() is itself a + no-op unless an identity policy is active. */ + if (ok && file->metadata && identity_active_enabled()) { + char* leaf = NULL; + int parent_fd = file_open_secure_parent(dir_path, &leaf, false); + if (parent_fd >= 0) { + if (!identity_apply_ownership_link(parent_fd, leaf, (int32_t)file->metadata->uid, + (int32_t)file->metadata->gid)) + ok = false; + close(parent_fd); + } else if (identity_copy_as_active()) { + /* The directory exists (ok) but its required --copy-as ownership could + not be applied because the confined parent could not be opened. */ + ok = false; + } + free(leaf); + } + free(dir_path); + return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; + } + + /* Symlink entry. (The process-wide --keep-dirlinks policy is set once by the + connection handler from the negotiated config, before any receiver/writer + threads start, so it is stable throughout this walk.) */ + + if (file->is_symlink) { + if (!file->symlink_target || file->path[0] == '\0' || + (!file_get_trust_sender() && has_path_traversal(file->path))) { + log_message(LOG_LEVEL_ERROR, "Invalid symlink entry received"); + return FILE_SAVE_ERROR; + } + char* link_path = path_cat(root_directory, file->path); + if (!link_path) + return FILE_SAVE_ERROR; + /* Restore the real target by stripping the sender's --munge-links marker. + Only unmunge when the policy was negotiated: a plain -l run must preserve + a source symlink whose target genuinely begins with the marker verbatim. */ + char* target = str_dup(file->symlink_target); + bool ok = target != NULL; + if (ok && config && config->munge_links) + file_symlink_unmunge(target); + /* Receiver-side trust boundary (independent of the sender): a target that + could escape the receive root (absolute, or relative-with-"..") is never + materialized. It is contained (the entry is skipped) rather than failing + the whole transfer, so a hostile sender can inject a broken symlink but + can never redirect it outside the root. --trust-sender deliberately + relaxes this receiver-side re-validation: a trusted sender's escaping + symlink target is copied verbatim (rsync -l parity). The low-level + leaf/destination confinement in file_symlink_at_secure still ensures the + link itself is placed inside the authorized root. */ + if (ok && !file_get_trust_sender() && !file_symlink_target_contained(target)) + ok = false; + if (!ok) { + /* Skip the escaping/empty target (contained) rather than abort. */ + free(target); + free(link_path); + return FILE_SAVE_SKIPPED; + } + char* parent = str_dup(link_path); + if (parent) { + /* Propagate a failed --copy-as ownership of the parent directory this + creates; every other failure mode stays best-effort as before. */ + ok = file_ensure_directory_secure(dirname(parent)); + free(parent); + } + if (ok) + ok = file_symlink_at_secure(link_path, target); + free(target); + /* P7 Wave D: apply the symlink's own metadata with no-follow primitives + (utimensat/lchown/fchmodat AT_SYMLINK_NOFOLLOW). -J/--omit-link-times + suppresses the timestamps; ownership stays gated by the identity policy. + A symlink has no children, so this can be applied immediately. */ + if (ok && config && config->use_metadata) + ok = file_restore_symlink_metadata(link_path, file->metadata, config->omit_link_times); + free(link_path); + return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; + } + + /* --hard-links/-H sibling: a later member of a link group arrives with no + payload and is installed as a hard link to (or, on link() failure, a + byte-identical copy of) the group's first member. Handled entirely here, + before the normal data-write paths (which would create an empty file). */ + if (file->link_group != 0 && !file->link_first && file->hardlink_target != NULL) { + return file_save_hardlink_sibling(root_directory, file, config); + } + + /* These options arrive from the client. They are names below the server + root, never independent filesystem roots. --temp-dir is confined exactly + like --backup-dir/--partial-dir: an absolute or `..`-escaping scratch + directory is rejected outright so nothing is ever created outside the + authorized destination root. */ + if ((backup_dir && (backup_dir[0] == '/' || has_path_traversal(backup_dir))) || + (partial_dir && (partial_dir[0] == '/' || has_path_traversal(partial_dir))) || + (temp_dir && (temp_dir[0] == '/' || has_path_traversal(temp_dir)))) + return FILE_SAVE_ERROR; + if (backup_dir && !(confined_backup = path_cat(root_directory, backup_dir))) + return FILE_SAVE_ERROR; + if (partial_dir && !(confined_partial = path_cat(root_directory, partial_dir))) { + free(confined_backup); + return FILE_SAVE_ERROR; + } + + const char* actual_root = use_partial_root ? confined_partial : root_directory; + destination_path = path_cat(root_directory, file->path); + disk_path = path_cat(actual_root, file->path); + if (destination_path == NULL || disk_path == NULL) { + free(confined_backup); + free(confined_partial); + free(destination_path); + free(disk_path); + return FILE_SAVE_ERROR; + } + + /* --delay-updates diverts the whole write into the staging tree; the rest of + this function is the immediate-install path. */ + if (config && config->delay_updates) { + FileSaveResult result = + file_stage_delayed_update(root_directory, destination_path, file, (Config*)config); + free(confined_backup); + free(confined_partial); + free(destination_path); + free(disk_path); + return result; + } + + /* --existing checks the final destination, not a temporary partial path. */ + if (config && config->existing && !file_path_exists_secure(destination_path)) { + free(confined_backup); + free(confined_partial); + free(destination_path); + free(disk_path); + return FILE_SAVE_SKIPPED; + } + + /* --ignore-existing checks the final destination before partial files or + overwrite policies can modify it. */ + if (config && config->ignore_existing) { + bool exists = file_path_exists_secure(destination_path); + if (exists) { + free(confined_backup); + free(confined_partial); + free(destination_path); + free(disk_path); + return FILE_SAVE_SKIPPED; + } + } + + /* --update is receiver-side policy: never replace a newer destination. + In partial-dir mode the entry that would be replaced is the real + destination, not the temporary partial file. The secure stat does not + require read permission on the destination. */ + const char* update_target = use_partial_root ? destination_path : disk_path; + if (config && config->update && file_destination_is_newer_secure(update_target, file->metadata)) { + free(confined_backup); + free(confined_partial); + free(destination_path); + free(disk_path); + return FILE_SAVE_SKIPPED; + } + + /* --force (rsync semantics): an incoming regular file may replace a + destination DIRECTORY by removing that (possibly non-empty, symlink-safe) + tree first, so the atomic temp+rename below can install the file. Only the + immediate-install path does this: a --delay-updates run stages into its own + tree and is unaffected here (its publication renames over regular files + only). The blocking directory is removed only after the --update / + --existing / --ignore-existing decisions above, which see it as an existing + destination entry. */ + if (config && config->force_delete && !file->is_dir && + file_directory_exists_secure(destination_path)) { + if (!file_remove_tree_secure(destination_path)) + goto fail; + } + + if (backup_enabled) { + /* Back up the entry that the incoming write will replace. When writing + through a partial dir the pre-existing destination file is the one to + preserve; any stale partial file is overwritten without a backup. */ + const char* replace_target = use_partial_root ? destination_path : disk_path; + struct stat backup_stat; + if (file_stat_secure(replace_target, &backup_stat)) { + if (backup_dir) { + backup_path = path_cat(confined_backup, file->path); + } else { + size_t path_len = strlen(replace_target); + size_t suffix_len = strlen(backup_suffix); + if (path_len > SIZE_MAX - suffix_len - 1) + goto fail; + backup_path = malloc(path_len + suffix_len + 1); + if (backup_path) { + memcpy(backup_path, replace_target, path_len); + memcpy(backup_path + path_len, backup_suffix, suffix_len + 1); + } + } + if (!backup_path) + goto fail; + parent_copy = str_dup(backup_path); + if (!parent_copy || !file_ensure_directory_secure(dirname(parent_copy))) + goto fail; + free(parent_copy); + parent_copy = NULL; + if (!file_rename_secure(replace_target, backup_path)) + goto fail; + free(backup_path); + backup_path = NULL; + } + } + + FileMetadata adjusted_metadata; + const FileMetadata* metadata = file->metadata; + if (metadata && config && config->chmod_spec && *config->chmod_spec) { + adjusted_metadata = *metadata; + if (!chmod_apply(adjusted_metadata.mode, config->chmod_spec, &adjusted_metadata.mode)) + goto fail; + metadata = &adjusted_metadata; + } + + /* A configured --temp-dir sends the temporary working copy to a scratch + directory resolved below the receive root; the engine then atomically + renames the completed file into the final destination directory. The + partial-dir flow already keeps its working copy in a separate directory + and --inplace writes directly, so neither diverts through the scratch + dir (matching rsync, where --inplace/--partial-dir supersede --temp-dir). */ + char* confined_temp = NULL; + bool use_temp_dir = temp_dir != NULL && !inplace && !use_partial_root; + if (use_temp_dir) { + confined_temp = path_cat(root_directory, temp_dir); + if (!confined_temp) + goto fail; + /* A user-supplied trailing slash would leave the scratch path ending in + "/", which has no final component to create/open. Normalize it away. */ + size_t temp_len = strlen(confined_temp); + while (temp_len > 1 && confined_temp[temp_len - 1] == '/') + confined_temp[--temp_len] = '\0'; + } + /* A --link-dest basis hit installs an atomic hard link (with a byte-copy + fallback); --inplace and the update/no-replace write variants do not + apply to a fresh hard link, whose inode attributes already match. The + existing/ignore-existing/update/backup preamble above has already made the + policy decision. */ + bool ok; + if (config && file->basis_link) { + ok = file_to_disk_secure_link_attrs(disk_path, file->basis_link, file->data->data, + file->data->size, config->preallocate, metadata, + preserve_executability, config->use_fsync, file->xattrs, + config->fake_super, confined_temp); + } else { + /* The plain no-replace / update / with-fsync engines, plus per-file xattr + (-X/-A) and --fake-super application on the written fd. */ + ok = file_to_disk_secure_attrs( + disk_path, file->data->data, file->data->size, inplace, sparse, + config && config->preallocate, metadata, preserve_executability, config && config->update, + config && config->ignore_existing, config && config->use_fsync, file->xattrs, + config ? config->fake_super : false, config ? config->partial : false, confined_temp); + } + free(confined_temp); + confined_temp = NULL; + if (!ok) + goto fail; + + /* --partial --partial-dir writes the complete file under the partial dir so + interrupted transfers leave a resumable copy there. Once the file is + fully written it must be atomically installed at the real destination; + otherwise completed transfers would linger under the partial dir. */ + if (use_partial_root) { + if (!file_rename_secure(disk_path, destination_path)) + goto fail; + } + + free(parent_copy); + free(backup_path); + free(confined_backup); + free(confined_partial); + free(destination_path); + free(disk_path); + return FILE_SAVE_WRITTEN; + +fail: + free(parent_copy); + free(backup_path); + free(confined_backup); + free(confined_partial); + free(destination_path); + free(disk_path); + return FILE_SAVE_ERROR; +} + +/* Receive a file's xattr block (when the config enables xattr transport) and + * attach it to `file`. Returns false on a malformed/oversized frame. */ +static bool receive_file_xattrs(File* file, int fd, const Config* config) { + if (!config->use_xattrs) + return true; + int xok = 0; + FileXattrList* list = xattr_receive(fd, &xok); + if (!xok) { + xattr_list_free(list); + return false; + } + file->xattrs = list; + return true; +} + +static File* receive_delta_file(int fd, const Config* config, const char* check_path, + void* old_data, unsigned long long old_size, bool* failed) { + if (!old_data) { + free(old_data); /* defensive: old_data is always non-NULL today */ + *failed = true; + return NULL; + } + + DeltaSignature* sig = delta_signature_create_seeded(old_data, old_size, config->delta_block_size, + (uint32_t)config->checksum_seed); + if (!sig) { + free(old_data); + *failed = true; + return NULL; + } + + Data* sig_data = delta_signature_serialize(sig); + if (!sig_data) { + delta_signature_destroy(sig); + free(old_data); + *failed = true; + return NULL; + } + + bool sig_sent = send_status(fd, STATUS_DELTA_SIGNATURE) && send_data(fd, sig_data); + data_destroy(sig_data); + + if (!sig_sent) { + delta_signature_destroy(sig); + free(old_data); + *failed = true; + return NULL; + } + + Status resp; + if (!receive_status(fd, &resp)) { + delta_signature_destroy(sig); + free(old_data); + *failed = true; + return NULL; + } + + if (resp == STATUS_DELTA_DATA) { + Data* delta_data = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); + if (!delta_data) { + delta_signature_destroy(sig); + free(old_data); + *failed = true; + return NULL; + } + + Data* raw_delta = delta_data; + if (config->use_compression && + !compression_should_skip_with_suffixes( + check_path, config->skip_compress_suffixes, + config->skip_compress_set ? config->skip_compress_count : -1)) { + raw_delta = data_decompress_limited(delta_data, MAX_RECEIVE_WHOLE_FILE_SIZE); + data_destroy(delta_data); + if (!raw_delta) { + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + } + + Delta* delta = delta_deserialize(raw_delta); + data_destroy(raw_delta); + if (!delta) { + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + + uint64_t new_size = delta->new_file_size; + if (new_size > MAX_RECEIVE_WHOLE_FILE_SIZE || new_size > SIZE_MAX) { + delta_destroy(delta); + free(old_data); + delta_signature_destroy(sig); + send_status(fd, STATUS_ERROR); + *failed = true; + return NULL; + } + void* new_data = delta_apply(old_data, old_size, delta, config->delta_block_size); + delta_destroy(delta); + + if (!new_data) { + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + + File* file = file_create(check_path); + if (!file) { + free(new_data); + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + + if (config->use_metadata) { + int meta_ok = 1; + file->metadata = metadata_receive(fd, &meta_ok); + if (!meta_ok) { + file_destroy(file); + free(new_data); + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + } + if (!receive_file_xattrs(file, fd, config)) { + file_destroy(file); + free(new_data); + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + + Data* replacement = data_create(new_data, (size_t)new_size); + if (replacement == NULL) { + file_destroy(file); + free(old_data); + delta_signature_destroy(sig); + send_status(fd, STATUS_ERROR); + *failed = true; + return NULL; + } + data_destroy(file->data); + file->data = replacement; + + free(old_data); + delta_signature_destroy(sig); + return file; + } + + if (resp == STATUS_NEXT) { + delta_signature_destroy(sig); + free(old_data); + + File* file = file_create(check_path); + if (!file) { + *failed = true; + return NULL; + } + + if (config->use_metadata) { + int meta_ok = 1; + file->metadata = metadata_receive(fd, &meta_ok); + if (!meta_ok) { + file_destroy(file); + *failed = true; + return NULL; + } + } + if (!receive_file_xattrs(file, fd, config)) { + file_destroy(file); + *failed = true; + return NULL; + } + + Data* file_data = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); + if (file_data == NULL) { + file_destroy(file); + *failed = true; + return NULL; + } + + if (config->use_compression && + !compression_should_skip_with_suffixes( + file->path, config->skip_compress_suffixes, + config->skip_compress_set ? config->skip_compress_count : -1)) { + Data* uncompressed = data_decompress_limited(file_data, MAX_RECEIVE_WHOLE_FILE_SIZE); + data_destroy(file_data); + if (uncompressed == NULL) { + file_destroy(file); + *failed = true; + return NULL; + } + if (uncompressed->size > MAX_FILE_DATA_SIZE) { + data_destroy(uncompressed); + file_destroy(file); + send_status(fd, STATUS_ERROR); + *failed = true; + return NULL; + } + file_data = uncompressed; + } + + data_destroy(file->data); + file->data = file_data; + return file; + } + + delta_signature_destroy(sig); + free(old_data); + send_status(fd, STATUS_ERROR); + *failed = true; + return NULL; +} + +/* ---- Alternate basis directories (--compare-dest / --copy-dest / --link-dest) ---- + * The receiver consults the ordered basis-dir list only when the destination + * entry is NOT already up to date. An "exact match" requires an equal size, + * an equal mtime (unless --size-only), and an equal content xxHash64, so a + * hard link / local copy is only ever made from byte-identical content. */ + +typedef struct BasisMatch { + bool hit; + BasisDestType type; + char* basis_path; /* owned absolute path of the matched basis file */ + struct stat st; /* fstat() of the matched basis file */ + Data* content; /* owned basis bytes (or empty Data), NULL when not loaded */ +} BasisMatch; + +static void basis_match_free(BasisMatch* match) { + if (!match) + return; + free(match->basis_path); + match->basis_path = NULL; + data_destroy(match->content); + match->content = NULL; + match->hit = false; + match->type = BASIS_DEST_NONE; +} + +/* Open `path` (via the secure, root-confined primitives) and require it to be + a regular file of exactly `expected_size` bytes. Returns an open read-only + descriptor and its fstat on success. */ +static bool basis_open_regular(const char* path, unsigned long long expected_size, int* out_fd, + struct stat* out_st) { + char* leaf = NULL; + int parent_fd = file_open_secure_parent(path, &leaf, false); + if (parent_fd < 0) + return false; + int fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW); + free(leaf); + close(parent_fd); + if (fd < 0) + return false; + struct stat st; + if (fstat(fd, &st) != 0 || !S_ISREG(st.st_mode) || + (unsigned long long)st.st_size != expected_size) { + close(fd); + return false; + } + *out_fd = fd; + *out_st = st; + return true; +} + +/* Read the whole remaining content of an open descriptor. A zero-length file + yields an empty Data (data pointer NULL). */ +static Data* basis_read_content(int fd, unsigned long long size) { + if (size == 0) + return data_create_reserve(0); + if (size > MAX_RECEIVE_WHOLE_FILE_SIZE || size > SIZE_MAX) + return NULL; + void* buf = protocol_alloc((size_t)size); + if (!buf) + return NULL; + size_t got = 0; + while (got < (size_t)size) { + ssize_t n = read(fd, (char*)buf + got, (size_t)size - got); + if (n <= 0) { + free(buf); + return NULL; + } + got += (size_t)n; + } + return data_create(buf, (size_t)size); +} + +/* --ignore-times forces every file to be updated, so no basis hit is ever + declared (matching rsync, where -I prevents link-dest from linking). */ +static bool basis_quick_matches(const Config* config, const struct stat* st, time_t check_mtime, + long check_mtime_nsec) { + if (config->size_only) + return true; + long mtime_nsec = 0; +#ifdef __linux__ + mtime_nsec = st->st_mtim.tv_nsec; +#endif + return metadata_mtime_matches(st->st_mtime, mtime_nsec, check_mtime, check_mtime_nsec, + config->modify_window); +} + +/* Search the basis-dir list in command-line order and return the first exact + match. When load_content is true the matched bytes are kept in out->content + so the caller can materialize the file without re-reading it. */ +static bool basis_match_find(const Config* config, const char* check_path, + unsigned long long check_size, time_t check_mtime, + long check_mtime_nsec, const uint8_t* check_digest, + size_t check_digest_len, bool load_content, BasisMatch* out) { + memset(out, 0, sizeof(*out)); + if (!config || !config_has_basis(config) || config->ignore_times) + return false; + for (int i = 0; i < config->basis_count; i++) { + const BasisDest* entry = &config->basis_dirs[i]; + char* basis_dir = path_cat(config->receive_root_directory, entry->path); + if (!basis_dir) + continue; + char* candidate = path_cat(basis_dir, check_path); + free(basis_dir); + if (!candidate) + continue; + + int fd; + struct stat st; + if (basis_open_regular(candidate, check_size, &fd, &st)) { + if (basis_quick_matches(config, &st, check_mtime, check_mtime_nsec)) { + Data* content = basis_read_content(fd, check_size); + if (content) { + uint8_t basis_digest[CHECKSUM_MAX_DIGEST_LEN]; + size_t basis_len = 0; + bool hashed = checksum_digest((ChecksumAlgo)config->checksum_algo, config->checksum_seed, + content->data, content->size, basis_digest, + sizeof(basis_digest), &basis_len); + if (hashed && basis_len == check_digest_len && check_digest_len > 0 && + memcmp(basis_digest, check_digest, check_digest_len) == 0) { + out->hit = true; + out->type = entry->type; + out->basis_path = candidate; + candidate = NULL; /* ownership transferred to out */ + out->st = st; + out->content = load_content ? content : NULL; + if (!load_content) + data_destroy(content); + close(fd); + return true; + } + } + data_destroy(content); + } + close(fd); + } + free(candidate); + } + return false; +} + +/* --------------------------------------------------------------------------- + * -y/--fuzzy similar-file delta basis. + * + * When a file must be transferred and the destination holds no usable content + * at the exact path (the destination file is absent, or is outside the delta + * engine's size bounds), --fuzzy lets the receiver reuse an EXISTING regular + * file in the SAME destination directory as the delta basis, so the sender + * transmits only the differences instead of the whole file. This is the + * rsync "find a similar file to use as a basis for a transfer" case (e.g. a + * file recreated under a new name whose old-named sibling is still present). + * + * The delta handshake is unchanged and receiver-driven, so the sender never + * learns the basis was a different file and needs no new protocol. Byte + * exactness never depends on which bytes the basis holds: the delta protocol + * only references basis blocks whose Adler-32 + xxHash32 checksums match the + * source, delta_apply validates every reference against the basis size, and a + * basis that shares nothing simply makes the sender reply STATUS_NEXT (full + * transfer). A fuzzy basis can therefore waste bandwidth but never corrupt a + * file. + * + * Similarity heuristic (deterministic, deliberately simpler than rsync's): + * * candidates are the target's sibling entries in its destination + * directory, opened through the confined root (file_open_secure_parent + + * openat O_NOFOLLOW, fstatat AT_SYMLINK_NOFOLLOW) -- symlinks are never + * followed and nothing outside the destination root is ever read; + * * dotfiles, directories, the target's own name, and the .fastsync-stage / + * temp scratch names are never candidates; + * * size gate = the delta engine's own bounds (delta_should_attempt: both + * files >= DELTA_MIN_FILE_SIZE, <= delta_max_file_size, ratio <= 10x), + * NOT rsync's ~1.5x size window; + * * name gate = Levenshtein edit distance between the basenames, accepted + * only when distance <= half the length of the longer basename; + * * the single best candidate (smallest distance; tie-break: size closest + * to the incoming file, then lexicographically smaller basename) is read + * and returned as the basis. + * ------------------------------------------------------------------------- */ + +/* A directory scan is linear in the number of entries; the fuzzy search stops + * after this many so a pathological huge directory cannot stall a transfer. + * The cap bounds the readdir() ITERATIONS, not the per-entry work: every + * entry that survives the (cheap) size and pre-name gates still runs an + * edit-distance DP, so the per-entry DP cost is separately bounded below by + * pre-pruning on the name length gap and the absent-character bound, and by + * trimming the common prefix/suffix before the DP runs on the middles only. */ +#define FUZZY_MAX_DIRECTORY_SCAN 4096 +/* Names longer than this never take part in fuzzy matching: the edit-distance + * DP below is O(len^2), so over-long names are bounded out of the search. */ +#define FUZZY_NAME_LIMIT 192 + +typedef struct { + char name[FUZZY_NAME_LIMIT + 1]; + unsigned long long size; + size_t distance; + unsigned long long size_gap; +} FuzzyCandidate; + +/* Two-row DP scratch, allocated once per directory scan (not per candidate) so + * a 4096-entry directory never performs 4096 malloc/free pairs. */ +typedef struct { + size_t* prev; + size_t* cur; +} FuzzyEditBuffer; + +static bool fuzzy_edit_buffer_init(FuzzyEditBuffer* buf) { + buf->prev = malloc((FUZZY_NAME_LIMIT + 1) * sizeof(size_t)); + buf->cur = malloc((FUZZY_NAME_LIMIT + 1) * sizeof(size_t)); + if (!buf->prev || !buf->cur) { + free(buf->prev); + free(buf->cur); + buf->prev = NULL; + buf->cur = NULL; + return false; + } + return true; +} + +static void fuzzy_edit_buffer_destroy(FuzzyEditBuffer* buf) { + free(buf->prev); + free(buf->cur); + buf->prev = NULL; + buf->cur = NULL; +} + +/* Cheap lower bounds used to reject a candidate BEFORE the DP: + * - any edit script must at least absorb the length gap: d >= |la - lb|; + * - any character of `a` that does not occur in `b` at all must be deleted or + * substituted at its own position: d >= (count of such characters). + * The acceptance gate is d*2 <= longer, so a candidate whose max of these two + * bounds already violates it can be skipped without computing the distance. */ +static size_t fuzzy_absent_char_bound(const char* a, size_t la, const char* b, size_t lb) { + if (lb == 0) + return la; + bool present[256] = {false}; + for (size_t i = 0; i < lb; i++) + present[(uint8_t)b[i]] = true; + size_t absent = 0; + for (size_t i = 0; i < la; i++) + if (!present[(uint8_t)a[i]]) + absent++; + return absent; +} + +/* Levenshtein edit distance between the two basenames. A shared prefix and a + * (non-overlapping) shared suffix can always be aligned at no cost, so the DP + * only runs over the differing middles; its two rows come from `buf` (allocated + * once by the caller). Callers enforce la, lb <= FUZZY_NAME_LIMIT. */ +static size_t fuzzy_edit_distance(FuzzyEditBuffer* buf, const char* a, size_t la, const char* b, + size_t lb) { + size_t p = 0; + while (p < la && p < lb && a[p] == b[p]) + p++; + /* Trim the common suffix (never overlapping the prefix). Working with two + moving end indices keeps the region arithmetic explicit and safe. */ + size_t ae = la; + size_t be = lb; + while (ae > p && be > p && a[ae - 1] == b[be - 1]) { + ae--; + be--; + } + size_t ma = ae - p; + size_t mb = be - p; + /* cppcheck-suppress knownConditionTrueFalse -- the prefix/suffix trims above + only run while the corresponding ends match, so a middle can remain; the + analysis unsoundly concludes the trims always consume everything. */ + if (ma == 0) + return mb; + if (mb == 0) + return ma; + const char* A = a + p; + const char* B = b + p; + size_t* prev = buf->prev; + size_t* cur = buf->cur; + for (size_t j = 0; j <= mb; j++) + prev[j] = j; + for (size_t i = 1; i <= ma; i++) { + cur[0] = i; + for (size_t j = 1; j <= mb; j++) { + size_t cost = A[i - 1] == B[j - 1] ? 0 : 1; + size_t del = prev[j] + 1; + size_t ins = cur[j - 1] + 1; + size_t sub = prev[j - 1] + cost; + size_t m = del < ins ? del : ins; + cur[j] = m < sub ? m : sub; + } + size_t* tmp = prev; + prev = cur; + cur = tmp; + } + return prev[mb]; +} + +/* Deterministic ordering of two fuzzy candidates: smallest edit distance, + * then the size closest to the incoming file, then the lexical basename. */ +static bool fuzzy_candidate_better(const FuzzyCandidate* cand, const FuzzyCandidate* best) { + if (!best->name[0]) + return true; + if (cand->distance != best->distance) + return cand->distance < best->distance; + if (cand->size_gap != best->size_gap) + return cand->size_gap < best->size_gap; + return strcmp(cand->name, best->name) < 0; +} + +/* Search the destination directory that will contain `check_path` for a + * similar regular file usable as a --fuzzy delta basis and return its full + * content in a malloc'd (protocol_alloc) buffer. Returns NULL (with *out_size + * = 0) when no candidate qualifies, which means the caller performs the normal + * whole-file transfer. */ +static void* fuzzy_basis_find_and_load(const Config* config, const char* check_path, + unsigned long long check_size, + unsigned long long* out_size) { + *out_size = 0; + if (!config || !config->receive_root_directory || !config->fuzzy || !config->use_delta || + !check_path || check_size < DELTA_MIN_FILE_SIZE || check_size > config->delta_max_file_size || + check_size > MAX_RECEIVE_WHOLE_FILE_SIZE) + return NULL; + + char* full_path = path_cat(config->receive_root_directory, check_path); + if (!full_path) + return NULL; + char* leaf = NULL; + int dir_fd = file_open_secure_parent(full_path, &leaf, false); + if (dir_fd < 0 || !leaf) { + free(leaf); + free(full_path); + return NULL; + } + size_t target_len = strlen(leaf); + /* A target basename longer than FUZZY_NAME_LIMIT can never pass the name gate + (every candidate name is bounded by the same limit), so skip the scan. */ + if (target_len > FUZZY_NAME_LIMIT) { + close(dir_fd); + free(leaf); + free(full_path); + return NULL; + } + + int scanfd = dup(dir_fd); + if (scanfd < 0) { + close(dir_fd); + free(leaf); + free(full_path); + return NULL; + } + DIR* dir = fdopendir(scanfd); + if (!dir) { + close(scanfd); + close(dir_fd); + free(leaf); + free(full_path); + return NULL; + } + + /* The DP scratch rows are allocated once per scan (not once per candidate). */ + FuzzyEditBuffer ebuf; + if (!fuzzy_edit_buffer_init(&ebuf)) { + closedir(dir); + close(dir_fd); + free(leaf); + free(full_path); + return NULL; + } + + FuzzyCandidate best; + memset(&best, 0, sizeof(best)); + const struct dirent* entry; + size_t scanned = 0; + /* readdir() yields entries in filesystem-dependent order, so the SET of + candidates seen is order-dependent; the winner is still deterministic + because every candidate is compared with the total ordering in + fuzzy_candidate_better (acceptable for a heuristic). */ + while (scanned < FUZZY_MAX_DIRECTORY_SCAN && (entry = readdir(dir)) != NULL) { + scanned++; + const char* name = entry->d_name; + size_t name_len = strlen(name); + if (name[0] == '.' || name_len == 0 || name_len > FUZZY_NAME_LIMIT || strcmp(name, leaf) == 0) + continue; + struct stat st; + if (fstatat(dir_fd, name, &st, AT_SYMLINK_NOFOLLOW) != 0 || !S_ISREG(st.st_mode)) + continue; + unsigned long long cand_size = (unsigned long long)st.st_size; + if (cand_size == 0 || cand_size > MAX_RECEIVE_WHOLE_FILE_SIZE || + !delta_should_attempt(cand_size, check_size, config->delta_max_file_size)) + continue; + /* Cheap pre-name gates run BEFORE the edit-distance DP. The edit distance + is bounded below by the length gap |la-lb| and by the number of + characters of one basename that are absent from the other (each such + position costs at least one op), so a candidate whose acceptance gate + (distance*2 <= longer) already fails on the max of those bounds is + skipped without running the DP. */ + size_t longer = target_len > name_len ? target_len : name_len; + size_t bound = longer - (target_len < name_len ? target_len : name_len); + size_t absent = fuzzy_absent_char_bound(leaf, target_len, name, name_len); + if (absent > bound) + bound = absent; + if (bound * 2 > longer) + continue; + size_t distance = fuzzy_edit_distance(&ebuf, leaf, target_len, name, name_len); + if (distance * 2 > longer) + continue; + FuzzyCandidate cand; + memcpy(cand.name, name, name_len + 1); + cand.size = cand_size; + cand.distance = distance; + cand.size_gap = cand_size > check_size ? cand_size - check_size : check_size - cand_size; + if (fuzzy_candidate_better(&cand, &best)) + best = cand; + } + closedir(dir); + free(leaf); + fuzzy_edit_buffer_destroy(&ebuf); + + void* basis = NULL; + if (best.name[0]) { + /* O_NONBLOCK: a name raced to a FIFO between the fstatat gate and this open + would otherwise block the receive thread forever on open(2); with it the + open fails (ENXIO) and the fstat/S_ISREG gate below would reject it too. + A regular file opened with O_NONBLOCK is unaffected. */ + int fd = openat(dir_fd, best.name, O_RDONLY | O_NOFOLLOW | O_NONBLOCK | O_CLOEXEC); + if (fd >= 0) { + struct stat st; + if (fstat(fd, &st) == 0 && S_ISREG(st.st_mode) && + (unsigned long long)st.st_size == best.size && best.size <= SIZE_MAX) { + basis = protocol_alloc((size_t)best.size); + if (basis) { + size_t got = 0; + while (got < (size_t)best.size) { + ssize_t n = read(fd, (char*)basis + got, (size_t)best.size - got); + if (n <= 0) { + free(basis); + basis = NULL; + break; + } + got += (size_t)n; + } + } + } + close(fd); + } + } + close(dir_fd); + free(full_path); + if (basis) + *out_size = best.size; + return basis; +} + +/* Read the remainder of a full-file transfer after the receiver has already + * sent STATUS_NEXT: receive the metadata frame (when enabled) followed by the + * data frame, and return an owned File. Shared by the plain full-transfer path + * and the --append-verify prefix-mismatch fallback (a clean full transfer + * instead of a corrupt prefix+tail blend). */ +static File* receive_full_file(int fd, const Config* config, const char* path) { + File* file = file_create(path); + if (!file) + return NULL; + if (config->use_metadata) { + int meta_ok = 1; + file->metadata = metadata_receive(fd, &meta_ok); + if (!meta_ok) { + file_destroy(file); + return NULL; + } + } + if (!receive_file_xattrs(file, fd, config)) { + file_destroy(file); + return NULL; + } + Data* file_data = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); + if (file_data == NULL) { + file_destroy(file); + return NULL; + } + if (config->use_compression && + !compression_should_skip_with_suffixes(file->path, config->skip_compress_suffixes, + config->skip_compress_set ? config->skip_compress_count + : -1)) { + Data* uncompressed = data_decompress_limited(file_data, MAX_RECEIVE_WHOLE_FILE_SIZE); + data_destroy(file_data); + if (uncompressed == NULL) { + file_destroy(file); + return NULL; + } + if (uncompressed->size > MAX_FILE_DATA_SIZE) { + data_destroy(uncompressed); + file_destroy(file); + return NULL; + } + file_data = uncompressed; + } + data_destroy(file->data); + file->data = file_data; + return file; +} + +File* receive_incremental_check(int fd, const Config* config, bool* skipped) { + if (!config || !skipped) { + send_status(fd, STATUS_ERROR); + return NULL; + } + *skipped = false; + char* check_path = receive_wire_str(fd); + if (check_path == NULL) { + return NULL; + } + + unsigned long long check_size; + long long check_mtime; + long long check_mtime_nsec; + uint8_t check_digest[CHECKSUM_MAX_DIGEST_LEN]; + size_t check_digest_len = 0; + if (!receive_n_data(fd, &check_size, sizeof(check_size)) || + !receive_n_data(fd, &check_mtime, sizeof(check_mtime))) { + free(check_path); + return NULL; + } + if (!receive_n_data(fd, &check_mtime_nsec, sizeof(check_mtime_nsec)) || check_mtime_nsec < 0 || + check_mtime_nsec >= 1000000000LL) { + free(check_path); + send_status(fd, STATUS_ERROR); + return NULL; + } + if ((config->checksum || config_has_basis(config))) { + uint8_t wire_len; + if (!receive_n_data(fd, &wire_len, sizeof(wire_len)) || wire_len == 0 || + wire_len > CHECKSUM_MAX_DIGEST_LEN || + wire_len != checksum_digest_len((ChecksumAlgo)config->checksum_algo)) { + free(check_path); + send_status(fd, STATUS_ERROR); + return NULL; + } + check_digest_len = wire_len; + if (!receive_n_data(fd, check_digest, check_digest_len)) { + free(check_path); + return NULL; + } + } + + if (check_size > MAX_RECEIVE_WHOLE_FILE_SIZE) { + free(check_path); + send_status(fd, STATUS_ERROR); + return NULL; + } + + if (has_path_traversal(check_path)) { + char* escaped_path = output_escape(check_path, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "Path traversal detected: %s", + escaped_path ? escaped_path : ""); + free(escaped_path); + free(check_path); + return NULL; + } + + char* full_path = path_cat(config->receive_root_directory, check_path); + if (!full_path) { + free(check_path); + send_status(fd, STATUS_ERROR); + return NULL; + } + + /* Open the existing destination entry (if any) once and keep the descriptor + until the quick-check below decides whether the old contents are needed. */ + struct stat st; + bool has_old_file = false; + int old_fd = -1; + char* leaf = NULL; + int parent_fd = file_open_secure_parent(full_path, &leaf, false); + if (parent_fd >= 0) { + old_fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW); + free(leaf); + close(parent_fd); + has_old_file = old_fd >= 0 && fstat(old_fd, &st) == 0 && S_ISREG(st.st_mode); + } + if (!has_old_file && old_fd >= 0) { + close(old_fd); + old_fd = -1; + } + unsigned long long old_size = has_old_file ? (unsigned long long)st.st_size : 0; + + /* Decide from metadata alone whether the receiver already holds the file + the sender is offering. The old contents are only read into memory when + a checksum comparison or a delta transfer actually requires them. */ + bool size_equal = has_old_file && old_size == check_size; + bool match_by_metadata = false; + if (size_equal && !config->ignore_times && !config->size_only) { + long long old_mtime_nsec = 0; +#ifdef __linux__ + old_mtime_nsec = st.st_mtim.tv_nsec; +#endif + match_by_metadata = metadata_mtime_matches(st.st_mtime, old_mtime_nsec, (time_t)check_mtime, + (long)check_mtime_nsec, config->modify_window); + } + + bool try_delta = config->use_delta && !config->whole_file && has_old_file && + delta_should_attempt(old_size, check_size, config->delta_max_file_size); + bool checksum_needs_read = size_equal && !config->ignore_times && config->checksum; + bool need_old_data = checksum_needs_read || try_delta; + + void* old_data = NULL; + if (need_old_data && has_old_file && old_size > 0 && old_size <= MAX_RECEIVE_WHOLE_FILE_SIZE && + old_size <= SIZE_MAX) { + old_data = protocol_alloc((size_t)old_size); + if (old_data) { + size_t got = 0; + while (got < (size_t)old_size) { + ssize_t n = read(old_fd, (char*)old_data + got, (size_t)old_size - got); + if (n <= 0) { + free(old_data); + old_data = NULL; + break; + } + got += (size_t)n; + } + } + } + + /* Quick-skip decision. If no content comparison is required this is final + and the old file was never read; if the read failed the file is not + skipped and the transfer proceeds with the full new contents. */ + bool match = false; + if (checksum_needs_read) { + uint8_t old_digest[CHECKSUM_MAX_DIGEST_LEN]; + size_t old_len = 0; + bool hashed = checksum_digest((ChecksumAlgo)config->checksum_algo, config->checksum_seed, + old_size == 0 ? "" : old_data, (size_t)old_size, old_digest, + sizeof(old_digest), &old_len); + match = hashed && old_len == check_digest_len && check_digest_len > 0 && + memcmp(old_digest, check_digest, check_digest_len) == 0; + } else if (size_equal && !config->ignore_times) { + match = config->size_only || match_by_metadata; + } + + if (match) { + free(old_data); + if (!send_status(fd, STATUS_OK)) { + close(old_fd); + free(full_path); + free(check_path); + return NULL; + } + close(old_fd); + free(full_path); + free(check_path); + *skipped = true; + return NULL; + } + + /* ---- Alternate basis directories ---- */ + if (config_has_basis(config)) { + BasisMatch basis; + basis_match_find(config, check_path, check_size, (time_t)check_mtime, (long)check_mtime_nsec, + check_digest, check_digest_len, true, &basis); + if (basis.hit) { + if (basis.type == BASIS_DEST_COMPARE) { + /* compare-dest never copies: an exact match only suppresses the data + for a file the destination does not already hold (sparse backup). + When the destination holds a DIFFERENT version FastSync falls back to + a normal transfer rather than deleting the stale entry the way rsync + does (see RSYNC_COMPAT.md). */ + basis_match_free(&basis); + if (!has_old_file) { + if (!send_status(fd, STATUS_OK)) { + close(old_fd); + free(full_path); + free(check_path); + return NULL; + } + free(old_data); + close(old_fd); + free(full_path); + free(check_path); + *skipped = true; + return NULL; + } + } else { + /* copy-dest / link-dest: materialize the unchanged file locally so the + sender can skip the data. The store engine re-applies the normal + existing/ignore-existing/update/backup/delay-updates policy. */ + File* materialized = file_create(check_path); + if (materialized && basis.content) { + materialized->data = basis.content; + basis.content = NULL; + materialized->metadata = file_metadata_create(NULL, &basis.st, false, false); + materialized->skip = true; /* receiver must not ack this as a data file */ + if (basis.type == BASIS_DEST_LINK) { + materialized->basis_link = basis.basis_path; + basis.basis_path = NULL; + } + if (!materialized->metadata) { + file_destroy(materialized); + materialized = NULL; + } + } else { + file_destroy(materialized); + materialized = NULL; + } + if (materialized) { + if (!send_status(fd, STATUS_OK)) { + basis_match_free(&basis); + file_destroy(materialized); + close(old_fd); + free(full_path); + free(check_path); + return NULL; + } + basis_match_free(&basis); + free(old_data); + close(old_fd); + free(full_path); + free(check_path); + *skipped = false; + return materialized; + } + /* Materialization setup failed: fall through to the normal transfer. */ + } + } + basis_match_free(&basis); + } + + /* ---- --append / --append-verify tail resume ---- + * When the existing destination file is SHORTER than the source, an append + * mode resumes it by negotiating a resume offset (the prefix length already + * present) from the receiver and transferring ONLY the tail. The receiver + * then reconstructs the full file (prefix + tail) and installs it through the + * normal atomic store path, so the result is byte-identical to the source. + * This takes precedence over block delta (a growing file is cheapest as a + * pure tail), and falls through to delta/full only when no shorter old file + * makes a resume possible. */ + bool append_resume = (config->append || config->append_verify) && has_old_file && + append_resume_eligible(old_size, check_size); + if (append_resume) { + /* Ensure the retained prefix (== the whole, shorter destination file) is + in memory; it is needed both to rebuild the full file and, for + --append-verify, to checksum it. A load failure is not fatal: the + resume is simply not possible and we fall through to the other paths. */ + if (old_data == NULL && old_size > 0 && old_size <= MAX_RECEIVE_WHOLE_FILE_SIZE && + old_size <= SIZE_MAX) { + old_data = protocol_alloc((size_t)old_size); + if (old_data) { + size_t got = 0; + while (got < (size_t)old_size) { + ssize_t n = read(old_fd, (char*)old_data + got, (size_t)old_size - got); + if (n <= 0) { + free(old_data); + old_data = NULL; + break; + } + got += (size_t)n; + } + } + } + if (old_data != NULL || old_size == 0) { + if (!send_status(fd, STATUS_APPEND) || !send_n_data(fd, &old_size, sizeof(old_size))) { + close(old_fd); + free(full_path); + free(check_path); + free(old_data); + return NULL; + } + bool verify = config->append_verify; + bool full_fallback = false; + if (verify) { + Status sig_status; + if (!receive_status(fd, &sig_status)) { + close(old_fd); + free(full_path); + free(check_path); + free(old_data); + return NULL; + } + if (sig_status != STATUS_APPEND_SIG) { + send_status(fd, STATUS_ERROR); + close(old_fd); + free(full_path); + free(check_path); + free(old_data); + return NULL; + } + uint64_t src_prefix_hash; + if (!receive_n_data(fd, &src_prefix_hash, sizeof(src_prefix_hash))) { + close(old_fd); + free(full_path); + free(check_path); + free(old_data); + return NULL; + } + /* Compare the retained prefix against the source prefix. A mismatch + must never be silently appended to: fall back to a full transfer so + the result is a byte-identical source copy. */ + uint64_t dst_prefix_hash = + old_size == 0 ? delta_xxhash64("", 0) : delta_xxhash64(old_data, (size_t)old_size); + if (dst_prefix_hash == src_prefix_hash) { + if (!send_status(fd, STATUS_APPEND_OK)) { + close(old_fd); + free(full_path); + free(check_path); + free(old_data); + return NULL; + } + } else { + if (!send_status(fd, STATUS_NEXT)) { + close(old_fd); + free(full_path); + free(check_path); + free(old_data); + return NULL; + } + full_fallback = true; + } + } + + if (full_fallback) { + /* Retained prefix differed: receive the sender's full transfer. */ + free(old_data); + old_data = NULL; + close(old_fd); + File* file = receive_full_file(fd, config, check_path); + free(check_path); + free(full_path); + return file; + } + + /* Receive the tail (STATUS_APPEND_DATA + metadata + tail bytes). */ + Status tail_status; + if (!receive_status(fd, &tail_status)) { + close(old_fd); + free(full_path); + free(check_path); + free(old_data); + return NULL; + } + if (tail_status != STATUS_APPEND_DATA) { + send_status(fd, STATUS_ERROR); + close(old_fd); + free(full_path); + free(check_path); + free(old_data); + return NULL; + } + FileMetadata* meta = NULL; + FileXattrList* append_xattrs = NULL; + if (config->use_metadata) { + int meta_ok = 1; + meta = metadata_receive(fd, &meta_ok); + if (!meta_ok) { + close(old_fd); + free(full_path); + free(check_path); + free(old_data); + return NULL; + } + } + if (config->use_xattrs) { + int xok = 0; + append_xattrs = xattr_receive(fd, &xok); + if (!xok) { + xattr_list_free(append_xattrs); + close(old_fd); + free(full_path); + free(check_path); + free(old_data); + return NULL; + } + } + Data* tail = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); + if (tail == NULL) { + xattr_list_free(append_xattrs); + close(old_fd); + free(full_path); + free(check_path); + free(old_data); + return NULL; + } + if (config->use_compression && + !compression_should_skip_with_suffixes( + check_path, config->skip_compress_suffixes, + config->skip_compress_set ? config->skip_compress_count : -1)) { + Data* uncompressed = data_decompress_limited(tail, MAX_RECEIVE_WHOLE_FILE_SIZE); + data_destroy(tail); + if (uncompressed == NULL) { + xattr_list_free(append_xattrs); + close(old_fd); + free(full_path); + free(check_path); + free(old_data); + return NULL; + } + if (uncompressed->size > MAX_FILE_DATA_SIZE) { + data_destroy(uncompressed); + xattr_list_free(append_xattrs); + close(old_fd); + free(full_path); + free(check_path); + free(old_data); + return NULL; + } + tail = uncompressed; + } + /* The tail must complete the file exactly; anything else is a protocol + violation (never a truncated or overrun file). */ + unsigned long long expected_tail; + if (!append_tail_length(old_size, check_size, &expected_tail) || + tail->size != (size_t)expected_tail) { + send_status(fd, STATUS_ERROR); + data_destroy(tail); + xattr_list_free(append_xattrs); + close(old_fd); + free(full_path); + free(check_path); + free(old_data); + return NULL; + } + size_t full_size = (size_t)check_size; + void* full = protocol_alloc(full_size ? full_size : 1); + if (!full) { + data_destroy(tail); + xattr_list_free(append_xattrs); + close(old_fd); + free(full_path); + free(check_path); + free(old_data); + return NULL; + } + if (old_size > 0 && old_data) + memcpy(full, old_data, (size_t)old_size); + if (tail->size > 0) + memcpy((char*)full + old_size, tail->data, tail->size); + data_destroy(tail); + free(old_data); + old_data = NULL; + + File* file = file_create(check_path); + if (!file) { + free(full); + xattr_list_free(append_xattrs); + close(old_fd); + free(full_path); + free(check_path); + return NULL; + } + file->metadata = meta; + file->xattrs = append_xattrs; + append_xattrs = NULL; + file->data = data_create(full, full_size); + if (!file->data) { /* data_create already freed full on failure */ + file_destroy(file); + close(old_fd); + free(full_path); + free(check_path); + return NULL; + } + close(old_fd); + free(full_path); + free(check_path); + return file; + } + } + + if (try_delta && old_data != NULL) { + bool delta_failed = false; + File* delta_file = + receive_delta_file(fd, config, check_path, old_data, old_size, &delta_failed); + old_data = NULL; /* receive_delta_file consumes the snapshot on every path */ + if (delta_file) { + close(old_fd); + free(full_path); + free(check_path); + return delta_file; + } + if (delta_failed) { + close(old_fd); + free(full_path); + free(check_path); + return NULL; + } + } + free(old_data); + old_data = NULL; + + /* ---- -y/--fuzzy similar-file delta basis ---- + * Reaching this point means the file must be transferred and the + * destination's own content at the exact path could not serve as a delta + * basis (it is absent, outside the delta size bounds, or unreadable). With + * --fuzzy the receiver tries an existing similar-named file in the same + * destination directory instead. receive_delta_file performs the whole + * handshake: when the sender judges the delta not worthwhile it replies + * STATUS_NEXT and the full content is received there, so a fuzzy attempt + * can only improve bandwidth, never fall through into the plain transfer + * below (that path is reserved for "no usable candidate was found"). */ + if (config->fuzzy && config->use_delta) { + unsigned long long fuzzy_size = 0; + void* fuzzy_basis = fuzzy_basis_find_and_load(config, check_path, check_size, &fuzzy_size); + if (fuzzy_basis != NULL) { + bool fuzzy_failed = false; + File* fuzzy_file = + receive_delta_file(fd, config, check_path, fuzzy_basis, fuzzy_size, &fuzzy_failed); + fuzzy_basis = NULL; /* receive_delta_file consumes the buffer on every path */ + if (fuzzy_file) { + close(old_fd); + free(full_path); + free(check_path); + return fuzzy_file; + } + if (fuzzy_failed) { + close(old_fd); + free(full_path); + free(check_path); + return NULL; + } + } + free(fuzzy_basis); + } + + if (!send_status(fd, STATUS_NEXT)) { + close(old_fd); + free(full_path); + free(check_path); + return NULL; + } + close(old_fd); + + File* file = receive_full_file(fd, config, check_path); + free(check_path); + free(full_path); + return file; +} + +File* file_receive(const Config* config, int file_descriptor) { + char* path = receive_wire_str(file_descriptor); + if (path == NULL) + return NULL; + if (path[0] == '\0' || (!file_get_trust_sender() && has_path_traversal(path))) { + char* escaped_path = output_escape(path, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "Invalid received file path: %s", + escaped_path ? escaped_path : ""); + free(escaped_path); + free(path); + return NULL; + } + File* file = file_create(path); + free(path); + if (file == NULL) + return NULL; + if (config->use_metadata) { + int meta_ok = 1; + file->metadata = metadata_receive(file_descriptor, &meta_ok); + if (!meta_ok) { + file_destroy(file); + return NULL; + } + } + if (!receive_file_xattrs(file, file_descriptor, config)) { + file_destroy(file); + return NULL; + } + Data* file_data = receive_data_limited(file_descriptor, MAX_RECEIVE_WHOLE_FILE_SIZE); + if (file_data == NULL) { + file_destroy(file); + return NULL; + } + if (config->use_compression && + !compression_should_skip_with_suffixes(file->path, config->skip_compress_suffixes, + config->skip_compress_set ? config->skip_compress_count + : -1)) { + Data* file_data_uncompressed = data_decompress_limited(file_data, MAX_RECEIVE_WHOLE_FILE_SIZE); + data_destroy(file_data); + if (file_data_uncompressed == NULL) { + file_destroy(file); + return NULL; + } + if (file_data_uncompressed->size > MAX_FILE_DATA_SIZE) { + data_destroy(file_data_uncompressed); + file_destroy(file); + return NULL; + } + file_data = file_data_uncompressed; + } + data_destroy(file->data); + file->data = file_data; + return file; +} + +/* ---- P7 Wave D: deferred directory times ---- */ + +void dir_time_list_init(DirTimeList* list) { + if (!list) + return; + list->paths = NULL; + list->entries = NULL; + list->count = 0; + list->capacity = 0; +} + +void dir_time_list_free(DirTimeList* list) { + if (!list) + return; + for (size_t i = 0; i < list->count; i++) + free(list->paths[i]); + free(list->paths); + free(list->entries); + list->paths = NULL; + list->entries = NULL; + list->count = 0; + list->capacity = 0; +} + +bool dir_time_list_add(DirTimeList* list, const char* wire_path, const FileMetadata* metadata) { + if (!list || !wire_path || !metadata) + return true; /* nothing to remember; never a hard error */ + if (list->count == list->capacity) { + size_t new_capacity = list->capacity == 0 ? 16 : list->capacity * 2; + if (new_capacity < list->capacity) + return false; + /* Assign each grown array as soon as its realloc succeeds: the old block is + already freed by then, so discarding the pointer would dangle. capacity + is advanced only after BOTH reallocs succeed, so a partial failure leaves + capacity no larger than the entries allocation (the paths array may be + over-allocated, which is harmless) -- never a mismatched list the next + add could write past. */ + char** grown_paths = realloc(list->paths, new_capacity * sizeof(char*)); + if (!grown_paths) + return false; + list->paths = grown_paths; + FileMetadata* grown_entries = realloc(list->entries, new_capacity * sizeof(FileMetadata)); + if (!grown_entries) + return false; + list->entries = grown_entries; + list->capacity = new_capacity; + } + char* copy = str_dup(wire_path); + if (!copy) + return false; + list->paths[list->count] = copy; + list->entries[list->count] = *metadata; + list->count++; + return true; +} + +void dir_time_list_apply(const DirTimeList* list, const char* root_directory) { + if (!list || !root_directory) + return; + for (size_t i = 0; i < list->count; i++) { + char* dir_path = path_cat(root_directory, list->paths[i]); + if (!dir_path) + continue; + char* leaf = NULL; + /* The parent walk is fd-relative and O_NOFOLLOW, so a symlink planted in a + parent component can never redirect the utimensat outside the root. */ + int parent_fd = file_open_secure_parent(dir_path, &leaf, false); + if (parent_fd < 0) { + free(dir_path); + continue; + } + /* A dir-time entry only records metadata: the directory is (deliberately) + not created from it, so an empty source directory (or one pruned by + -m/--prune-empty-dirs) may well not exist here. Skip absent paths + QUIETLY rather than warning for every one, and apply the times only to a + real directory that does exist. AT_SYMLINK_NOFOLLOW keeps a same-named + symlink from being followed; a pre-existing regular file/symlink is not a + directory, so it is left completely untouched. */ + struct stat st; + if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) != 0 || !S_ISDIR(st.st_mode)) { + close(parent_fd); + free(leaf); + free(dir_path); + continue; + } + struct timespec times[2] = { + {.tv_sec = 0, .tv_nsec = UTIME_OMIT}, + {.tv_sec = list->entries[i].mtime_sec, .tv_nsec = list->entries[i].mtime_nsec}}; + if (list->entries[i].atime_valid) { + times[0].tv_sec = list->entries[i].atime_sec; + times[0].tv_nsec = list->entries[i].atime_nsec; + } + if (utimensat(parent_fd, leaf, times, AT_SYMLINK_NOFOLLOW) != 0) { + char* escaped_path = output_escape(dir_path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "Failed to set directory timestamps on %s: %s", + escaped_path ? escaped_path : "", strerror(errno)); + free(escaped_path); + } + close(parent_fd); + free(leaf); + free(dir_path); + } +} + +/* Receive an explicit directory entry (--dirs): a STATUS_MKDIR frame carries + the destination path and, when metadata is negotiated, the directory's + metadata frame. The same path validation as a regular file applies + (non-empty, relative-or-mirrored, no traversal), and the created File is + routed through the regular store_file sink so single-threaded and -m + receivers handle directories identically. The metadata is NOT applied here: + the sink accumulates it into a DirTimeList that is applied only after the + whole transfer (children would otherwise clobber the directory mtime). */ +File* file_receive_directory(int file_descriptor, const Config* config) { + char* path = receive_wire_str(file_descriptor); + if (path == NULL) + return NULL; + if (path[0] == '\0' || (!file_get_trust_sender() && has_path_traversal(path))) { + char* escaped_path = output_escape(path, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "Invalid received directory path: %s", + escaped_path ? escaped_path : ""); + free(escaped_path); + free(path); + return NULL; + } + File* file = file_create(path); + free(path); + if (file == NULL) + return NULL; + file->is_dir = true; + if (config && config->use_metadata) { + int meta_ok = 1; + file->metadata = metadata_receive(file_descriptor, &meta_ok); + if (!meta_ok) { + file_destroy(file); + return NULL; + } + } + return file; +} + +/* Receive one directory-time entry from a STATUS_DIR_TIMES frame: the + * destination-relative wire path and (when metadata is negotiated) the + * directory's metadata frame. The created File is an is_dir, dir_time_only + * entry routed through the regular store_file sink: the sink records its + * metadata into the deferred DirTimeList but never creates the directory (the + * scanner captures every traversed directory, including empty ones). Unlike a + * STATUS_MKDIR entry, this one must not create anything. */ +File* file_receive_dir_time(int file_descriptor, const Config* config) { + char* path = receive_wire_str(file_descriptor); + if (path == NULL) + return NULL; + if (path[0] == '\0' || (!file_get_trust_sender() && has_path_traversal(path))) { + char* escaped_path = output_escape(path, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "Invalid received directory-time path: %s", + escaped_path ? escaped_path : ""); + free(escaped_path); + free(path); + return NULL; + } + File* file = file_create(path); + free(path); + if (!file) + return NULL; + file->is_dir = true; + file->dir_time_only = true; + if (config && config->use_metadata) { + int meta_ok = 1; + file->metadata = metadata_receive(file_descriptor, &meta_ok); + if (!meta_ok) { + file_destroy(file); + return NULL; + } + } + return file; +} + +/* Receive a --hard-links/-H sibling frame (the leading STATUS_HARDLINK code has + already been consumed): the destination path, the run-local link-group id, + and the first (data-carrying) member's destination-relative wire path. The + created File carries no payload; it is installed beneath the receive root as + a hard link to (or, on link failure, a byte-identical copy of) the first + member. All paths are validated like every other received path (non-empty, + relative, no traversal). */ +File* file_receive_hardlink(int file_descriptor) { + char* path = receive_wire_str(file_descriptor); + if (path == NULL) + return NULL; + if (path[0] == '\0' || (!file_get_trust_sender() && has_path_traversal(path))) { + char* escaped_path = output_escape(path, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "Invalid received hard-link path: %s", + escaped_path ? escaped_path : ""); + free(escaped_path); + free(path); + send_status(file_descriptor, STATUS_ERROR); + return NULL; + } + int gid; + if (!receive_int(file_descriptor, &gid) || gid <= 0) { + free(path); + return NULL; + } + char* target = receive_wire_str(file_descriptor); + if (!target) { + free(path); + return NULL; + } + if (target[0] == '\0' || (!file_get_trust_sender() && has_path_traversal(target))) { + char* escaped = output_escape(target, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "Invalid hard-link target path: %s", + escaped ? escaped : ""); + free(escaped); + free(target); + free(path); + send_status(file_descriptor, STATUS_ERROR); + return NULL; + } + File* file = file_create(path); + free(path); + if (file == NULL) { + free(target); + return NULL; + } + file->link_group = gid; + file->link_first = false; + file->hardlink_target = target; + return file; +} + +/* Receive a symlink entry (the leading STATUS_SYMLINK code has already been + consumed): the destination path and the (sender-munged, if --munge-links) + symlink target string, then metadata when negotiated. The created File is + routed through the regular store_file sink, which creates the link beneath + the receive root (unmungeing the target first). */ +File* file_receive_symlink(int file_descriptor, const Config* config) { + char* path = receive_wire_str(file_descriptor); + if (path == NULL) + return NULL; + if (path[0] == '\0' || (!file_get_trust_sender() && has_path_traversal(path))) { + char* escaped_path = output_escape(path, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "Invalid received symlink path: %s", + escaped_path ? escaped_path : ""); + free(escaped_path); + free(path); + send_status(file_descriptor, STATUS_ERROR); + return NULL; + } + char* target = receive_wire_str(file_descriptor); + if (!target) { + free(path); + return NULL; + } + if (target[0] == '\0') { + char* escaped = output_escape(target, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "Invalid received symlink target: %s", + escaped ? escaped : ""); + free(escaped); + free(target); + free(path); + send_status(file_descriptor, STATUS_ERROR); + return NULL; + } + File* file = file_create(path); + free(path); + if (!file) { + free(target); + return NULL; + } + if (config && config->use_metadata) { + int meta_ok = 1; + file->metadata = metadata_receive(file_descriptor, &meta_ok); + if (!meta_ok) { + file_destroy(file); + free(target); + return NULL; + } + } + file->is_symlink = true; + file->symlink_target = target; + return file; +} + +/* Receive a device/special node frame (--devices/--specials): the leading + * STATUS_SPECIAL code has already been consumed. Payload: the destination path, + * the metadata frame (whose mode's S_IFMT bits carry the node kind), and two + * int32 rdev major/minor fields. The created File carries no payload and is + * recreated by file_save_to_disk_full (mknod/mkfifo, privilege-gated and + * confined). rdev is validated here (non-negative, range-checked) so a bogus + * value cannot drive a dangerous node on the receiver. */ +File* file_receive_special(int file_descriptor) { + char* path = receive_wire_str(file_descriptor); + if (path == NULL) + return NULL; + if (path[0] == '\0' || (!file_get_trust_sender() && has_path_traversal(path))) { + char* escaped_path = output_escape(path, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "Invalid received special path: %s", + escaped_path ? escaped_path : ""); + free(escaped_path); + free(path); + send_status(file_descriptor, STATUS_ERROR); + return NULL; + } + int meta_ok = 1; + FileMetadata* metadata = metadata_receive(file_descriptor, &meta_ok); + if (!meta_ok) { + free(path); + send_status(file_descriptor, STATUS_ERROR); + return NULL; + } + int32_t major = 0; + int32_t minor = 0; + if (!receive_n_data(file_descriptor, &major, sizeof(major)) || + !receive_n_data(file_descriptor, &minor, sizeof(minor))) { + free(path); + file_metadata_destroy(metadata); + send_status(file_descriptor, STATUS_ERROR); + return NULL; + } + /* A node kind must be present; without metadata mode there is no S_IFMT to + recreate from. */ + if (!metadata) { + log_message(LOG_LEVEL_ERROR, "Special node sent without metadata (mode)"); + free(path); + send_status(file_descriptor, STATUS_ERROR); + return NULL; + } + if (!file_special_rdev_valid(major, minor, metadata->mode)) { + log_message(LOG_LEVEL_ERROR, "Invalid special rdev received (%d:%d)", (int)major, (int)minor); + free(path); + file_metadata_destroy(metadata); + send_status(file_descriptor, STATUS_ERROR); + return NULL; + } + File* file = file_create(path); + free(path); + if (file == NULL) { + file_metadata_destroy(metadata); + return NULL; + } + file->metadata = metadata; + file->is_special = true; + file->rdev_major = major; + file->rdev_minor = minor; + return file; +} + +/* Read a delete-manifest frame (the STATUS_MANIFEST leading code has already + been consumed): a keep-set entry count followed by that many + destination-relative paths, then a protected-prefix count followed by that + many destination-relative prefixes, then (protocol 2.10.0+) a missing-args + count followed by that many destination-relative delete paths. The frame is + self-delimiting (the counts are authoritative), so the caller decides what to + do next and continues reading the following STATUS_* frame. Every section is + validated identically: an entry must be non-empty, relative and traversal-free + and the aggregate length across ALL sections is capped by MAX_MANIFEST_BYTES + (so the missing-args deletion requests are confined like the rest of the + manifest). Returns an owned DeleteManifest, or NULL after sending STATUS_ERROR + when the frame is malformed (bad count, empty/absolute path, path traversal, + or an aggregate size beyond MAX_MANIFEST_BYTES). */ +static bool receive_manifest_section(int fd, ArrayList* list, size_t* manifest_bytes) { + int count; + if (!receive_int(fd, &count)) { + send_status(fd, STATUS_ERROR); + return false; + } + if (count < 0 || count > MAX_MANIFEST_ENTRIES) { + send_status(fd, STATUS_ERROR); + return false; + } + for (int i = 0; i < count; i++) { + char* s = receive_wire_str(fd); + size_t entry_size = s ? strlen(s) : 0; + if (!s || s[0] == '\0' || s[0] == '/' || has_path_traversal(s) || + entry_size > MAX_MANIFEST_BYTES - *manifest_bytes || + (*manifest_bytes += entry_size) > MAX_MANIFEST_BYTES || !array_list_add(list, s)) { + free(s); + send_status(fd, STATUS_ERROR); + return false; + } + } + return true; +} + +DeleteManifest* receive_manifest_entries(int fd) { + DeleteManifest* manifest = calloc(1, sizeof(DeleteManifest)); + if (!manifest) { + send_status(fd, STATUS_ERROR); + return NULL; + } + manifest->keeps = array_list_create(free); + manifest->protected = array_list_create(free); + manifest->missing = array_list_create(free); + if (!manifest->keeps || !manifest->protected || !manifest->missing) { + delete_manifest_free(manifest); + send_status(fd, STATUS_ERROR); + return NULL; + } + size_t manifest_bytes = 0; + if (!receive_manifest_section(fd, manifest->keeps, &manifest_bytes) || + !receive_manifest_section(fd, manifest->protected, &manifest_bytes) || + !receive_manifest_section(fd, manifest->missing, &manifest_bytes)) { + delete_manifest_free(manifest); + return NULL; + } + return manifest; +} + +void delete_manifest_free(DeleteManifest* manifest) { + if (!manifest) + return; + array_list_delete(manifest->keeps); + array_list_delete(manifest->protected); + array_list_delete(manifest->missing); + free(manifest); +} + +/* Remove every destination entry under the receive root that is not in the + keep-set, bounded by MAX_SERVER_DELETE_COUNT (or a smaller client + --max-delete=NUM, which is all-or-nothing), using the symlink-safe delete + walker. With --delay-updates the not-yet-published staging directory is a + direct child of the receive root and must not be treated as a set of extras; + the manifest's protected prefixes (paths excluded on the source) and the + alternate basis directories are never destination content and are skipped at + any depth. Prints a notice and returns true on success. */ +bool manifest_delete_extras(const Config* config, DeleteManifest* manifest) { + if (!config || !manifest || !manifest->keeps) + return false; + fprintf(stderr, "Deleting files not in manifest...\n"); + /* Protected entries: + - the --delay-updates staging name, protected only as a DIRECT child of the + receive root (a nested destination directory that happens to be named + .fastsync-stage is ordinary content); + - alternate basis directories (--compare-dest / --copy-dest / --link-dest) + at any depth: they are extra comparison snapshots the user pointed at, + not destination content, and deleting them would destroy the very files a + --link-dest run just linked into place; + - the sender-side protected prefixes (source paths excluded by filters), at + any depth, so an excluded destination mirror survives --delete unless + --delete-excluded opts back into removing it. */ + int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count + + (manifest->protected ? manifest->protected->size : 0); + DeleteSkipEntry* skips = NULL; + if (skip_count > 0) { + skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); + if (!skips) + return false; + int idx = 0; + if (config->delay_updates) { + skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; + skips[idx].top_level_only = true; + idx++; + } + for (int i = 0; i < config->basis_count; i++) { + skips[idx].prefix = config->basis_dirs[i].path; + skips[idx].top_level_only = false; + idx++; + } + for (int i = 0; i < manifest->protected->size; i++) { + skips[idx].prefix = (const char*)manifest->protected->items[i]; + skips[idx].top_level_only = false; + idx++; + } + } + /* A client --max-delete=NUM smaller than the server's hard bound replaces it + for this run; both still bound the walk. The walker is all-or-nothing, so + a run that would delete more than the bound removes nothing and fails with + an error that names the bound that was hit. */ + bool user_limited = + config->max_delete >= 0 && (size_t)config->max_delete < MAX_SERVER_DELETE_COUNT; + size_t cap = user_limited ? (size_t)config->max_delete : MAX_SERVER_DELETE_COUNT; + size_t deleted_count = 0; + DeleteWalkResult result = delete_extras_limited(config->receive_root_directory, manifest->keeps, + cap, skips, skip_count, &deleted_count); + free(skips); + if (result == DELETE_WALK_LIMIT_EXCEEDED) { + if (user_limited) { + log_message(LOG_LEVEL_ERROR, + "deletion stopped: the destination holds more than --max-delete=%d extraneous " + "entries; no files were deleted", + config->max_delete); + } else { + log_message(LOG_LEVEL_ERROR, + "deletion stopped: the destination holds more than %u extraneous entries " + "(server deletion limit); no files were deleted", + (unsigned)MAX_SERVER_DELETE_COUNT); + } + return false; + } + if (result != DELETE_WALK_OK) { + log_message(LOG_LEVEL_ERROR, "deletion failed while removing extraneous files"); + return false; + } + return true; +} + +/* --delete-missing-args exact-path deletions: each destination mirror in + manifest->missing is an explicit user request, so it is removed even when the + ordinary extras walk (with its protected prefixes) would leave it alone. The + --delay-updates staging directory and basis snapshots are receiver artifacts + and stay protected exactly as in the extras walker. A regular file or + symlink is unlinked, an empty directory removed, and a NON-empty directory is + removed recursively only when --delete or --force is in effect (rsync parity: + the man page says a non-empty directory mirror is only deleted with --force + or --delete); otherwise it is left with a warning and the run continues. A + mirror that does not exist is a no-op. Returns false only on a genuine error + (a confinement failure on a validated path or an I/O error), which fails the + run. */ +bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest) { + if (!config || !manifest) + return false; + if (!manifest->missing || manifest->missing->size == 0) + return true; + fprintf(stderr, "Deleting destination mirrors of missing source arguments...\n"); + int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count; + DeleteSkipEntry* skips = NULL; + if (skip_count > 0) { + skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); + if (!skips) + return false; + int idx = 0; + if (config->delay_updates) { + skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; + skips[idx].top_level_only = true; + idx++; + } + for (int i = 0; i < config->basis_count; i++) { + skips[idx].prefix = config->basis_dirs[i].path; + skips[idx].top_level_only = false; + idx++; + } + } + bool ok = true; + for (int i = 0; i < manifest->missing->size; i++) { + const char* rel = (const char*)manifest->missing->items[i]; + if (!rel || *rel == '\0' || *rel == '/' || has_path_traversal(rel)) { + /* Defensive only: receive_manifest_entries already validated every + section identically, so a controlled peer never reaches this branch. */ + log_message(LOG_LEVEL_ERROR, "invalid missing-args delete path"); + ok = false; + continue; + } + bool at_root = strchr(rel, '/') == NULL; + if (path_under_skip_prefix(rel, at_root, skips, skip_count)) { + char* escaped = output_escape(rel, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, + "missing-args path '%s' is protected (staging directory or basis snapshot); " + "not deleting", + escaped ? escaped : ""); + free(escaped); + continue; + } + char* full = path_cat(config->receive_root_directory, rel); + if (!full) { + ok = false; + continue; + } + char* leaf = NULL; + int parent_fd = file_open_secure_parent(full, &leaf, false); + if (parent_fd < 0) { + /* The mirror's parent directory may itself not exist on the destination + (a deeper missing entry whose leading directories were never created). + That is a no-op -- there is nothing to delete -- matching + file_remove_tree_secure's absent-path handling; only a genuine I/O + error (EACCES, a symlink loop, ...) fails the run. */ + bool absent = errno == ENOENT || errno == ENOTDIR; + free(full); + free(leaf); + if (!absent) + ok = false; + continue; + } + struct stat st; + if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) != 0) { + /* Already absent: nothing to delete (a no-op, not a deletion). */ + if (errno != ENOENT) + ok = false; + close(parent_fd); + free(leaf); + free(full); + continue; + } + bool removed = false; + if (S_ISDIR(st.st_mode)) { + if (unlinkat(parent_fd, leaf, AT_REMOVEDIR) == 0) { + removed = true; + } else if (errno == ENOTEMPTY || errno == EEXIST) { + close(parent_fd); + parent_fd = -1; + free(leaf); + leaf = NULL; + if (config->use_delete || config->force_delete) { + if (!file_remove_tree_secure(full)) + ok = false; + else + removed = true; + } else { + char* escaped = output_escape(rel, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, + "missing-args destination '%s' is a non-empty directory; use --force or " + "--delete to remove it", + escaped ? escaped : ""); + free(escaped); + } + } else if (errno != ENOENT) { + ok = false; + } + } else { + if (unlinkat(parent_fd, leaf, 0) == 0) { + removed = true; + } else if (errno != ENOENT) { + ok = false; + } + } + if (removed) { + char* escaped = output_escape(rel, log_get_8_bit_output()); + fprintf(stderr, " Deleted: %s\n", escaped ? escaped : ""); + free(escaped); + } + if (parent_fd >= 0) + close(parent_fd); + free(leaf); + free(full); + if (!ok) + break; + } + free(skips); + return ok; +} + +/* Commit every deletion family the manifest carries. The --delete-missing-args + exact-path deletions run FIRST: they are explicit user requests and must not + be blocked by the extras walker's filter-exclusion protection (a protected + leftover inside a missing-argument directory must not make that user-requested + removal fail). The ordinary extras walk then runs when --delete is active. + Returns true when there was nothing to do or every requested deletion + committed. */ +bool manifest_delete_all(const Config* config, DeleteManifest* manifest) { + if (!config || !manifest) + return false; + if (config->delete_missing_args && !manifest_delete_missing_args(config, manifest)) + return false; + if (config->use_delete && !manifest_delete_extras(config, manifest)) + return false; + return true; +} diff --git a/src/shared/file_receive.h b/src/shared/file_receive.h new file mode 100644 index 0000000..555a342 --- /dev/null +++ b/src/shared/file_receive.h @@ -0,0 +1,97 @@ +#ifndef FILE_RECEIVE_H +#define FILE_RECEIVE_H + +#include "config.h" +#include "file_types.h" +#include + +/* Server-side file receive/save path. */ + +File* file_receive(const Config* config, int file_descriptor); +File* file_receive_directory(int file_descriptor, const Config* config); +File* file_receive_dir_time(int file_descriptor, const Config* config); +File* file_receive_hardlink(int file_descriptor); +File* file_receive_symlink(int file_descriptor, const Config* config); +File* file_receive_special(int file_descriptor); +bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode); +File* receive_incremental_check(int fd, const Config* config, bool* skipped); + +/* P7 Wave D directory-time accumulator. The receiver collects the metadata of + * every directory it creates/receives (STATUS_MKDIR with metadata and/or the + * trailing STATUS_DIR_TIMES frame(s)) and applies the times only at the END of the + * transfer, after all children have been written and after the delete / + * --delay-updates phases have committed (writing or removing a child bumps the + * parent's mtime). -O/--omit-dir-times skips the application entirely. The + * list owns deep copies of the paths and metadata; freed on every path. */ +typedef struct { + char** paths; /* owned, destination-relative wire paths */ + FileMetadata* entries; /* owned, parallel to paths */ + size_t count; + size_t capacity; +} DirTimeList; + +void dir_time_list_init(DirTimeList* list); +void dir_time_list_free(DirTimeList* list); +/* Deep-copy one directory's path + metadata into the list. Returns false on + * allocation failure (the caller fails the transfer). */ +bool dir_time_list_add(DirTimeList* list, const char* wire_path, const FileMetadata* metadata); +/* Apply every accumulated directory's mtime (and atime when captured) beneath + * `root_directory`, confined fd-relative. Best-effort per entry: an absent + * directory (an empty/pruned source dir that was deliberately not created) or a + * non-directory at the path is skipped QUIETLY, an unreachable one with a + * warning, and never fatal. */ +void dir_time_list_apply(const DirTimeList* list, const char* root_directory); + +/* A received delete-manifest frame: the keep-set (`keeps`, destination-relative + paths the sender transferred/keeps) plus `protected`, destination-relative + prefixes the sender asks the receiver never to delete (paths excluded on the + source, protected at any depth). When --delete-excluded is given the sender + transmits an empty protected list so excluded destination mirrors are treated + as ordinary extras. With --delete-missing-args a third section (`missing`) + carries the destination mirrors of explicitly-listed source entries that do + not exist: each is an exact deletion request, independent of the ordinary + extras walk (never blocked by the protected prefixes) and processed when the + manifest is committed. */ +typedef struct DeleteManifest { + ArrayList* keeps; + ArrayList* protected; + ArrayList* missing; +} DeleteManifest; + +void delete_manifest_free(DeleteManifest* manifest); +/* Read a delete-manifest frame: keep count + keeps, then protected count + + protected prefixes, then missing count + missing paths (self-delimiting; the + leading STATUS_MANIFEST code has been consumed). Returns an owned + DeleteManifest, or NULL after signalling STATUS_ERROR on a malformed frame. */ +DeleteManifest* receive_manifest_entries(int fd); +/* Remove destination entries under config->receive_root_directory that are not + in `manifest` (bounded, all-or-nothing walk; staging-dir, basis-dir and + protected-prefix skips). `--max-delete` and `--force` are honored here. The + caller decides WHEN to run it based on the negotiated delete timing. Returns + false (and the transfer fails) when the deletion cannot be committed. */ +bool manifest_delete_extras(const Config* config, DeleteManifest* manifest); +/* --delete-missing-args exact-path deletions: remove each destination mirror + in `manifest->missing` (never blocked by the protected prefixes, staging dir + and basis dirs excluded). A regular file/symlink is unlinked; an empty + directory is removed; a NON-empty directory is removed recursively only when + --delete or --force is in effect, otherwise it is left with a warning (rsync + parity). A missing path is a no-op. Returns false only on a genuine + confinement or I/O error (the run then fails); tolerated per-path cases are + reported and skipped. */ +bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest); +/* Run every deletion family the manifest carries: the --delete-missing-args + exact-path deletions first (user requests are not blocked by exclusion + protection), then the ordinary extras walk when --delete is active. Returns + true when nothing to do or everything committed. */ +bool manifest_delete_all(const Config* config, DeleteManifest* manifest); + +/* Outcome of a single file_save_to_disk operation. The receiver needs to + distinguish "written" from "skipped" so --remove-source-files can be told + which sources were actually stored. */ +typedef enum { FILE_SAVE_ERROR = 0, FILE_SAVE_WRITTEN = 1, FILE_SAVE_SKIPPED = 2 } FileSaveResult; + +FileSaveResult file_save_to_disk_full(const char* root_directory, const File* file, + const Config* config); +bool file_save_to_disk(const char* root_directory, const File* file, const Config* config); + +#endif diff --git a/src/shared/file_send.c b/src/shared/file_send.c new file mode 100644 index 0000000..e7bcffb --- /dev/null +++ b/src/shared/file_send.c @@ -0,0 +1,181 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "charset.h" +#include "compression.h" +#include "data.h" +#include "file.h" +#include "log.h" +#include "metadata.h" +#include "protocol.h" +#include "xattr.h" + +/* Transmit a device/special node (--devices / --specials) as a STATUS_SPECIAL + * frame: the destination path, the metadata frame (whose mode's S_IFMT bits + * carry the node kind) and the device rdev major/minor. The receiver validates + * the kind and rdev and recreates the node (privilege-gating the mknod). */ +bool file_send_special(const File* file, int file_descriptor, bool use_metadata) { + if (!file || !file_wire_path(file)) + return false; + if (!send_status(file_descriptor, STATUS_SPECIAL)) + return false; + if (!send_wire_str(file_descriptor, file_wire_path(file))) + return false; + if (use_metadata && !metadata_send(file_descriptor, file->metadata)) + return false; + int32_t major = file->rdev_major; + int32_t minor = file->rdev_minor; + return send_n_data(file_descriptor, &major, sizeof(major)) && + send_n_data(file_descriptor, &minor, sizeof(minor)); +} + +bool file_send_single_calls(File* file, int file_descriptor, bool use_metadata, + int compression_level, bool send_path) { + return file_send_single_calls_with_skip(file, file_descriptor, use_metadata, compression_level, + send_path, NULL, -1, 0, false); +} + +bool file_send_single_calls_with_skip(File* file, int file_descriptor, bool use_metadata, + int compression_level, bool send_path, + char* const* skip_suffixes, int skip_count, + int compression_threads, bool send_xattrs) { + if (!file || !file->path || !file->data || (file->data->size != 0 && !file->data->data)) + return false; + const Data* data_to_send = file->data; + Data* compressed_data = NULL; + if (compression_level > 0 && + !compression_should_skip_with_suffixes(file->path, skip_suffixes, skip_count)) { + compressed_data = + data_compress_with_threads(file->data, compression_level, compression_threads); + if (compressed_data == NULL) { + log_message(LOG_LEVEL_ERROR, "Failed to compress file data"); + return false; + } + data_to_send = compressed_data; + } + if (send_path && !send_wire_str(file_descriptor, file_wire_path(file))) { + data_destroy(compressed_data); + return false; + } + if (use_metadata && !metadata_send(file_descriptor, file->metadata)) { + data_destroy(compressed_data); + return false; + } + if (send_xattrs && !xattr_send(file_descriptor, file ? file->xattrs : NULL)) { + data_destroy(compressed_data); + return false; + } + if (!send_data(file_descriptor, data_to_send)) { + data_destroy(compressed_data); + return false; + } + data_destroy(compressed_data); + return true; +} + +bool file_send_sendfile(File* file, int file_descriptor, bool use_metadata, int compression_level, + bool send_path) { + return file_send_sendfile_with_skip(file, file_descriptor, use_metadata, compression_level, + send_path, NULL, -1, 0, false); +} + +bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_metadata, + int compression_level, bool send_path, char* const* skip_suffixes, + int skip_count, int compression_threads, bool send_xattrs) { + if (!file || !file->path || !file->data) + return false; + if (compression_level > 0) + return file_send_single_calls_with_skip(file, file_descriptor, use_metadata, compression_level, + send_path, skip_suffixes, skip_count, + compression_threads, send_xattrs); + + if (send_path && !send_wire_str(file_descriptor, file_wire_path(file))) + return false; + if (use_metadata && !metadata_send(file_descriptor, file->metadata)) + return false; + if (send_xattrs && !xattr_send(file_descriptor, file ? file->xattrs : NULL)) + return false; + + int fd = file_open_for_read(file->path); + if (fd == -1) { + log_perror("Could not open file for sendfile"); + return false; + } + + unsigned long long file_size = file->data->size; + struct stat source_stat; + if (fstat(fd, &source_stat) != 0 || !S_ISREG(source_stat.st_mode) || + (unsigned long long)source_stat.st_size < file_size) { + close(fd); + return false; + } + if (!send_n_data(file_descriptor, &file_size, sizeof(unsigned long long))) { + close(fd); + return false; + } + + /* sendfile cannot encrypt TLS records. Keep the framing identical but + route encrypted transfers through the deadline-aware IO layer. */ + if (io_get_ssl() != NULL) { + unsigned char buffer[64 * 1024]; + unsigned long long remaining = file_size; + bool ok = true; + while (remaining > 0) { + size_t want = remaining > sizeof(buffer) ? sizeof(buffer) : (size_t)remaining; + ssize_t got = read(fd, buffer, want); + if (got <= 0 || !send_n_data(file_descriptor, buffer, (size_t)got)) { + ok = false; + break; + } + remaining -= (unsigned long long)got; + } + close(fd); + return ok; + } + + off_t offset = 0; + struct timespec deadline; + clock_gettime(CLOCK_MONOTONIC, &deadline); + deadline.tv_sec += 60; + while ((unsigned long long)offset < file_size) { + struct timespec now; + clock_gettime(CLOCK_MONOTONIC, &now); + long long remaining = (long long)(deadline.tv_sec - now.tv_sec) * 1000LL + + (deadline.tv_nsec - now.tv_nsec) / 1000000LL; + if (remaining <= 0) { + close(fd); + return false; + } + struct pollfd pfd = {.fd = file_descriptor, .events = POLLOUT}; + int timeout = remaining > INT_MAX ? INT_MAX : (int)remaining; + int polled = poll(&pfd, 1, timeout); + if (polled <= 0 || (pfd.revents & (POLLERR | POLLHUP | POLLNVAL))) { + close(fd); + return false; + } + ssize_t sent = sendfile(file_descriptor, fd, &offset, file_size - offset); + if (sent == -1) { + if (errno == EAGAIN || errno == EINTR) + continue; + log_perror("sendfile failed"); + close(fd); + return false; + } + if (sent == 0) { + close(fd); + return false; + } + } + + close(fd); + return true; +} diff --git a/src/shared/file_send.h b/src/shared/file_send.h new file mode 100644 index 0000000..b48d33c --- /dev/null +++ b/src/shared/file_send.h @@ -0,0 +1,22 @@ +#ifndef FILE_SEND_H +#define FILE_SEND_H + +#include "file_types.h" +#include + +/* Client-side file send path. */ + +bool file_send_special(const File* file, int file_descriptor, bool use_metadata); +bool file_send_single_calls(File* file, int file_descriptor, bool use_metadata, + int compression_level, bool send_path); +bool file_send_single_calls_with_skip(File* file, int file_descriptor, bool use_metadata, + int compression_level, bool send_path, + char* const* skip_suffixes, int skip_count, + int compression_threads, bool send_xattrs); +bool file_send_sendfile(File* file, int file_descriptor, bool use_metadata, int compression_level, + bool send_path); +bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_metadata, + int compression_level, bool send_path, char* const* skip_suffixes, + int skip_count, int compression_threads, bool send_xattrs); + +#endif diff --git a/src/shared/file_store.c b/src/shared/file_store.c new file mode 100644 index 0000000..b14da25 --- /dev/null +++ b/src/shared/file_store.c @@ -0,0 +1,242 @@ +#include +#include +#include +#include +#include +#include +#include +#include + +#include "file_store.h" +#include "metadata.h" +#include "utils.h" + +static int authorized_root_fd = -1; +static char* authorized_root_path; + +static bool path_is_within_root(const char* root, const char* path) { + size_t root_length = strlen(root); + return strncmp(root, path, root_length) == 0 && + (path[root_length] == '\0' || path[root_length] == '/'); +} + +bool file_store_set_authorized_root(int fd, const char* canonical_path) { + char* new_path = canonical_path ? str_dup(canonical_path) : NULL; + if (canonical_path && !new_path) { + authorized_root_fd = -1; + free(authorized_root_path); + authorized_root_path = NULL; + return false; + } + free(authorized_root_path); + authorized_root_path = new_path; + authorized_root_fd = fd; + return true; +} + +int file_store_open_secure_parent(const char* path, char** leaf_out) { + char* copy = str_dup(path); + if (!copy) + return -1; + char* parent = dirname(copy); + const char* slash = strrchr(path, '/'); + char* leaf = str_dup(slash ? slash + 1 : path); + if (!leaf) { + free(copy); + return -1; + } + int fd; + if (authorized_root_fd >= 0) { + if (!authorized_root_path || path[0] != '/' || + !path_is_within_root(authorized_root_path, path)) { + free(copy); + free(leaf); + return -1; + } + fd = dup(authorized_root_fd); + if (fd < 0) { + free(copy); + free(leaf); + return -1; + } + size_t root_length = strlen(authorized_root_path); + char* relative = str_dup(path + root_length); + if (!relative) { + free(copy); + free(leaf); + close(fd); + return -1; + } + free(copy); + copy = relative; + parent = dirname(copy); + } else { + fd = (parent[0] == '/') ? open("/", O_RDONLY | O_DIRECTORY | O_CLOEXEC) + : open(".", O_RDONLY | O_DIRECTORY | O_CLOEXEC); + } + if (fd < 0) { + free(copy); + free(leaf); + return -1; + } + char* save = NULL; + char* component = strtok_r(parent, "/", &save); + while (component) { + if (strcmp(component, "..") == 0) { + close(fd); + free(copy); + free(leaf); + return -1; + } + if (strcmp(component, ".") != 0) { + int next = openat(fd, component, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + if (next < 0 && errno == ENOENT) { + if (mkdirat(fd, component, 0755) == 0 || errno == EEXIST) + next = openat(fd, component, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + } + if (next < 0) { + close(fd); + free(copy); + free(leaf); + return -1; + } + close(fd); + fd = next; + } + component = strtok_r(NULL, "/", &save); + } + free(copy); + *leaf_out = leaf; + return fd; +} + +bool file_store_rename_secure(const char* old_path, const char* new_path) { + char *old_leaf = NULL, *new_leaf = NULL; + int old_parent = file_store_open_secure_parent(old_path, &old_leaf); + int new_parent = file_store_open_secure_parent(new_path, &new_leaf); + bool ok = old_parent >= 0 && new_parent >= 0 && + renameat(old_parent, old_leaf, new_parent, new_leaf) == 0; + if (old_parent >= 0) + close(old_parent); + if (new_parent >= 0) + close(new_parent); + free(old_leaf); + free(new_leaf); + return ok; +} + +static bool write_all(int fd, const void* data, unsigned long long size) { + const unsigned char* p = data; + unsigned long long done = 0; + while (done < size) { + ssize_t n = write(fd, p + done, (size_t)(size - done)); + if (n < 0 && errno == EINTR) + continue; + if (n <= 0) + return false; + done += (unsigned long long)n; + } + return true; +} + +/* A run of NUL bytes at least this long is emitted as a hole (lseek) rather + * than written, so the resulting file is genuinely sparse on the filesystem. */ +#define SPARSE_HOLE_MIN 4096U + +/* Sparse-aware writer (--sparse/-S). Walks `data`; any all-zero run of at + * least SPARSE_HOLE_MIN bytes is skipped with lseek(SEEK_CUR) so the block is + * never allocated (a real hole on the destination); every other byte is written + * normally. The file is pre-sized with ftruncate by the callers before this + * runs, so holes are guaranteed and the offset bookkeeping stays correct + * (each lseek advances the fd offset exactly as a write of that many bytes + * would). After the final run, ftruncate(size) guarantees the logical size is + * exactly `size` even when the tail was a hole. The full file image is in + * memory, so no wire change is needed. Returns false on I/O error. */ +bool file_store_write_sparse(int fd, const unsigned char* data, unsigned long long size) { + unsigned long long i = 0; + while (i < size) { + if (data[i] == 0) { + unsigned long long run_start = i; + while (i < size && data[i] == 0) + i++; + unsigned long long run_len = i - run_start; + if (run_len >= SPARSE_HOLE_MIN) { + if (lseek(fd, (off_t)run_len, SEEK_CUR) < 0) + return false; + } else if (!write_all(fd, data + run_start, run_len)) { + return false; + } + } else { + unsigned long long run_start = i; + while (i < size && data[i] != 0) + i++; + if (!write_all(fd, data + run_start, i - run_start)) + return false; + } + } + return ftruncate(fd, (off_t)size) == 0; +} + +bool file_store_write_secure(const char* path, const void* data, unsigned long long data_size, + bool inplace, bool sparse, const FileMetadata* metadata, + bool preserve_executability) { + char* leaf = NULL; + int dirfd = file_store_open_secure_parent(path, &leaf); + if (dirfd < 0) + return false; + int fd = -1; + bool ok = false; + if (inplace) { + fd = openat(dirfd, leaf, O_WRONLY | O_CREAT | O_TRUNC | O_CLOEXEC | O_NOFOLLOW, 0644); + if (fd >= 0) { + if (sparse && data_size > 0) { + if (ftruncate(fd, (off_t)data_size) == 0) + ok = file_store_write_sparse(fd, data, data_size); + } else { + ok = write_all(fd, data, data_size); + } + if (ok && metadata) + ok = file_restore_metadata_fd(fd, metadata, preserve_executability); + } + } else { + int tmp_size = snprintf(NULL, 0, ".%s.tmp.%ld.%u", leaf, (long)getpid(), 99U); + if (tmp_size < 0) { + close(dirfd); + free(leaf); + return false; + } + char* tmp = malloc((size_t)tmp_size + 1); + if (!tmp) { + close(dirfd); + free(leaf); + return false; + } + for (unsigned int i = 0; i < 100 && !ok; ++i) { + snprintf(tmp, (size_t)tmp_size + 1, ".%s.tmp.%ld.%u", leaf, (long)getpid(), i); + fd = openat(dirfd, tmp, O_WRONLY | O_CREAT | O_EXCL | O_CLOEXEC | O_NOFOLLOW, 0600); + if (fd < 0) + continue; + if (sparse && data_size > 0) + ok = ftruncate(fd, (off_t)data_size) == 0; + if (ok || (!sparse || data_size == 0)) + ok = (sparse && data_size > 0) + ? file_store_write_sparse(fd, (const unsigned char*)data, data_size) + : write_all(fd, data, data_size); + if (ok && metadata) + ok = file_restore_metadata_fd(fd, metadata, preserve_executability); + if (close(fd) != 0) + ok = false; + fd = -1; + if (ok && renameat(dirfd, tmp, dirfd, leaf) != 0) + ok = false; + if (!ok) + unlinkat(dirfd, tmp, 0); + } + free(tmp); + } + if (fd >= 0) + close(fd); + close(dirfd); + free(leaf); + return ok; +} diff --git a/src/shared/file_store.h b/src/shared/file_store.h new file mode 100644 index 0000000..4bfb39a --- /dev/null +++ b/src/shared/file_store.h @@ -0,0 +1,21 @@ +#ifndef FILE_STORE_H +#define FILE_STORE_H + +#include "file.h" +#include + +bool file_store_set_authorized_root(int fd, const char* canonical_path); +int file_store_open_secure_parent(const char* path, char** leaf_out); +bool file_store_rename_secure(const char* old_path, const char* new_path); +bool file_store_write_secure(const char* path, const void* data, unsigned long long data_size, + bool inplace, bool sparse, const FileMetadata* metadata, + bool preserve_executability); +/* Sparse-aware write (--sparse/-S): every all-zero run of at least + * SPARSE_HOLE_MIN bytes is skipped with lseek(SEEK_CUR) so it becomes a real + * hole; every other byte is written. The caller pre-sizes the file with + * ftruncate; this function also ftruncate()s to `size` at the end so a trailing + * hole keeps the exact logical length. Shared by the file_store and file write + * paths. Returns false on write/lseek/ftruncate error. */ +bool file_store_write_sparse(int fd, const unsigned char* data, unsigned long long size); + +#endif diff --git a/src/shared/file_types.h b/src/shared/file_types.h new file mode 100644 index 0000000..9564d82 --- /dev/null +++ b/src/shared/file_types.h @@ -0,0 +1,97 @@ +#ifndef FILE_TYPES_H +#define FILE_TYPES_H + +#include "data.h" +#include "xattr.h" +#include +#include + +typedef enum { FILE_TYPE_REGULAR, FILE_TYPE_SYMLINK, FILE_TYPE_DIR } FileType; + +typedef struct { + mode_t mode; + uid_t uid; + gid_t gid; + time_t mtime_sec; + long mtime_nsec; + /* Optional access time (-U/--atimes) and creation/birth time (-N/--crtimes), + * appended for protocol 2.12.0. The SENDER sets the corresponding *_valid + * flag only when the preserve option is active (and, for crtime, only when + * the source platform exposed a birth time via statx STATX_BTIME). The wire + * always carries the fields and the flags; a false flag tells the receiver to + * ignore the value. */ + bool atime_valid; + time_t atime_sec; + long atime_nsec; + bool crtime_valid; + time_t crtime_sec; + long crtime_nsec; +} FileMetadata; + +typedef struct { + char* path; + /* Sender-side override for the path transmitted on the wire (and used for + * the delete manifest / change output). NULL means "use `path`". With + * -R + --files-from this holds the entry's bare relative destination path, + * while `path` stays the absolute local source path the client reads from. + * Never populated on the receiver. */ + char* send_path; + Data* data; + FileMetadata* metadata; + bool skip; + /* True when this entry is an explicit directory entry (--dirs mode): the + * receiver creates the directory instead of writing a regular file. */ + bool is_dir; + /* Receiver-only (P7 Wave D): this is a STATUS_DIR_TIMES entry. It carries a + * traversed source directory's metadata for DEFERRED application, but must + * NEVER create the directory: the scanner captures every traversed directory + * (including empty ones whose parents no child write created), so creation + * would resurrect the empty dirs that FastSync deliberately never transfers. + * file_save_to_disk_full short-circuits such an entry as FILE_SAVE_SKIPPED, + * and the sink still accumulates the metadata into its DirTimeList. */ + bool dir_time_only; + /* Receiver-only, --link-dest: when set, install the destination entry as a + * hard link to this absolute (root-confined) path instead of writing + * `data`. The matching code has already verified the link target's content + * equals the incoming file, and `data` is kept as the cross-filesystem + * fallback (a local copy) if the hard link cannot be created. */ + char* basis_link; + /* --hard-links (-H), sender + receiver wire state. link_group is a run-local + * id shared by every member of one source inode (0 = not part of a group). + * The FIRST member (link_first == true) carries its data on the wire and is + * written normally; every sibling (link_first == false) carries NO data and + * hardlink_target holds the first member's wire path so the receiver can link + * to (or copy from) the already-installed first member. */ + int link_group; + bool link_first; + char* hardlink_target; + /* Symlink-type entry (-l/--links, or -k/--copy-dirlinks' keep-as-symlink + * branch). When true, `symlink_target` holds the (sender-munged, if + * --munge-links) target string that is carried on the wire; the receiver + * creates a symlink to (an unmunged) target instead of writing regular-file + * data. `data` is empty for a symlink entry. Sender + receiver state. */ + bool is_symlink; + char* symlink_target; + /* Phase 4 special/devices: when `is_special` is true this entry is a device + * or special node to be RECREATED on the destination (mknod/mkfifo) rather + * than written from `data`. The concrete node kind is derived from the + * metadata mode's S_IFMT bits (receiver-validated), and rdev_major/minor + * carry the device major/minor numbers for char/block devices. CROSSES the + * wire (protocol 2.13.0). */ + bool is_special; + int32_t rdev_major; + int32_t rdev_minor; + /* Phase-4 xattrs (-X/--xattrs, -A/--acls). Sender: captured from the source + * file when use_xattrs is set; transmitted in the per-file metadata frame. + * Receiver: parsed off the wire, attached here, and applied fd-relative on + * the written file. NULL/0 == the file carries no xattrs. */ + FileXattrList* xattrs; +} File; + +/* The path that should be sent on the wire and used for the receiver-side + * destination layout (see send_path). */ +static inline const char* file_wire_path(const File* file) { + return file && file->send_path ? file->send_path : (file ? file->path : NULL); +} + +#endif diff --git a/src/shared/filter.c b/src/shared/filter.c new file mode 100644 index 0000000..8e7ae77 --- /dev/null +++ b/src/shared/filter.c @@ -0,0 +1,427 @@ +#include "filter.h" +#include "log.h" +#include "utils.h" +#include +#include +#include +#include + +/* ---- Single rule parsing ---- */ + +static bool rule_text_is_unsupported_word(const char* p, size_t len) { + static const char* const words[] = {"merge", "dir-merge", "hide", "show", + "protect", "risk", "clear"}; + for (size_t i = 0; i < sizeof(words) / sizeof(words[0]); i++) { + size_t wl = strlen(words[i]); + if (len == wl && strncmp(p, words[i], wl) == 0) + return true; + } + return false; +} + +/* rsync include/exclude rule modifiers we do NOT implement. A rule whose +/- is + * immediately followed by one of these is rejected instead of being silently + * parsed as a literal pattern. */ +static bool is_unsupported_rule_modifier(char c) { + return c == '!' || c == 'C' || c == 's' || c == 'r' || c == 'p' || c == 'x'; +} + +FilterRule* filter_rule_parse(const char* line, char* err, size_t err_size) { + if (err && err_size > 0) + err[0] = '\0'; + if (!line) + return NULL; + char* text = str_dup(line); + if (!text) { + if (err) + snprintf(err, err_size, "memory allocation failed"); + return NULL; + } + size_t len = strlen(text); + while (len > 0 && (text[len - 1] == '\n' || text[len - 1] == '\r')) + text[--len] = '\0'; + + const char* p = text; + while (*p == ' ' || *p == '\t') + p++; + if (*p == '\0') { + snprintf(err, err_size, "empty filter rule"); + free(text); + return NULL; + } + + FilterAction action = FILTER_ACTION_EXCLUDE; + if (*p == '+' || *p == '-') { + action = *p == '+' ? FILTER_ACTION_INCLUDE : FILTER_ACTION_EXCLUDE; + p++; + /* rsync attaches rule modifiers directly to the +/- (e.g. "-s foo"). Only + * the '/' anchor modifier is supported; anything else is a clear error + * rather than a silently-ignored literal. */ + if (*p != ' ' && *p != '\t' && *p != '\0' && is_unsupported_rule_modifier(*p)) { + snprintf(err, err_size, + "filter rule modifier '%c' is not supported (only the '/' anchor after +/- " + "is implemented; put a space between +/- and the pattern)", + *p); + free(text); + return NULL; + } + while (*p == ' ' || *p == '\t') + p++; + } else { + /* ':' (dir-merge) and '.' (merge) are rsync filter-rule shorthands. At the + * start of a rule they mean "merge this file", so reject them instead of + * silently turning them into inert exclude patterns. */ + if (*p == ':' || *p == '.' || *p == '!') { + snprintf(err, err_size, + "filter rule starting with '%c' is not supported (merge/dir-merge/list-clear " + "shorthands are not implemented; use +/- include/exclude rules)", + *p); + free(text); + return NULL; + } + const char* sp = p; + while (*sp != '\0' && *sp != ' ' && *sp != '\t') + sp++; + size_t word_len = (size_t)(sp - p); + if (rule_text_is_unsupported_word(p, word_len)) { + snprintf(err, err_size, + "'%.*s' filter directives are not supported (only +/- include/exclude rules " + "with an optional '/' anchor and trailing '/' dir marker)", + (int)word_len, p); + free(text); + return NULL; + } + if (word_len == strlen("include") && strncmp(p, "include", word_len) == 0) { + action = FILTER_ACTION_INCLUDE; + p = sp; + } else if (word_len == strlen("exclude") && strncmp(p, "exclude", word_len) == 0) { + action = FILTER_ACTION_EXCLUDE; + p = sp; + } + while (*p == ' ' || *p == '\t') + p++; + } + + if (*p == '\0') { + snprintf(err, err_size, "filter rule has no pattern"); + free(text); + return NULL; + } + + /* A pattern beginning with '/' is anchored (either as "-/foo" or "- /foo"). */ + bool anchored = false; + if (*p == '/') { + anchored = true; + p++; + while (*p == ' ' || *p == '\t') + p++; + } + if (*p == '\0') { + snprintf(err, err_size, "filter rule has no pattern after '/' anchor"); + free(text); + return NULL; + } + + /* Pattern runs to the end of the rule; a single trailing '/' marks dir-only. */ + size_t pat_len = strlen(p); + bool dir_only = false; + if (pat_len > 1 && p[pat_len - 1] == '/') { + dir_only = true; + pat_len--; + } else if (pat_len == 1 && p[0] == '/') { + /* "//" anchored with nothing after: meaningless. */ + snprintf(err, err_size, "filter rule has no pattern"); + free(text); + return NULL; + } + + FilterRule* rule = calloc(1, sizeof(FilterRule)); + if (!rule) { + snprintf(err, err_size, "memory allocation failed"); + free(text); + return NULL; + } + rule->pattern = malloc(pat_len + 1); + if (!rule->pattern) { + free(rule); + snprintf(err, err_size, "memory allocation failed"); + free(text); + return NULL; + } + memcpy(rule->pattern, p, pat_len); + rule->pattern[pat_len] = '\0'; + rule->action = action; + rule->anchored = anchored; + rule->dir_only = dir_only; + rule->owner = NULL; + free(text); + return rule; +} + +void filter_rule_free(FilterRule* rule) { + if (!rule) + return; + free(rule->pattern); + free(rule->owner); + free(rule); +} + +/* ---- Ordered rule lists ---- */ + +FilterRuleList* filter_rule_list_create(void) { + return calloc(1, sizeof(FilterRuleList)); +} + +bool filter_rule_list_add(FilterRuleList* list, FilterRule* rule) { + if (!list || !rule) + return false; + if (list->count == list->capacity) { + int new_cap = list->capacity > 0 ? list->capacity * 2 : 8; + FilterRule** grown = realloc(list->items, (size_t)new_cap * sizeof(FilterRule*)); + if (!grown) + return false; + list->items = grown; + list->capacity = new_cap; + } + list->items[list->count++] = rule; + return true; +} + +bool filter_rule_list_parse_append(FilterRuleList* list, const char* line, char* err, + size_t err_size) { + FilterRule* rule = filter_rule_parse(line, err, err_size); + if (!rule) + return false; + if (!filter_rule_list_add(list, rule)) { + filter_rule_free(rule); + snprintf(err, err_size, "memory allocation failed"); + return false; + } + return true; +} + +void filter_rule_list_free(FilterRuleList* list) { + if (!list) + return; + for (int i = 0; i < list->count; i++) + filter_rule_free(list->items[i]); + free(list->items); + free(list); +} + +static bool set_rule_owner(FilterRule* rule, const char* owner) { + char* dup = str_dup(owner ? owner : ""); + if (!dup) + return false; + free(rule->owner); + rule->owner = dup; + return true; +} + +/* ---- CVS default excludes (-C) ---- */ + +typedef struct { + const char* pattern; + bool dir_only; +} CvsDefaultRule; + +static const CvsDefaultRule CVS_DEFAULTS[] = { + {"RCS", false}, {"SCCS", false}, {"CVS", false}, {"CVS.adm", false}, + {"RCSLOG", false}, {"cvslog.*", false}, {"tags", false}, {"TAGS", false}, + {".make.state", false}, {".nse_depinfo", false}, {"*~", false}, {"#*", false}, + {".#*", false}, {",*", false}, {"_$*", false}, {"*$", false}, + {"*.old", false}, {"*.bak", false}, {"*.BAK", false}, {"*.orig", false}, + {"*.rej", false}, {".del-*", false}, {"*.a", false}, {"*.olb", false}, + {"*.o", false}, {"*.obj", false}, {"*.so", false}, {"*.exe", false}, + {"*.Z", false}, {"*.elc", false}, {"*.ln", false}, {"core", false}, + {".svn/", true}, {".git/", true}, {".hg/", true}, {".bzr/", true}, +}; + +static bool cvs_rule_list_append(FilterRuleList* list) { + for (size_t i = 0; i < sizeof(CVS_DEFAULTS) / sizeof(CVS_DEFAULTS[0]); i++) { + FilterRule* rule = calloc(1, sizeof(FilterRule)); + if (!rule) + return false; + rule->action = FILTER_ACTION_EXCLUDE; + rule->dir_only = CVS_DEFAULTS[i].dir_only; + size_t plen = strlen(CVS_DEFAULTS[i].pattern); + if (rule->dir_only && plen > 0 && CVS_DEFAULTS[i].pattern[plen - 1] == '/') + plen--; /* keep the cleaned pattern, matching filter_rule_parse */ + rule->pattern = malloc(plen + 1); + if (!rule->pattern) { + free(rule); + return false; + } + memcpy(rule->pattern, CVS_DEFAULTS[i].pattern, plen); + rule->pattern[plen] = '\0'; + if (!set_rule_owner(rule, "")) { + filter_rule_free(rule); + return false; + } + if (!filter_rule_list_add(list, rule)) { + filter_rule_free(rule); + return false; + } + } + return true; +} + +FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count, bool cvs_exclude, + char* err, size_t err_size) { + if (err && err_size > 0) + err[0] = '\0'; + FilterRuleList* list = filter_rule_list_create(); + if (!list) { + snprintf(err, err_size, "memory allocation failed"); + return NULL; + } + for (int i = 0; i < rule_count; i++) { + if (!rule_texts || !rule_texts[i]) + continue; + FilterRule* rule = filter_rule_parse(rule_texts[i], err, err_size); + if (!rule) { + filter_rule_list_free(list); + return NULL; + } + if (!set_rule_owner(rule, "")) { + filter_rule_free(rule); + filter_rule_list_free(list); + snprintf(err, err_size, "memory allocation failed"); + return NULL; + } + if (!filter_rule_list_add(list, rule)) { + filter_rule_free(rule); + filter_rule_list_free(list); + snprintf(err, err_size, "memory allocation failed"); + return NULL; + } + } + if (cvs_exclude && !cvs_rule_list_append(list)) { + filter_rule_list_free(list); + snprintf(err, err_size, "memory allocation failed"); + return NULL; + } + return list; +} + +/* ---- Per-directory .rsync-filter files ---- */ + +FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bool* exists, + char* err, size_t err_size) { + if (err && err_size > 0) + err[0] = '\0'; + if (exists) + *exists = false; + char* filter_path = path_cat(dir_path, ".rsync-filter"); + if (!filter_path) { + snprintf(err, err_size, "memory allocation failed"); + return NULL; + } + FILE* fp = fopen(filter_path, "r"); + free(filter_path); + if (!fp) { + if (errno == ENOENT || errno == ENOTDIR) + return filter_rule_list_create(); + log_message(LOG_LEVEL_WARNING, "Could not read .rsync-filter in %s: %s", dir_path, + strerror(errno)); + return filter_rule_list_create(); + } + if (exists) + *exists = true; + FilterRuleList* list = filter_rule_list_create(); + if (!list) { + fclose(fp); + snprintf(err, err_size, "memory allocation failed"); + return NULL; + } + char* line = NULL; + size_t line_cap = 0; + ssize_t n; + bool ok = true; + while ((n = getline(&line, &line_cap, fp)) != -1) { + const char* p = line; + while (*p == ' ' || *p == '\t') + p++; + if (*p == '\0' || *p == '\n' || *p == '\r' || *p == '#') + continue; + FilterRule* rule = filter_rule_parse(p, err, err_size); + if (!rule) { + ok = false; + break; + } + if (!set_rule_owner(rule, owner_rel)) { + filter_rule_free(rule); + snprintf(err, err_size, "memory allocation failed"); + ok = false; + break; + } + if (!filter_rule_list_add(list, rule)) { + filter_rule_free(rule); + snprintf(err, err_size, "memory allocation failed"); + ok = false; + break; + } + } + free(line); + fclose(fp); + if (!ok) { + filter_rule_list_free(list); + return NULL; + } + return list; +} + +/* ---- Rule matching ---- */ + +/* Match a pattern that contains '/' (non-anchored) against the end of the + * relative path, starting at any path-component boundary. */ +static bool glob_suffix_match(const char* pattern, const char* str) { + if (glob_match(pattern, str)) + return true; + for (const char* slash = strchr(str, '/'); slash; slash = strchr(slash + 1, '/')) { + if (glob_match(pattern, slash + 1)) + return true; + } + return false; +} + +static FilterAction rule_matches(const FilterRule* rule, const char* rel_path, const char* leaf, + bool is_dir) { + if (!rule || !rule->pattern) + return FILTER_ACTION_NONE; + if (rule->dir_only && !is_dir) + return FILTER_ACTION_NONE; + /* A rule applies only to entries below its owner directory. */ + const char* rel2 = rel_path; + if (rule->owner && rule->owner[0] != '\0') { + size_t owner_len = strlen(rule->owner); + if (strncmp(rule->owner, rel_path, owner_len) != 0) + return FILTER_ACTION_NONE; + if (rel_path[owner_len] != '/') + return FILTER_ACTION_NONE; + rel2 = rel_path + owner_len + 1; + } + if (rel2[0] == '\0') + return FILTER_ACTION_NONE; + bool matched; + if (rule->anchored) { + matched = glob_match(rule->pattern, rel2); + } else if (strchr(rule->pattern, '/') != NULL) { + matched = glob_suffix_match(rule->pattern, rel2); + } else { + matched = glob_match(rule->pattern, leaf); + } + return matched ? rule->action : FILTER_ACTION_NONE; +} + +FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf, + bool is_dir) { + if (!list) + return FILTER_ACTION_NONE; + for (int i = 0; i < list->count; i++) { + FilterAction action = rule_matches(list->items[i], rel_path, leaf, is_dir); + if (action != FILTER_ACTION_NONE) + return action; + } + return FILTER_ACTION_NONE; +} diff --git a/src/shared/filter.h b/src/shared/filter.h new file mode 100644 index 0000000..8c43f27 --- /dev/null +++ b/src/shared/filter.h @@ -0,0 +1,83 @@ +#ifndef FILTER_H +#define FILTER_H + +#include +#include + +/* rsync-style filter rule engine (client-side file selection). + * + * Supported rule syntax (documented subset): + * [+|-] [anchored '/' prefix] pattern [trailing '/' for dir-only] + * + * "+ PATTERN" include rule (first match wins) + * "- PATTERN" exclude rule + * "PATTERN" implicit exclude rule (rsync default) + * "include PATTERN" / "exclude PATTERN" word forms + * leading '/' after the +/- anchors the pattern to its owner directory + * (the transfer root for command-line/-C rules, the directory that + * contains a .rsync-filter file for per-directory rules) + * a trailing '/' makes the rule match directories only + * + * Rejected explicitly (no silent no-ops): the rsync merge/dir-merge/list-clear + * shorthands written as a rule that starts with ':' or '.' or '!', the + * merge/dir-merge/hide/show/protect/risk/clear words, and every include/exclude + * rule modifier other than '/' (! C s r p x). The pattern must be separated + * from +/- by a space (or a single '/' anchor), exactly like rsync's + * "-s foo"/"-p ..." modifier syntax is refused. + */ + +typedef enum { + FILTER_ACTION_NONE = 0, /* no rule matched */ + FILTER_ACTION_EXCLUDE = -1, + FILTER_ACTION_INCLUDE = 1 +} FilterAction; + +typedef struct { + FilterAction action; + bool anchored; /* pattern anchored to the rule's owner directory */ + bool dir_only; /* pattern had a trailing '/': matches directories only */ + char* owner; /* owning directory rel path ("" == transfer root) */ + char* pattern; /* cleaned glob pattern (no leading '/', no trailing '/') */ +} FilterRule; + +typedef struct { + FilterRule** items; /* owned array of rule pointers */ + int count; + int capacity; +} FilterRuleList; + +/* Parse a single filter-rule line (no trailing newline required). Returns an + * owned rule, or NULL on unsupported/invalid syntax with a message in `err`. */ +FilterRule* filter_rule_parse(const char* line, char* err, size_t err_size); +void filter_rule_free(FilterRule* rule); + +FilterRuleList* filter_rule_list_create(void); +/* Append a fully-parsed rule (takes ownership). Returns false on OOM. */ +bool filter_rule_list_add(FilterRuleList* list, FilterRule* rule); +/* Parse `line` and append it. Returns false and fills `err` on bad syntax. */ +bool filter_rule_list_parse_append(FilterRuleList* list, const char* line, char* err, + size_t err_size); +void filter_rule_list_free(FilterRuleList* list); + +/* Build the command-line filter set: `rule_texts` (--filter=RULE in the order + * given, 0..rule_count) followed by the -C CVS default excludes when + * cvs_exclude is true. All rules are owned by "" (the transfer root). + * Returns NULL on unsupported rule text (message in `err`). */ +FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count, bool cvs_exclude, + char* err, size_t err_size); + +/* Read "/.rsync-filter" and return its rules, each owned by + * `owner_rel`. A missing file yields an empty list with *exists=false; an + * unreadable file is treated as missing. Returns NULL only on parse or + * allocation failure (message in `err`). */ +FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bool* exists, + char* err, size_t err_size); + +/* Evaluate an entry against one ordered rule list. Returns FILTER_ACTION_NONE + * when no rule matched, otherwise the first matching rule's action. + * `rel_path` is the entry's path relative to the transfer root ("" == root), + * `leaf` its final name, `is_dir` whether it is a directory. */ +FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf, + bool is_dir); + +#endif diff --git a/src/shared/hardlink.c b/src/shared/hardlink.c new file mode 100644 index 0000000..36a1cf5 --- /dev/null +++ b/src/shared/hardlink.c @@ -0,0 +1,127 @@ +#include "hardlink.h" + +#include +#include +#include + +#include "log.h" +#include "utils.h" + +/* ---- Sender-side detection table ---- */ + +HardLinkTable* hardlink_table_create(void) { + HardLinkTable* table = calloc(1, sizeof(HardLinkTable)); + if (!table) + return NULL; + if (mtx_init(&table->mutex, mtx_plain) != thrd_success) { + free(table); + return NULL; + } + table->next_gid = 1; + return table; +} + +static void hardlink_item_destroy(HardLinkItem* item) { + if (!item) + return; + free(item->first_path); + item->first_path = NULL; +} + +void hardlink_table_destroy(HardLinkTable* table) { + if (!table) + return; + for (size_t i = 0; i < table->count; i++) + hardlink_item_destroy(&table->items[i]); + free(table->items); + table->items = NULL; + table->count = 0; + table->capacity = 0; + mtx_destroy(&table->mutex); + free(table); +} + +static HardLinkItem* hardlink_table_find_locked(HardLinkTable* table, dev_t dev, ino_t ino) { + for (size_t i = 0; i < table->count; i++) { + if (table->items[i].dev == dev && table->items[i].ino == ino) + return &table->items[i]; + } + return NULL; +} + +static bool hardlink_table_add_locked(HardLinkTable* table, dev_t dev, ino_t ino, const char* path, + int gid, HardLinkItem** out) { + if (table->count == table->capacity) { + size_t new_capacity = table->capacity == 0 ? 8 : table->capacity * 2; + if (new_capacity < table->capacity) + return false; + HardLinkItem* grown = realloc(table->items, new_capacity * sizeof(HardLinkItem)); + if (!grown) + return false; + table->items = grown; + table->capacity = new_capacity; + } + HardLinkItem* item = &table->items[table->count]; + char* dup = str_dup(path); + if (!dup) + return false; + memset(item, 0, sizeof(*item)); + item->dev = dev; + item->ino = ino; + item->gid = gid; + item->first_path = dup; + table->count++; + *out = item; + return true; +} + +bool hardlink_table_assign(HardLinkTable* table, const char* wire_path, dev_t dev, ino_t ino, + int* gid, bool* is_first, char** first_path_out) { + if (!table || !wire_path || !gid || !is_first || !first_path_out) + return false; + if (mtx_lock(&table->mutex) != thrd_success) + return false; + bool ok = true; + const HardLinkItem* item = hardlink_table_find_locked(table, dev, ino); + int next_gid; + if (item) { + *is_first = false; + char* dup = str_dup(item->first_path); + if (!dup) { + ok = false; + } else { + *gid = item->gid; + *first_path_out = dup; + } + next_gid = -1; + } else { + if (table->next_gid <= 0) { + ok = false; + next_gid = -1; + } else { + next_gid = table->next_gid; + HardLinkItem* created = NULL; + if (!hardlink_table_add_locked(table, dev, ino, wire_path, next_gid, &created)) { + ok = false; + } else { + char* dup = str_dup(wire_path); + if (!dup) { + hardlink_item_destroy(created); + table->count--; + ok = false; + } else { + *is_first = true; + *gid = next_gid; + *first_path_out = dup; + } + } + } + } + if (ok && next_gid > 0) + table->next_gid++; + mtx_unlock(&table->mutex); + if (!ok) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed while detecting hard links"); + } + return ok; +} diff --git a/src/shared/hardlink.h b/src/shared/hardlink.h new file mode 100644 index 0000000..55eb342 --- /dev/null +++ b/src/shared/hardlink.h @@ -0,0 +1,66 @@ +#ifndef HARDLINK_H +#define HARDLINK_H + +#include +#include +#include +#include + +/* + * --hard-links / -H support. + * + * Sender side: a HardLinkTable detects regular files on the source that share + * an (st_dev, st_ino) identity (a `cp -al`-style hard-linked tree) and assigns + * each distinct inode a stable, run-local link-group id. The first member + * encountered carries the file data; every later member is marked as a sibling + * (no data payload) that the receiver creates as a hard link to the first + * member's destination file. Grouping is scoped by st_dev so inode reuse + * across different filesystems is never conflated. The table is mutex-guarded + * so the parallel (multi-threaded) scanner COULD share one instance across its + * worker threads; the first-thread-to-call designates the data-carrying member, + * which is safe because a hard-link group's members are byte-identical. (In + * practice the sender forces the sequential scanner whenever -H is on; the + * mutex guards the shared table for any path that supplies one.) + * + * ORDERING (why there is no receiver-side handshake): the receiver stores every + * file - including a hard-link group's first member - through a SINGLE writer + * thread draining a single FIFO queue driven by a single receive thread, so + * wire order == write order and every sibling is processed AFTER its group's + * first member. The sender additionally forces the sequential scanner with -H + * so the first-member frame always precedes its siblings on the wire. Sibling + * install therefore needs no present/wait registry: it hard-links to the first + * member (or copies it) knowing that path is already installed - or that, if + * the first member was skipped (already up to date), its destination still + * exists. This guarantee is REQUIRED; do not introduce a concurrent + * multi-writer receiver for -H without re-adding an ordering mechanism. + */ + +typedef struct HardLinkItem { + dev_t dev; + ino_t ino; + int gid; + char* first_path; /* wire path of the group's data-carrying first member */ +} HardLinkItem; + +typedef struct HardLinkTable { + mtx_t mutex; + HardLinkItem* items; + size_t count; + size_t capacity; + int next_gid; +} HardLinkTable; + +HardLinkTable* hardlink_table_create(void); +void hardlink_table_destroy(HardLinkTable* table); + +/* Assign a link-group id to the regular file at `wire_path` with (dev, ino). + * On the first encounter the file becomes the group's first (data-carrying) + * member (*is_first = true) and a fresh gid is allocated. On a later member + * *is_first = false and *first_path_out is set to a malloc'd copy of the first + * member's wire path (the caller stores it and owns it; on the first member + * path the returned *first_path_out is a malloc'd copy of its own wire path). + * Returns false on allocation failure (transfer should abort). */ +bool hardlink_table_assign(HardLinkTable* table, const char* wire_path, dev_t dev, ino_t ino, + int* gid, bool* is_first, char** first_path_out); + +#endif diff --git a/src/shared/identity.c b/src/shared/identity.c new file mode 100644 index 0000000..5a39f0e --- /dev/null +++ b/src/shared/identity.c @@ -0,0 +1,748 @@ +#include "identity.h" +#include "log.h" +#include "utils.h" +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/* The active identity snapshot lives in a per-process global. The TCP server + * forks one child process per connection, so a connection never shares this + * with another; within a connection the multithreaded receiver reads it without + * mutation. This is what lets the fd-relative metadata path consult the + * negotiated policy without threading a Config through every write helper. */ +typedef struct { + bool numeric_ids; + bool chown_uid_set; + int32_t chown_uid; + bool chown_gid_set; + int32_t chown_gid; + IdentityMap* usermap; + int usermap_count; + IdentityMap* groupmap; + int groupmap_count; + /* --super / --no-super tri-state (SUPER_MODE_AUTO when unset). Snapshotted + * per connection so privilege_super_permitted() can gate super-user + * activities without a Config argument. */ + int super_mode; + /* --copy-as=USER[:GROUP]: snapshotted so the ownership resolver can force the + * target ids without a Config argument. */ + bool copy_as_set; + int32_t copy_as_uid; + int32_t copy_as_gid; + bool set; +} IdentityActive; + +static IdentityActive g_identity; + +static void identity_active_reset(void) { + free(g_identity.usermap); + free(g_identity.groupmap); + g_identity.usermap = NULL; + g_identity.groupmap = NULL; + g_identity.usermap_count = 0; + g_identity.groupmap_count = 0; + g_identity.numeric_ids = false; + g_identity.chown_uid_set = false; + g_identity.chown_uid = 0; + g_identity.chown_gid_set = false; + g_identity.chown_gid = 0; + g_identity.super_mode = SUPER_MODE_AUTO; + g_identity.copy_as_set = false; + g_identity.copy_as_uid = 0; + g_identity.copy_as_gid = 0; + g_identity.set = false; +} + +void identity_clear_active(void) { + identity_active_reset(); +} + +bool identity_set_active(const Config* config) { + identity_active_reset(); + if (!config) + return true; + g_identity.numeric_ids = config->numeric_ids; + g_identity.chown_uid_set = config->chown_uid_set; + g_identity.chown_uid = config->chown_uid; + g_identity.chown_gid_set = config->chown_gid_set; + g_identity.chown_gid = config->chown_gid; + g_identity.super_mode = config->super_mode; + g_identity.copy_as_set = config->copy_as_set; + g_identity.copy_as_uid = config->copy_as_uid; + g_identity.copy_as_gid = config->copy_as_gid; + if (config->usermap_count > 0) { + g_identity.usermap = calloc((size_t)config->usermap_count, sizeof(IdentityMap)); + if (!g_identity.usermap) + goto alloc_failed; + memcpy(g_identity.usermap, config->usermap, + (size_t)config->usermap_count * sizeof(IdentityMap)); + g_identity.usermap_count = config->usermap_count; + } + if (config->groupmap_count > 0) { + g_identity.groupmap = calloc((size_t)config->groupmap_count, sizeof(IdentityMap)); + if (!g_identity.groupmap) + goto alloc_failed; + memcpy(g_identity.groupmap, config->groupmap, + (size_t)config->groupmap_count * sizeof(IdentityMap)); + g_identity.groupmap_count = config->groupmap_count; + } + g_identity.set = true; + /* A root receiver would honor any client-supplied ownership request (a + --usermap/--groupmap/--chown/--copy-as, or raw ids under --numeric-ids). + Surface that prominently; a privileged daemon applying arbitrary client + ownership is a deliberate, opt-in choice the operator should be aware of. */ + if (geteuid() == 0) + log_message(LOG_LEVEL_WARNING, + "identity mapping active and running as root: client-supplied " + "ownership (usermap/groupmap/chown/numeric-ids) will be honored; " + "run the daemon as an unprivileged user unless intended"); + /* --super explicitly requests super-user activities, but FastSync never + elevates privileges: when the receiver is not already root the kernel will + refuse those confined attempts and each is skipped per entry. Warn exactly + once at activation time (never abort) so the operator knows the flag cannot + succeed on this host. */ + if (g_identity.super_mode == SUPER_MODE_ON && geteuid() != 0) + log_message(LOG_LEVEL_WARNING, + "--super requested but the receiver is not privileged; super-user " + "activities (ownership, device nodes) will be attempted but refused " + "by the kernel and skipped per entry"); + return true; + +alloc_failed: + /* Never proceed with a partial (count-left-zero) map: that would silently + apply the WRONG ownership policy. Fail closed and let the caller refuse + the connection. */ + log_message(LOG_LEVEL_ERROR, "memory allocation failed while activating identity policy"); + identity_active_reset(); + return false; +} + +bool privilege_super_permitted(void) { + return privilege_super_mode_permitted(g_identity.super_mode); +} + +bool privilege_super_mode_permitted(int mode) { + /* AUTO and ON both attempt the confined operation; OFF forbids it even for a + * root receiver. AUTO is the historical FastSync behavior (always attempt + * and let the kernel refuse an unprivileged call, which the caller skips), so + * it must stay permissive or a group-only chown that a non-root receiver is + * allowed to make would regress. */ + return mode != SUPER_MODE_OFF; +} + +bool identity_active_enabled(void) { + /* numeric_ids is included: this set only gates identity_apply_ownership, + which runs only when metadata is present (a -M/--preserve transfer). A + standalone --numeric-ids (no ownership-affecting flag) carries no + metadata, never reaches identity_apply_ownership, and therefore correctly + stays inert; combined with -M it activates raw-id application. --super / + --no-super does NOT enable ownership: it only permits or forbids the + already-requested super-user activities, so a --super with no explicit + identity flag must never silently apply client-chosen ownership. */ + return g_identity.set && + (g_identity.numeric_ids || g_identity.chown_uid_set || g_identity.chown_gid_set || + g_identity.usermap_count > 0 || g_identity.groupmap_count > 0 || g_identity.copy_as_set); +} + +bool identity_ownership_requested(const Config* config) { + if (!config) + return false; + /* Every value that makes the receiver act on a client-chosen owner, plus an + * explicit --super (super-user device-node activities). Pure config, so the + * daemon gate can evaluate it before identity_set_active(). */ + return config->numeric_ids || config->chown_uid_set || config->chown_gid_set || + config->usermap_count > 0 || config->groupmap_count > 0 || config->copy_as_set || + config->fake_super || config->super_mode == SUPER_MODE_ON; +} + +bool identity_copy_as_active(void) { + return g_identity.set && g_identity.copy_as_set; +} + +bool identity_copy_as_refused(const Config* config) { + if (!config || !config->copy_as_set) + return false; + /* The safe-subset --copy-as needs a privileged (root) receiver, and an + * operator/--no-super veto forbids the ownership change even for root. This + * is deliberately a pure function of the config and the current effective uid + * (never the active snapshot) because the server evaluates it at the + * pre-STATUS_OK config gate, before identity_set_active() has run. */ + return geteuid() != 0 || config->super_mode == SUPER_MODE_OFF; +} + +bool identity_wire_valid(const Config* config) { + if (!config) + return false; + if (config->usermap_count < 0 || config->usermap_count > MAX_IDENTITY_MAP || + config->groupmap_count < 0 || config->groupmap_count > MAX_IDENTITY_MAP) + return false; + if (config->chown_uid_set && config->chown_uid < IDENTITY_MATCH_ANY) + return false; + if (config->chown_gid_set && config->chown_gid < IDENTITY_MATCH_ANY) + return false; + for (int i = 0; i < config->usermap_count; i++) { + if (config->usermap[i].from < IDENTITY_MATCH_ANY || config->usermap[i].to < IDENTITY_CURRENT) + return false; + } + for (int i = 0; i < config->groupmap_count; i++) { + if (config->groupmap[i].from < IDENTITY_MATCH_ANY || config->groupmap[i].to < IDENTITY_CURRENT) + return false; + } + /* Defense-in-depth: a --copy-as block must never carry a negative (sentinel) + * id into the ownership path. receive_copy_as_options already rejects them, + * but identity_wire_valid is the shared validation used by both the receiver + * and unit tests, so re-assert it here. */ + if (config->copy_as_set && (config->copy_as_uid < 0 || config->copy_as_gid < 0)) + return false; + return true; +} + +/* ---- CLI-time name/number resolution ---- */ + +/* Parse a single FROM/TO token into an int32 id. Returns 0 on success, -1 on a + * malformed or unresolvable token. When is_group, name lookups use the group + * database; otherwise the user database. A `*` token returns IDENTITY_MATCH_ANY + * / IDENTITY_CURRENT (the same -1 value, disambiguated by the caller's + * position). An `@`-prefixed or bare-decimal token is a numeric id. */ +static int identity_resolve_token(const char* token, bool is_group, int32_t* out) { + if (!token || *token == '\0') + return -1; + if (strcmp(token, "*") == 0) { + *out = IDENTITY_MATCH_ANY; + return 0; + } + const char* num = (token[0] == '@') ? token + 1 : token; + if (*num != '\0') { + bool all_digits = true; + for (const char* p = num; *p; p++) + if (*p < '0' || *p > '9') + all_digits = false; + if (all_digits) { + char* endptr = NULL; + errno = 0; + long val = strtol(num, &endptr, 10); + if (errno == 0 && endptr && *endptr == '\0' && val >= 0 && val <= INT32_MAX) { + *out = (int32_t)val; + return 0; + } + return -1; + } + } + /* A name (or a name-like numeric that failed strict numeric parse). */ + if (is_group) { + struct group* gr = getgrnam(token); + if (!gr) + return -1; + *out = (int32_t)gr->gr_gid; + return 0; + } + struct passwd* pw = getpwnam(token); + if (!pw) + return -1; + *out = (int32_t)pw->pw_uid; + return 0; +} + +static int identity_append_rule(IdentityMap** map, int* count, int32_t from, int32_t to) { + if (*count >= MAX_IDENTITY_MAP) + return -1; + IdentityMap* grown = realloc(*map, (size_t)(*count + 1) * sizeof(IdentityMap)); + if (!grown) + return -1; + *map = grown; + (*map)[*count].from = from; + (*map)[*count].to = to; + (*count)++; + return 0; +} + +int identity_parse_map(Config* config, const char* value, bool is_group) { + if (!config || !value || *value == '\0') { + log_message(LOG_LEVEL_ERROR, "%smap requires a value", is_group ? "--group" : "--user"); + return -1; + } + char* list = str_dup(value); + if (!list) + return -1; + const char* optname = is_group ? "--groupmap" : "--usermap"; + char* saveptr = NULL; + for (char* rule = strtok_r(list, ",", &saveptr); rule; rule = strtok_r(NULL, ",", &saveptr)) { + char* colon = strchr(rule, ':'); + if (!colon || colon == rule) { + /* Log before freeing: `rule` points into the str_dup'd list. */ + log_message(LOG_LEVEL_ERROR, "%s rules must be FROM:TO (got '%s')", optname, rule); + free(list); + return -1; + } + *colon = '\0'; + char* from_token = rule; + char* to_token = colon + 1; + if (*to_token == '\0') { + free(list); + log_message(LOG_LEVEL_ERROR, "%s rule 'FROM:' is missing the TO value (got '%s')", optname, + value); + return -1; + } + int32_t from_id, to_id; + if (identity_resolve_token(from_token, is_group, &from_id) != 0 || + identity_resolve_token(to_token, is_group, &to_id) != 0) { + free(list); + log_message(LOG_LEVEL_ERROR, + "%s could not resolve '%s' (name must exist on the source; use " + "@N for a numeric id)", + optname, value); + return -1; + } + if (identity_append_rule(is_group ? &config->groupmap : &config->usermap, + is_group ? &config->groupmap_count : &config->usermap_count, from_id, + to_id) != 0) { + free(list); + log_message(LOG_LEVEL_ERROR, "%s has too many rules (max %d)", optname, MAX_IDENTITY_MAP); + return -1; + } + } + free(list); + return 0; +} + +/* Split --chown=USER:GROUP on the first UNESCAPED colon, honoring backslash + * escapes (a `\:` is a literal colon inside a name; a lone backslash before any + * other character is kept verbatim). Both sides are returned as malloc'd + * strings (the absent side is NULL). */ +static int identity_split_chown(const char* value, char** puser, char** pgroup) { + size_t len = strlen(value); + char* user = malloc(len + 1); + char* group = malloc(len + 1); + if (!user || !group) { + free(user); + free(group); + return -1; + } + const char* p = value; + size_t ui = 0; + bool split_seen = false; + size_t gi = 0; + while (*p) { + if (*p == '\\' && p[1] == ':') { + /* an escaped colon: a literal ':' in the current side's name */ + if (split_seen) + group[gi++] = ':'; + else + user[ui++] = ':'; + p += 2; + continue; + } + if (*p == ':') { + split_seen = true; + p++; + continue; + } + if (split_seen) + group[gi++] = *p; + else + user[ui++] = *p; + p++; + } + user[ui] = '\0'; + group[gi] = '\0'; + char* u = str_dup(user); + char* g = str_dup(group); + free(user); + free(group); + if (!u || !g) { + free(u); + free(g); + return -1; + } + *puser = u; + *pgroup = g; + return 0; +} + +int identity_parse_chown(Config* config, const char* value) { + if (!config || !value || *value == '\0') { + log_message(LOG_LEVEL_ERROR, "--chown requires a value (USER:GROUP, USER, or :GROUP)"); + return -1; + } + /* Reject more than one UNESCAPED colon (a name or group may not contain an + * unescaped ':' in the spec). The scan is escape-aware: a `\:` is a literal + * colon inside a name, not a field separator. */ + int colons = 0; + bool saw_colon = false; + const char* p = value; + while (*p) { + if (*p == '\\' && p[1] == ':') { + p += 2; + continue; + } + if (*p == ':') { + colons++; + saw_colon = true; + } + p++; + } + if (colons > 1) { + log_message(LOG_LEVEL_ERROR, "--chown must have at most one ':' (got '%s')", value); + return -1; + } + + char *user = NULL, *group = NULL; + if (identity_split_chown(value, &user, &group) != 0) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed for --chown"); + return -1; + } + int ret = 0; + if (!saw_colon) { + /* --chown=USER: owner only. */ + if (*user == '\0') { + log_message(LOG_LEVEL_ERROR, "--chown requires a user or group (got '%s')", value); + ret = -1; + } else if (identity_resolve_token(user, false, &config->chown_uid) != 0) { + log_message(LOG_LEVEL_ERROR, + "--chown could not resolve user '%s' (use a name that exists " + "on the source, '*', or @N)", + value); + ret = -1; + } else { + config->chown_uid_set = true; + } + } else { + /* --chown=USER:GROUP, --chown=:GROUP, --chown=USER: */ + if (*user != '\0') { + if (identity_resolve_token(user, false, &config->chown_uid) != 0) { + log_message(LOG_LEVEL_ERROR, "--chown could not resolve user '%s'", value); + ret = -1; + goto done; + } + config->chown_uid_set = true; + } + if (*group != '\0') { + if (identity_resolve_token(group, true, &config->chown_gid) != 0) { + log_message(LOG_LEVEL_ERROR, "--chown could not resolve group '%s'", value); + ret = -1; + goto done; + } + config->chown_gid_set = true; + } + if (!*user && !*group) { + log_message(LOG_LEVEL_ERROR, "--chown must set a user, a group, or both (got '%s')", value); + ret = -1; + } + } +done: + free(user); + free(group); + return ret; +} + +/* uid_t/gid_t are unsigned and may hold a value wider than the signed int32 the + * wire (and the identity policy) uses. Reject such an id instead of truncating + * it to an out-of-range (possibly negative sentinel) value. */ +static bool identity_id_fits_int32(unsigned long id) { + return id <= (unsigned long)INT32_MAX; +} + +/* Resolve one --copy-as id token. A '*' token means the caller's current + * effective uid (user) or gid (group). Returns 0 on success. On failure sets + * *overflow when a '*' id was wider than int32 so the caller can log the + * specific message; otherwise the token was simply unresolvable. */ +static int identity_resolve_copy_as_id(const char* token, bool is_group, int32_t* out, + bool* overflow) { + *overflow = false; + if (strcmp(token, "*") == 0) { + unsigned long current = is_group ? (unsigned long)getegid() : (unsigned long)geteuid(); + if (!identity_id_fits_int32(current)) { + *overflow = true; + return -1; + } + *out = (int32_t)current; + return 0; + } + return identity_resolve_token(token, is_group, out); +} + +int identity_parse_copy_as(Config* config, const char* value) { + if (!config || !value || *value == '\0') { + log_message(LOG_LEVEL_ERROR, "--copy-as requires USER[:GROUP]"); + return -1; + } + /* --copy-as=USER[:GROUP] is the whole grammar: at most one field separator. + * (Unlike --chown there is no escaped-colon form; a name containing ':' is + * simply not expressible, and the extra colon is a clear parse error.) */ + int colons = 0; + for (const char* p = value; *p; p++) + if (*p == ':') + colons++; + if (colons > 1) { + char* escaped = output_escape(value, false); + log_message(LOG_LEVEL_ERROR, "--copy-as must be USER[:GROUP] (got '%s')", + escaped ? escaped : ""); + free(escaped); + return -1; + } + + char* spec = str_dup(value); + if (!spec) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed for --copy-as"); + return -1; + } + const char* user_token = spec; + const char* group_token = NULL; + char* colon = strchr(spec, ':'); + if (colon) { + *colon = '\0'; + group_token = colon + 1; + } + + /* The spec is untrusted user input echoed back in error paths: escape it once + * (8-bit-safe) so a control byte cannot forge a log line. */ + char* escaped_spec = output_escape(value, false); + const char* shown = escaped_spec ? escaped_spec : ""; + int ret = -1; + + if (*user_token == '\0') { + log_message(LOG_LEVEL_ERROR, "--copy-as is missing the user (got '%s')", shown); + goto done; + } + bool overflow = false; + int32_t uid; + if (identity_resolve_copy_as_id(user_token, false, &uid, &overflow) != 0) { + if (overflow) + log_message(LOG_LEVEL_ERROR, "--copy-as: current user id %lu exceeds INT32_MAX", + (unsigned long)geteuid()); + else + log_message(LOG_LEVEL_ERROR, + "--copy-as could not resolve user (use a name that exists on the " + "source, '*', or @N): %s", + shown); + goto done; + } + + int32_t gid; + if (group_token) { + if (*group_token == '\0') { + log_message(LOG_LEVEL_ERROR, "--copy-as group is empty (got '%s')", shown); + goto done; + } + if (identity_resolve_copy_as_id(group_token, true, &gid, &overflow) != 0) { + if (overflow) + log_message(LOG_LEVEL_ERROR, "--copy-as: current group id %lu exceeds INT32_MAX", + (unsigned long)getegid()); + else + log_message(LOG_LEVEL_ERROR, "--copy-as could not resolve group (got '%s')", shown); + goto done; + } + } else { + /* Group omitted: use the user's primary gid. A numeric id with no local + * passwd entry has no primary gid to look up, so fall back to gid == uid + * (the rsync-style numeric convention; documented divergence). */ + struct passwd* pw = getpwuid((uid_t)uid); + if (pw) { + if (!identity_id_fits_int32((unsigned long)pw->pw_gid)) { + log_message(LOG_LEVEL_ERROR, + "--copy-as: primary group id %lu for the requested user exceeds INT32_MAX", + (unsigned long)pw->pw_gid); + goto done; + } + gid = (int32_t)pw->pw_gid; + } else { + gid = uid; + } + } + /* The group-default and gid==uid fallbacks must never store a negative + * (sentinel) value; the explicit numeric path is already capped by + * identity_resolve_token. */ + if (uid < 0 || gid < 0) { + log_message(LOG_LEVEL_ERROR, "--copy-as resolved id does not fit in int32 (got '%s')", shown); + goto done; + } + + config->copy_as_set = true; + config->copy_as_uid = uid; + config->copy_as_gid = gid; + /* Ownership application needs the metadata path (the source uid/gid must be + * transmitted); imply it exactly like --chown/--usermap/--groupmap. */ + config->use_metadata = true; + ret = 0; + +done: + free(escaped_spec); + free(spec); + return ret; +} + +/* ---- Receiver-side ownership application ---- */ + +static bool identity_map_lookup(const IdentityMap* map, int count, int32_t source_id, + int32_t* out_to) { + for (int i = 0; i < count; i++) { + if (map[i].from == IDENTITY_MATCH_ANY || map[i].from == source_id) { + *out_to = map[i].to; + return true; + } + } + return false; +} + +/* Resolve the target ownership from the negotiated policy against the entry's + * current stat. Shared by the fd (regular file) and no-follow (symlink) apply + * paths. Returns false when no side is to be changed. */ +static bool identity_resolve_targets(const struct stat* st, int32_t source_uid, int32_t source_gid, + uid_t* out_uid, gid_t* out_gid) { + bool set_uid = false; + bool set_gid = false; + uid_t uid = 0; + gid_t gid = 0; + + /* --copy-as (P7 Wave E) has the highest priority: it forces BOTH the owner + * and group of every written entry to the requested ids, beating usermap / + * groupmap / --chown / --numeric-ids and the best-effort name lookup. Only + * skip when the entry already carries exactly those ids. */ + if (g_identity.copy_as_set) { + uid = (uid_t)g_identity.copy_as_uid; + gid = (gid_t)g_identity.copy_as_gid; + if (st->st_uid == uid && st->st_gid == gid) + return false; + *out_uid = uid; + *out_gid = gid; + return true; + } + + int32_t target; + if (identity_map_lookup(g_identity.usermap, g_identity.usermap_count, source_uid, &target)) { + uid = target == IDENTITY_CURRENT ? geteuid() : (uid_t)target; + set_uid = true; + } else if (g_identity.chown_uid_set) { + uid = g_identity.chown_uid == IDENTITY_CURRENT ? geteuid() : (uid_t)g_identity.chown_uid; + set_uid = true; + } else if (g_identity.numeric_ids) { + uid = (uid_t)source_uid; + set_uid = true; + } else { + /* Best-effort name mapping against the receiver's own database: if the + * transmitted (numeric) id resolves to a name present on this machine, + * re-resolve it. On a shared-account host this is the identity operation; + * when the id has no name here, the user side is left alone. */ + struct passwd* pw = getpwuid((uid_t)source_uid); + if (pw) { + const struct passwd* mapped = getpwnam(pw->pw_name); + if (mapped) { + uid = mapped->pw_uid; + set_uid = true; + } + } + } + + if (identity_map_lookup(g_identity.groupmap, g_identity.groupmap_count, source_gid, &target)) { + gid = target == IDENTITY_CURRENT ? getegid() : (gid_t)target; + set_gid = true; + } else if (g_identity.chown_gid_set) { + gid = g_identity.chown_gid == IDENTITY_CURRENT ? getegid() : (gid_t)g_identity.chown_gid; + set_gid = true; + } else if (g_identity.numeric_ids) { + gid = (gid_t)source_gid; + set_gid = true; + } else { + struct group* gr = getgrgid((gid_t)source_gid); + if (gr) { + const struct group* mapped = getgrnam(gr->gr_name); + if (mapped) { + gid = mapped->gr_gid; + set_gid = true; + } + } + } + + if (!set_uid && !set_gid) + return false; + /* An unset side keeps the file's current id so the other side can change. */ + if (!set_uid) + uid = st->st_uid; + if (!set_gid) + gid = st->st_gid; + /* Only change ownership when the target differs (avoid needless syscalls and + * any chance of clearing setuid/setgid on an already-correct entry). */ + if (st->st_uid == uid && st->st_gid == gid) + return false; + *out_uid = uid; + *out_gid = gid; + return true; +} + +static void identity_log_chown_failure(const char* what, uid_t uid, gid_t gid) { + /* EPERM/EACCES are expected when the receiver is not privileged (e.g. the CI + * `nobody` user): warn and continue, never abort the transfer. Any other + * error (EIO/EROFS/ENOSPC/...) is a real failure and must not be silently + * downgraded to a warning. + * + * --copy-as is different: the whole point of the flag is that the target + * ownership is REQUIRED (the pre-flight gate already refused an unprivileged + * receiver). If the chown still fails with EPERM/EACCES (a capability- + * restricted root, root-squash, or a read-only mount) the run would be + * silently producing the WRONG ownership, so surface it at ERROR. The + * caller (identity_apply_ownership*) then reports the ENTRY as failed rather + * than as written, which becomes a FILE_SAVE_ERROR and fails the transfer + * (fail-fast) instead of reporting overall success with the wrong owner. */ + if (errno == EPERM || errno == EACCES) { + if (identity_copy_as_active()) + log_message(LOG_LEVEL_ERROR, + "could not apply --copy-as ownership on %s (uid=%ld gid=%ld): %s; " + "entry was written with the wrong owner", + what, (long)uid, (long)gid, strerror(errno)); + else + log_message(LOG_LEVEL_WARNING, + "could not apply ownership (uid=%ld gid=%ld): %s; leaving as-is", (long)uid, + (long)gid, strerror(errno)); + } else { + log_message(LOG_LEVEL_ERROR, "failed to apply ownership on %s (uid=%ld gid=%ld): %s", what, + (long)uid, (long)gid, strerror(errno)); + } +} + +bool identity_apply_ownership(int fd, int32_t source_uid, int32_t source_gid) { + /* Ownership application is OFF unless the client requested an identity flag. + * This is the controlled gate: a default (or plain -M) transfer never changes + * ownership, byte-for-byte preserving FastSync's existing behavior. --no-super + * additionally forbids it even when the receiver is root. */ + if (!identity_active_enabled() || !privilege_super_permitted() || fd < 0) + return true; + struct stat st; + if (fstat(fd, &st) != 0) + return !identity_copy_as_active(); + uid_t uid; + gid_t gid; + if (!identity_resolve_targets(&st, source_uid, source_gid, &uid, &gid)) + return true; + if (fchown(fd, uid, gid) != 0) { + identity_log_chown_failure("file", uid, gid); + /* A required --copy-as ownership that did not land is a per-entry failure; + * every other policy stays best-effort (rsync parity). */ + return !identity_copy_as_active(); + } + return true; +} + +bool identity_apply_ownership_link(int parent_fd, const char* leaf, int32_t source_uid, + int32_t source_gid) { + if (!identity_active_enabled() || !privilege_super_permitted() || parent_fd < 0 || !leaf) + return true; + struct stat st; + if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) != 0) + return !identity_copy_as_active(); + uid_t uid; + gid_t gid; + if (!identity_resolve_targets(&st, source_uid, source_gid, &uid, &gid)) + return true; + if (fchownat(parent_fd, leaf, uid, gid, AT_SYMLINK_NOFOLLOW) != 0) { + identity_log_chown_failure("no-follow entry", uid, gid); + return !identity_copy_as_active(); + } + return true; +} diff --git a/src/shared/identity.h b/src/shared/identity.h new file mode 100644 index 0000000..592c5e7 --- /dev/null +++ b/src/shared/identity.h @@ -0,0 +1,133 @@ +#ifndef IDENTITY_H +#define IDENTITY_H + +#include "config.h" +#include +#include +#include + +/* + * Identity mapping: --numeric-ids / --usermap / --groupmap / --chown / --copy-as. + * + * FastSync transmits uid/gid numerically (int32 on the wire) and, by design, + * NEVER applies client-supplied ownership unless a user explicitly opts in with + * an identity flag below. This module is the controlled, opt-in, + * privilege-gated path for applying ownership on the receiver: the wire config + * is snapshotted once per connection via identity_set_active() and applied + * through an fd-relative fchown() in the receiver's metadata-restore path. + * + * Because only numeric ids cross the wire, name-based values are resolved to + * numbers at CLI parse time using the CLIENT (sender) machine's databases. On + * a shared-account source/destination this reproduces rsync's semantics; a + * genuinely different destination database is a documented divergence (see + * RSYNC_COMPAT.md). + */ + +/* Parse one --usermap= / --groupmap= value (comma-separated FROM:TO rules, + * first match wins) into config->usermap / config->groupmap. is_group selects + * the group tables and name databases. Returns 0 on success, -1 on a + * malformed spec or an unresolvable name (never a silent no-op). */ +int identity_parse_map(Config* config, const char* value, bool is_group); + +/* Parse --chown=USER:GROUP. Supports USER:GROUP, USER (owner only), :GROUP + * (group only), '*' (current/root as appropriate) and numeric ids. Returns 0 + * on success, -1 on a malformed spec / unresolvable name. */ +int identity_parse_chown(Config* config, const char* value); + +/* Parse --copy-as=USER[:GROUP] (P7 Wave E). USER is resolved with the same + * user-database rules as --chown (a name, @N/bare N numeric id, or '*' meaning + * the client's current euid); when ':GROUP' is present the group is resolved + * with the group database ('*' meaning the client's egid). When the group is + * omitted, the user's primary gid is used (getpwuid(uid)->pw_gid); if the + * resolved user is a numeric id with no local passwd entry, gid falls back to + * uid. On success sets copy_as_set/copy_as_uid/copy_as_gid and forces + * metadata transmission (ownership application needs the metadata path). + * Returns 0 on success, -1 on a malformed / empty / unresolvable spec (never a + * silent no-op). */ +int identity_parse_copy_as(Config* config, const char* value); + +/* True when a --copy-as request is active but the receiver is not permitted to + * perform the privileged ownership application it needs. This is the up-front + * refusal predicate: the server rejects the whole transfer at the config + * handshake rather than silently ignoring the requested ownership. It is a + * pure function of the config mode and the current effective uid (it does NOT + * read the active snapshot, so it is valid at the pre-STATUS_OK gate, before + * identity_set_active() has run). `super_mode` is the EFFECTIVE mode after any + * server-side policy veto. */ +bool identity_copy_as_refused(const Config* config); + +/* True when the CURRENT per-connection snapshot has a --copy-as active (i.e. + * identity_set_active() has run against a config with copy_as_set). The + * --fake-super owner replay consults this so a copy-as run never lets the + * recorded source owner overwrite the forced target owner. Reads the active + * snapshot, so call identity_set_active() first (the receiver does, before any + * write). */ +bool identity_copy_as_active(void); + +/* Receiver-side snapshot of the negotiated identity config. The server calls + * identity_set_active() once per connection (before any file write) using the + * config received over the wire; the snapshot is a deep copy so the caller may + * free its Config immediately. identity_clear_active() releases it. + * + * Returns true on success. On an allocation failure while deep-copying a + * requested usermap/groupmap it logs a LOG_LEVEL_ERROR, leaves the snapshot + * cleared (never a partial/wrong policy) and returns false; the caller must + * refuse the connection. */ +bool identity_set_active(const Config* config); +void identity_clear_active(void); + +/* True when any ownership-affecting identity option is present in the active + * snapshot. Ownership stays OFF ("do not apply") for every transfer that + * requests none of them, preserving FastSync's existing behavior. --super / + * --no-super alone does NOT enable ownership; an explicit identity flag + * (--numeric-ids / --chown / --usermap / --groupmap / --copy-as) is required. */ +bool identity_active_enabled(void); + +/* Pure, config-only predicate: true when the client requested ANY + * client-chosen ownership or super-user activity (--numeric-ids, --chown, + * --usermap/--groupmap, --copy-as, --fake-super, or an explicit --super). Used + * by the daemon module gate to decide whether a module's per-module opt-in is + * required; it never reads the per-connection snapshot. */ +bool identity_ownership_requested(const Config* config); + +/* Apply the negotiated ownership to an already-written file descriptor. + * source_uid/source_gid are the transmitted numeric ids. Resolution order: + * --copy-as (highest priority, forces both ids), then a matching + * usermap/groupmap rule, then --chown, then --numeric-ids (raw), then a + * best-effort name lookup on the receiver's own databases (skipped when the + * transmitted id has no name on this system). Only calls fchown() when the + * result differs from the current value. + * + * Returns false ONLY when an active --copy-as ownership application failed: its + * forced ownership is REQUIRED, so the caller must treat the entry as failed + * rather than reporting success with the wrong owner. For every other identity + * policy an fchown EPERM/EACCES is logged and ignored and true is returned + * (rsync parity: the transfer must not abort). A no-op when no identity policy + * is active returns true. */ +bool identity_apply_ownership(int fd, int32_t source_uid, int32_t source_gid); + +/* P7 Wave D: the no-follow (symlink) counterpart. Resolves the same + * usermap/groupmap/chown/numeric-ids/copy-as policy but applies it with + * fchownat(..., AT_SYMLINK_NOFOLLOW) so a symlink's own ownership is changed + * without ever dereferencing it. A no-op unless an identity flag is active. + * The return value follows identity_apply_ownership(): false only when an + * active --copy-as application failed. */ +bool identity_apply_ownership_link(int parent_fd, const char* leaf, int32_t source_uid, + int32_t source_gid); + +/* Receiver-side wire validation of the resolved identity fields. */ +bool identity_wire_valid(const Config* config); + +/* P7 Wave E receiver-side permission gate for super-user activities (ownership + * application and char/block device-node creation). `privilege_super_permitted` + * consults the per-connection snapshot (call identity_set_active() first); + * `privilege_super_mode_permitted` is the pure mode predicate and is what + * callers holding a Config use (the config-frame gate, file_receive). Both + * return false only for SUPER_MODE_OFF; SUPER_MODE_ON and SUPER_MODE_AUTO (the + * default) permit a confined attempt, matching FastSync's historical + * best-effort behavior where an unprivileged attempt is refused by the kernel + * and skipped. Neither EVER elevates privileges. */ +bool privilege_super_permitted(void); +bool privilege_super_mode_permitted(int mode); + +#endif \ No newline at end of file diff --git a/src/shared/log.c b/src/shared/log.c index 07a5d27..d013bbe 100644 --- a/src/shared/log.c +++ b/src/shared/log.c @@ -1,42 +1,144 @@ #include "log.h" +#include +#include #include #include +#include #include static const char* log_level_strings[] = {"DEBUG", "INFO", "WARN", "ERROR"}; static LogLevel current_log_level = LOG_LEVEL_WARNING; +static uint32_t current_debug_flags = 0; +static uint32_t info_flags = 0; +static bool info_flags_explicit = false; static FILE* log_fp = NULL; +static _Thread_local bool eight_bit_output; +static LogStderrMode stderr_mode = LOG_STDERR_ERRORS; void set_log_level(LogLevel level) { current_log_level = level; } +void set_log_debug_flags(uint32_t flags) { + current_debug_flags = flags; +} + +uint32_t get_log_debug_flags(void) { + return current_debug_flags; +} + +bool log_debug_enabled(LogDebugFlag flag) { + return current_log_level <= LOG_LEVEL_DEBUG && (current_debug_flags & flag) != 0; +} + +void set_log_info_flags(uint32_t flags) { + info_flags = flags; + info_flags_explicit = true; +} + +uint32_t get_log_info_flags(void) { + return info_flags; +} + void log_set_file(FILE* fp) { log_fp = fp; } +void log_set_8_bit_output(bool enabled) { + eight_bit_output = enabled; +} + +bool log_get_8_bit_output(void) { + return eight_bit_output; +} + +void log_set_stderr_mode(LogStderrMode mode) { + stderr_mode = mode; +} + +LogStderrMode log_get_stderr_mode(void) { + return stderr_mode; +} + +static inline void write_message(FILE* dest_io, LogLevel log_level, struct tm t, const char* format, + va_list args) { + fprintf(dest_io, "%04d-%02d-%02d %02d:%02d:%02d [%s]: ", t.tm_year + 1900, t.tm_mon + 1, + t.tm_mday, t.tm_hour, t.tm_min, t.tm_sec, log_level_strings[log_level]); + + vfprintf(dest_io, format, args); + fprintf(dest_io, "\n"); +} + void log_message(LogLevel log_level, const char* format, ...) { if (log_level < current_log_level) return; + if (log_level < 0 || log_level >= (int)(sizeof(log_level_strings) / sizeof(log_level_strings[0]))) + return; time_t now = time(NULL); - const struct tm* t = localtime(&now); + struct tm t; + if (!localtime_r(&now, &t)) + return; - fprintf(stderr, "%04d-%02d-%02d %02d:%02d:%02d [%s]: ", t->tm_year + 1900, t->tm_mon + 1, - t->tm_mday, t->tm_hour, t->tm_min, t->tm_sec, log_level_strings[log_level]); + FILE* dest_io = stdout; + if (stderr_mode == LOG_STDERR_ALL || log_level == LOG_LEVEL_ERROR) { + dest_io = stderr; + } va_list args; va_start(args, format); - vfprintf(stderr, format, args); + write_message(dest_io, log_level, t, format, args); va_end(args); - fprintf(stderr, "\n"); if (log_fp) { - fprintf(log_fp, "%04d-%02d-%02d %02d:%02d:%02d [%s]: ", t->tm_year + 1900, t->tm_mon + 1, - t->tm_mday, t->tm_hour, t->tm_min, t->tm_sec, log_level_strings[log_level]); va_start(args, format); - vfprintf(log_fp, format, args); + write_message(log_fp, log_level, t, format, args); va_end(args); - fprintf(log_fp, "\n"); - fflush(log_fp); } } + +void log_debug_message(LogDebugFlag flag, const char* format, ...) { + if (current_log_level > LOG_LEVEL_DEBUG || !(current_debug_flags & flag)) + return; + + time_t now = time(NULL); + struct tm t; + if (!localtime_r(&now, &t)) + return; + + va_list args; + va_start(args, format); + write_message(stdout, LOG_LEVEL_DEBUG, t, format, args); + va_end(args); + + if (log_fp) { + va_start(args, format); + write_message(log_fp, LOG_LEVEL_DEBUG, t, format, args); + va_end(args); + } +} + +void log_info_message(LogInfoFlag flag, const char* format, ...) { + if ((info_flags_explicit && (info_flags & flag) == 0) || + (!info_flags_explicit && current_log_level > LOG_LEVEL_DEBUG)) + return; + + time_t now = time(NULL); + struct tm t; + if (!localtime_r(&now, &t)) + return; + + va_list args; + va_start(args, format); + write_message(stdout, LOG_LEVEL_INFO, t, format, args); + va_end(args); + + if (log_fp) { + va_start(args, format); + write_message(log_fp, LOG_LEVEL_INFO, t, format, args); + va_end(args); + } +} + +void log_perror(const char* context) { + log_message(LOG_LEVEL_ERROR, "%s: %s", context, strerror(errno)); +} diff --git a/src/shared/log.h b/src/shared/log.h index 0aa622b..0e5a2ff 100644 --- a/src/shared/log.h +++ b/src/shared/log.h @@ -2,11 +2,45 @@ #define LOG_H #include +#include +#include typedef enum { LOG_LEVEL_DEBUG, LOG_LEVEL_INFO, LOG_LEVEL_WARNING, LOG_LEVEL_ERROR } LogLevel; +typedef enum { LOG_STDERR_ERRORS, LOG_STDERR_ALL } LogStderrMode; + +typedef enum { + LOG_DEBUG_IO = 1u << 0, + LOG_DEBUG_PROTO = 1u << 1, + LOG_DEBUG_PACK = 1u << 2, + LOG_DEBUG_UTIL = 1u << 3, + LOG_DEBUG_ALL = (1u << 4) - 1, +} LogDebugFlag; + +typedef enum { + LOG_INFO_COPY = 1u << 0, + LOG_INFO_MISC = 1u << 1, + LOG_INFO_SKIP = 1u << 2, + LOG_INFO_STATS = 1u << 3, + LOG_INFO_ALL = LOG_INFO_COPY | LOG_INFO_MISC | LOG_INFO_SKIP | LOG_INFO_STATS, +} LogInfoFlag; void log_message(LogLevel log_level, const char* message, ...); +void log_perror(const char* context); void set_log_level(LogLevel level); +void set_log_debug_flags(uint32_t flags); +uint32_t get_log_debug_flags(void); +/* True when a log_debug_message() call with the same flag would actually emit: + * the debug log level is enabled AND the flag is selected. Hot paths use this + * to skip expensive message formatting/escaping when the line is filtered. */ +bool log_debug_enabled(LogDebugFlag flag); +void log_debug_message(LogDebugFlag flag, const char* message, ...); +void set_log_info_flags(uint32_t flags); +uint32_t get_log_info_flags(void); +void log_info_message(LogInfoFlag flag, const char* message, ...); void log_set_file(FILE* fp); +void log_set_8_bit_output(bool enabled); +bool log_get_8_bit_output(void); +void log_set_stderr_mode(LogStderrMode mode); +LogStderrMode log_get_stderr_mode(void); #endif diff --git a/src/shared/metadata.c b/src/shared/metadata.c index 4134c13..b1cff1e 100644 --- a/src/shared/metadata.c +++ b/src/shared/metadata.c @@ -1,7 +1,9 @@ #include "metadata.h" #include "file.h" +#include "identity.h" #include "log.h" #include "protocol.h" +#include "utils.h" #include #include #include @@ -24,6 +26,29 @@ typedef char static_assert_mode_t_fits[(sizeof(mode_t) <= sizeof(int32_t)) ? 1 : typedef char static_assert_uid_t_fits[(sizeof(uid_t) <= sizeof(int32_t)) ? 1 : -1]; typedef char static_assert_gid_t_fits[(sizeof(gid_t) <= sizeof(int32_t)) ? 1 : -1]; +bool metadata_mtime_matches(time_t left_sec, long left_nsec, time_t right_sec, long right_nsec, + int modify_window) { + int64_t left = (int64_t)left_sec; + int64_t right = (int64_t)right_sec; + int64_t seconds; + int64_t nanoseconds; + + if (left > right || (left == right && left_nsec >= right_nsec)) { + seconds = left - right; + nanoseconds = (int64_t)left_nsec - (int64_t)right_nsec; + } else { + seconds = right - left; + nanoseconds = (int64_t)right_nsec - (int64_t)left_nsec; + } + if (nanoseconds < 0) { + seconds--; + nanoseconds += 1000000000LL; + } + if (modify_window == 0) + return left == right; + return seconds < modify_window || (seconds == modify_window && nanoseconds == 0); +} + void metadata_to_buf(char** buf, const FileMetadata* m) { int32_t present = (m != NULL) ? 1 : 0; memcpy(*buf, &present, sizeof(present)); @@ -45,15 +70,35 @@ void metadata_to_buf(char** buf, const FileMetadata* m) { int64_t mtime_nsec = (int64_t)m->mtime_nsec; memcpy(*buf, &mtime_nsec, sizeof(mtime_nsec)); *buf += sizeof(mtime_nsec); + int32_t atime_valid = m->atime_valid ? 1 : 0; + memcpy(*buf, &atime_valid, sizeof(atime_valid)); + *buf += sizeof(atime_valid); + int64_t atime_sec = (int64_t)m->atime_sec; + memcpy(*buf, &atime_sec, sizeof(atime_sec)); + *buf += sizeof(atime_sec); + int64_t atime_nsec = (int64_t)m->atime_nsec; + memcpy(*buf, &atime_nsec, sizeof(atime_nsec)); + *buf += sizeof(atime_nsec); + int32_t crtime_valid = m->crtime_valid ? 1 : 0; + memcpy(*buf, &crtime_valid, sizeof(crtime_valid)); + *buf += sizeof(crtime_valid); + int64_t crtime_sec = (int64_t)m->crtime_sec; + memcpy(*buf, &crtime_sec, sizeof(crtime_sec)); + *buf += sizeof(crtime_sec); + int64_t crtime_nsec = (int64_t)m->crtime_nsec; + memcpy(*buf, &crtime_nsec, sizeof(crtime_nsec)); + *buf += sizeof(crtime_nsec); } FileMetadata* metadata_from_buf(char** buf) { int32_t present; memcpy(&present, *buf, sizeof(present)); *buf += sizeof(present); + if (present != 0 && present != 1) + return NULL; if (!present) return NULL; - FileMetadata* m = malloc(sizeof(FileMetadata)); + FileMetadata* m = protocol_alloc(sizeof(FileMetadata)); if (m == NULL) return NULL; int32_t mode; @@ -76,10 +121,41 @@ FileMetadata* metadata_from_buf(char** buf) { memcpy(&mtime_nsec, *buf, sizeof(mtime_nsec)); *buf += sizeof(mtime_nsec); m->mtime_nsec = (long)mtime_nsec; + int32_t atime_valid; + memcpy(&atime_valid, *buf, sizeof(atime_valid)); + *buf += sizeof(atime_valid); + int64_t atime_sec; + memcpy(&atime_sec, *buf, sizeof(atime_sec)); + *buf += sizeof(atime_sec); + int64_t atime_nsec; + memcpy(&atime_nsec, *buf, sizeof(atime_nsec)); + *buf += sizeof(atime_nsec); + int32_t crtime_valid; + memcpy(&crtime_valid, *buf, sizeof(crtime_valid)); + *buf += sizeof(crtime_valid); + int64_t crtime_sec; + memcpy(&crtime_sec, *buf, sizeof(crtime_sec)); + *buf += sizeof(crtime_sec); + int64_t crtime_nsec; + memcpy(&crtime_nsec, *buf, sizeof(crtime_nsec)); + *buf += sizeof(crtime_nsec); + m->atime_valid = atime_valid != 0; + m->atime_sec = (time_t)atime_sec; + m->atime_nsec = (long)atime_nsec; + m->crtime_valid = crtime_valid != 0; + m->crtime_sec = (time_t)crtime_sec; + m->crtime_nsec = (long)crtime_nsec; + if (present != 1 || mtime_nsec < 0 || mtime_nsec >= 1000000000LL || mode < 0 || uid < 0 || + gid < 0 || atime_valid < 0 || atime_valid > 1 || crtime_valid < 0 || crtime_valid > 1 || + (atime_valid && (atime_nsec < 0 || atime_nsec >= 1000000000LL)) || + (crtime_valid && (crtime_nsec < 0 || crtime_nsec >= 1000000000LL))) { + free(m); + return NULL; + } return m; } -bool metadata_send(int file_descriptor, FileMetadata* m) { +bool metadata_send(int file_descriptor, const FileMetadata* m) { if (m == NULL) { int32_t zero = 0; return send_n_data(file_descriptor, &zero, sizeof(zero)); @@ -90,12 +166,24 @@ bool metadata_send(int file_descriptor, FileMetadata* m) { int32_t gid = (int32_t)m->gid; int64_t mtime_sec = (int64_t)m->mtime_sec; int64_t mtime_nsec = (int64_t)m->mtime_nsec; + int32_t atime_valid = m->atime_valid ? 1 : 0; + int64_t atime_sec = (int64_t)m->atime_sec; + int64_t atime_nsec = (int64_t)m->atime_nsec; + int32_t crtime_valid = m->crtime_valid ? 1 : 0; + int64_t crtime_sec = (int64_t)m->crtime_sec; + int64_t crtime_nsec = (int64_t)m->crtime_nsec; return send_n_data(file_descriptor, &present, sizeof(present)) && send_n_data(file_descriptor, &mode, sizeof(mode)) && send_n_data(file_descriptor, &uid, sizeof(uid)) && send_n_data(file_descriptor, &gid, sizeof(gid)) && send_n_data(file_descriptor, &mtime_sec, sizeof(mtime_sec)) && - send_n_data(file_descriptor, &mtime_nsec, sizeof(mtime_nsec)); + send_n_data(file_descriptor, &mtime_nsec, sizeof(mtime_nsec)) && + send_n_data(file_descriptor, &atime_valid, sizeof(atime_valid)) && + send_n_data(file_descriptor, &atime_sec, sizeof(atime_sec)) && + send_n_data(file_descriptor, &atime_nsec, sizeof(atime_nsec)) && + send_n_data(file_descriptor, &crtime_valid, sizeof(crtime_valid)) && + send_n_data(file_descriptor, &crtime_sec, sizeof(crtime_sec)) && + send_n_data(file_descriptor, &crtime_nsec, sizeof(crtime_nsec)); } FileMetadata* metadata_receive(int file_descriptor, int* ok) { @@ -105,12 +193,17 @@ FileMetadata* metadata_receive(int file_descriptor, int* ok) { *ok = 0; return NULL; } - if (!present) { + if (present == 0) { if (ok) *ok = 1; return NULL; } - FileMetadata* m = malloc(sizeof(FileMetadata)); + if (present != 1) { + if (ok) + *ok = 0; + return NULL; + } + FileMetadata* m = protocol_alloc(sizeof(FileMetadata)); if (m == NULL) { if (ok) *ok = 0; @@ -156,23 +249,195 @@ FileMetadata* metadata_receive(int file_descriptor, int* ok) { return NULL; } m->mtime_nsec = (long)mtime_nsec; + int32_t atime_valid; + if (!receive_n_data(file_descriptor, &atime_valid, sizeof(atime_valid))) { + free(m); + if (ok) + *ok = 0; + return NULL; + } + int64_t atime_sec; + if (!receive_n_data(file_descriptor, &atime_sec, sizeof(atime_sec))) { + free(m); + if (ok) + *ok = 0; + return NULL; + } + int64_t atime_nsec; + if (!receive_n_data(file_descriptor, &atime_nsec, sizeof(atime_nsec))) { + free(m); + if (ok) + *ok = 0; + return NULL; + } + int32_t crtime_valid; + if (!receive_n_data(file_descriptor, &crtime_valid, sizeof(crtime_valid))) { + free(m); + if (ok) + *ok = 0; + return NULL; + } + int64_t crtime_sec; + if (!receive_n_data(file_descriptor, &crtime_sec, sizeof(crtime_sec))) { + free(m); + if (ok) + *ok = 0; + return NULL; + } + int64_t crtime_nsec; + if (!receive_n_data(file_descriptor, &crtime_nsec, sizeof(crtime_nsec))) { + free(m); + if (ok) + *ok = 0; + return NULL; + } + m->atime_valid = atime_valid != 0; + m->atime_sec = (time_t)atime_sec; + m->atime_nsec = (long)atime_nsec; + m->crtime_valid = crtime_valid != 0; + m->crtime_sec = (time_t)crtime_sec; + m->crtime_nsec = (long)crtime_nsec; + if (mtime_nsec < 0 || mtime_nsec >= 1000000000LL || mode < 0 || uid < 0 || gid < 0 || + atime_valid < 0 || atime_valid > 1 || crtime_valid < 0 || crtime_valid > 1 || + (atime_valid && (atime_nsec < 0 || atime_nsec >= 1000000000LL)) || + (crtime_valid && (crtime_nsec < 0 || crtime_nsec >= 1000000000LL))) { + free(m); + if (ok) + *ok = 0; + return NULL; + } if (ok) *ok = 1; return m; } -void file_restore_metadata(const char* path, FileMetadata* metadata) { +static mode_t metadata_mode(const FileMetadata* metadata, mode_t current_mode, + bool preserve_executability) { + const mode_t execute_bits = S_IXUSR | S_IXGRP | S_IXOTH; + if (preserve_executability) + return (current_mode & 0777 & ~execute_bits) | (metadata->mode & execute_bits); + return metadata->mode & 0777 & ~(S_IWGRP | S_IWOTH); +} + +void file_restore_metadata(const char* path, const FileMetadata* metadata, + bool preserve_executability) { if (metadata == NULL) return; - if (chmod(path, metadata->mode & 07777 & ~(S_ISUID | S_ISGID)) != 0) - log_message(LOG_LEVEL_WARNING, "Failed to chmod %s: %s", path, strerror(errno)); - if (chown(path, metadata->uid, metadata->gid) != 0) - log_message(LOG_LEVEL_WARNING, "Failed to chown %s: %s", path, strerror(errno)); + struct stat current; + mode_t current_mode = stat(path, ¤t) == 0 ? current.st_mode : 0; + mode_t safe_mode = metadata_mode(metadata, current_mode, preserve_executability); + if (chmod(path, safe_mode) != 0) { + char* escaped_path = output_escape(path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "Failed to chmod %s: %s", + escaped_path ? escaped_path : "", strerror(errno)); + free(escaped_path); + } + /* Never apply client-supplied ownership. The descriptor API below is the + receiver write path; retain this legacy API only for compatibility. */ struct timespec times[2]; times[0].tv_sec = 0; times[0].tv_nsec = UTIME_OMIT; times[1].tv_sec = metadata->mtime_sec; times[1].tv_nsec = metadata->mtime_nsec; - if (utimensat(AT_FDCWD, path, times, 0) != 0) - log_message(LOG_LEVEL_WARNING, "Failed to set timestamps on %s: %s", path, strerror(errno)); + if (metadata->atime_valid) { + times[0].tv_sec = metadata->atime_sec; + times[0].tv_nsec = metadata->atime_nsec; + } + if (metadata->crtime_valid) { + log_message(LOG_LEVEL_DEBUG, + "crtime (birth time) %lld.%09ld transmitted for %s but not applied: no portable " + "setter exists", + (long long)metadata->crtime_sec, metadata->crtime_nsec, path); + } + if (utimensat(AT_FDCWD, path, times, 0) != 0) { + char* escaped_path = output_escape(path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "Failed to set timestamps on %s: %s", + escaped_path ? escaped_path : "", strerror(errno)); + free(escaped_path); + } +} + +bool file_restore_symlink_metadata(const char* path, const FileMetadata* metadata, + bool omit_link_times) { + if (path == NULL || metadata == NULL) + return !identity_copy_as_active(); + char* leaf = NULL; + int parent_fd = file_open_secure_parent(path, &leaf, false); + if (parent_fd < 0) + return !identity_copy_as_active(); + /* Ownership (only when the identity policy is active) via lchown semantics: + fchownat with AT_SYMLINK_NOFOLLOW never dereferences the link. A failed + REQUIRED --copy-as ownership marks the entry failed; every other policy is + best-effort. */ + bool owned = identity_apply_ownership_link(parent_fd, leaf, (int32_t)metadata->uid, + (int32_t)metadata->gid); + /* Symlink mode: not settable on Linux (fchmodat AT_SYMLINK_NOFOLLOW returns + EOPNOTSUPP/ENOTSUP); attempt it for platforms that support it and quietly + ignore the unsupported case so the transfer never fails over it. */ + mode_t link_mode = metadata->mode & 0777; + if (fchmodat(parent_fd, leaf, link_mode, AT_SYMLINK_NOFOLLOW) != 0 && errno != EOPNOTSUPP && + errno != ENOTSUP && errno != ENOSYS) { + log_message(LOG_LEVEL_DEBUG, "Could not set symlink mode on %s: %s", path, strerror(errno)); + } + if (!omit_link_times) { + struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT}, + {.tv_sec = metadata->mtime_sec, .tv_nsec = metadata->mtime_nsec}}; + if (metadata->atime_valid) { + times[0].tv_sec = metadata->atime_sec; + times[0].tv_nsec = metadata->atime_nsec; + } + if (utimensat(parent_fd, leaf, times, AT_SYMLINK_NOFOLLOW) != 0) { + char* escaped_path = output_escape(path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "Failed to set symlink timestamps on %s: %s", + escaped_path ? escaped_path : "", strerror(errno)); + free(escaped_path); + } + } + close(parent_fd); + free(leaf); + return owned; +} + +bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, bool preserve_executability) { + if (fd < 0 || metadata == NULL) + return metadata == NULL; + bool ok = true; + struct stat current; + if (fstat(fd, ¤t) != 0) + return false; + mode_t safe_mode = metadata_mode(metadata, current.st_mode, preserve_executability); + if (fchmod(fd, safe_mode) != 0) + ok = false; + /* Client uid/gid values are deliberately not authoritative UNLESS the client + explicitly opted in with an identity flag (--numeric-ids / --usermap / + --groupmap / --chown). identity_apply_ownership is the controlled, + privilege-gated path: it consults the negotiated policy, resolves the + target ids, and applies them via an fd-relative fchown() that is confined + to the just-written file (EPERM/EACCES are logged, never fatal) -- EXCEPT + for an active --copy-as, whose forced ownership is REQUIRED: a failure + marks this entry as failed instead of reporting a wrong-owner write as + success. With no identity flag set it is a no-op, so a default or plain -M + transfer keeps FastSync's existing behavior of never applying client + ownership. */ + if (!identity_apply_ownership(fd, (int32_t)metadata->uid, (int32_t)metadata->gid)) + ok = false; + struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT}, + {.tv_sec = metadata->mtime_sec, .tv_nsec = metadata->mtime_nsec}}; + if (metadata->atime_valid) { + times[0].tv_sec = metadata->atime_sec; + times[0].tv_nsec = metadata->atime_nsec; + } + /* --crtimes captures and transmits the source birth time, but there is no + * portable way to set a birth time (utimensat can only set atime/mtime), so + * the receiver deliberately does NOT apply it. This is explicit, honest + * unsupported-attribute handling: log a debug note and continue — never fail + * the transfer and never pretend the crtime was applied. */ + if (metadata->crtime_valid) { + log_message(LOG_LEVEL_DEBUG, + "crtime (birth time) %lld.%09ld transmitted but not applied: no portable setter", + (long long)metadata->crtime_sec, metadata->crtime_nsec); + } + if (futimens(fd, times) != 0) + ok = false; + return ok; } diff --git a/src/shared/metadata.h b/src/shared/metadata.h index 28312f5..faf5194 100644 --- a/src/shared/metadata.h +++ b/src/shared/metadata.h @@ -5,6 +5,7 @@ #include #include #include +#include /* * Wire format (introduced in protocol version 2.0.0): @@ -14,6 +15,12 @@ * int32_t gid (was gid_t, platform-dependent) * int64_t mtime_sec (was time_t, platform-dependent) * int64_t mtime_nsec (was long, platform-dependent) + * int32_t atime_valid (-U/--atimes; protocol 2.12.0) + * int64_t atime_sec + * int64_t atime_nsec + * int32_t crtime_valid (-N/--crtimes; protocol 2.12.0) + * int64_t crtime_sec + * int64_t crtime_nsec * * Prior to 2.0.0 the wire format used the raw platform-dependent types, * which broke compatiblity across different systems. All fields are now @@ -22,13 +29,29 @@ /* Size of metadata fields on wire, excluding the int32_t `present` field that * is always sent first. The total wire size for present metadata is - * sizeof(int32_t) + FILE_METADATA_WIRE_SIZE (32 bytes on most platforms). */ -#define FILE_METADATA_WIRE_SIZE (sizeof(int32_t) * 3 + sizeof(int64_t) * 2) + * sizeof(int32_t) + FILE_METADATA_WIRE_SIZE (68 bytes on most platforms). */ +#define FILE_METADATA_WIRE_SIZE (sizeof(int32_t) * 5 + sizeof(int64_t) * 6) void metadata_to_buf(char** buf, const FileMetadata* m); FileMetadata* metadata_from_buf(char** buf); -bool metadata_send(int file_descriptor, FileMetadata* m); +bool metadata_send(int file_descriptor, const FileMetadata* m); FileMetadata* metadata_receive(int file_descriptor, int* ok); -void file_restore_metadata(const char* path, FileMetadata* metadata); +void file_restore_metadata(const char* path, const FileMetadata* metadata, + bool preserve_executability); +bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, bool preserve_executability); +/* P7 Wave D: apply a SYMLINK's own metadata using no-follow primitives only + * (utimensat/lchown/fchmodat with AT_SYMLINK_NOFOLLOW), confined fd-relative + * under the authorized root. `omit_link_times` (-J/--omit-link-times) + * suppresses the timestamps; the link's mode/ownership are still attempted + * (ownership stays gated by the identity policy and by default is not applied). + * A null metadata or an unfollowable parent is a harmless no-op. Returns false + * only when a REQUIRED --copy-as ownership application failed, so the caller can + * report the entry as failed instead of claiming a wrong-owner success. */ +bool file_restore_symlink_metadata(const char* path, const FileMetadata* metadata, + bool omit_link_times); + +/* Compare timestamps using rsync's whole-second modification window. */ +bool metadata_mtime_matches(time_t left_sec, long left_nsec, time_t right_sec, long right_nsec, + int modify_window); #endif diff --git a/src/shared/motd.c b/src/shared/motd.c new file mode 100644 index 0000000..9fcd1ea --- /dev/null +++ b/src/shared/motd.c @@ -0,0 +1,76 @@ +#include "motd.h" +#include "protocol.h" +#include +#include +#include +#include + +char* motd_read_file(const char* path) { + if (!path || path[0] == '\0') + return NULL; + FILE* fp = fopen(path, "rb"); + if (!fp) + return NULL; + char* buffer = malloc(MOTD_MAX_BYTES + 1); + if (!buffer) { + fclose(fp); + return NULL; + } + /* fread stops at the bound; a larger file is truncated rather than read + * unbounded. ferror distinguishes a truncated read from an I/O failure. */ + size_t total = fread(buffer, 1, MOTD_MAX_BYTES, fp); + if (ferror(fp)) { + free(buffer); + fclose(fp); + return NULL; + } + fclose(fp); + buffer[total] = '\0'; + return buffer; +} + +char* motd_render(const char* motd, bool eight_bit_output) { + if (!motd) + return NULL; + size_t length = strlen(motd); + if (length > (SIZE_MAX - 1) / 5) + return NULL; + char* rendered = malloc(length * 5 + 1); + if (!rendered) + return NULL; + size_t out = 0; + for (size_t i = 0; i < length; i++) { + unsigned char byte = (unsigned char)motd[i]; + if (byte == '\n' || byte == '\t') { + rendered[out++] = (char)byte; + } else if ((byte >= 32 && byte <= 126) || (eight_bit_output && byte >= 128)) { + rendered[out++] = (char)byte; + } else { + rendered[out++] = '\\'; + rendered[out++] = '#'; + rendered[out++] = (char)('0' + ((byte >> 6) & 7)); + rendered[out++] = (char)('0' + ((byte >> 3) & 7)); + rendered[out++] = (char)('0' + (byte & 7)); + } + } + rendered[out] = '\0'; + return rendered; +} + +bool motd_send(int file_descriptor, const char* motd) { + return send_str(file_descriptor, motd ? motd : ""); +} + +char* motd_receive(int file_descriptor) { + char* motd = receive_str(file_descriptor); + if (!motd) + return NULL; + /* Guard against a hostile/oversized peer: receive_str already bounded the + * frame at MAX_STRING_SIZE and consumed it, so discarding an over-bound + * body here keeps the stream framed while refusing to display it. */ + if (strlen(motd) > MOTD_MAX_BYTES) { + free(motd); + return NULL; + } + return motd; +} diff --git a/src/shared/motd.h b/src/shared/motd.h new file mode 100644 index 0000000..49cb3f9 --- /dev/null +++ b/src/shared/motd.h @@ -0,0 +1,54 @@ +#ifndef MOTD_H +#define MOTD_H + +#include + +/* Daemon Message-Of-The-Day (Wave C). + * + * The daemon listener (fastsync-server --daemon) may advertise a `motd file` + * configured in its globals. When a client connects with a host::module/path + * destination and the module gate accepts the connection, the server sends the + * MOTD as a single string frame BEFORE any transfer data (rsync sends its MOTD + * as the first thing from the server at the start of a daemon connection). + * The client reads that frame right after the config/status handshake and + * displays it on stdout unless --no-motd was given. + * + * The MOTD is ordinary display text, never a secret, so it uses the normal + * (non-redacted) string primitive. The exchange is strictly server->client + * and happens on the daemon listener path only; the --stdio SSH path has no + * MOTD. + * + * No PROTOCOL_VERSION bump is involved: the frame is sent and read + * symmetrically by every 2.15.0 daemon build (the strict same-version + * handshake rejects any other version before the frame), so it cannot + * desynchronize a peer. */ + +/* Upper bound on the MOTD bytes the server will read from disk and put on the + * wire. Kept far below MAX_STRING_SIZE (64 KB) so a huge/hostile motd file + * can never produce an unbounded frame or allocation. */ +#define MOTD_MAX_BYTES 4096 + +/* Read a daemon MOTD file, bounded to MOTD_MAX_BYTES. Returns a malloc'd + * NUL-terminated copy of the file content (bytes beyond the bound are + * truncated) or NULL when path is NULL/empty, the file cannot be opened or + * read, or allocation fails. An absent or unreadable motd file is NOT an + * error: the caller simply sends an empty MOTD frame and continues. */ +char* motd_read_file(const char* path); + +/* Render MOTD text for terminal display. Newlines and tabs are preserved so + * a multi-line motd still reads naturally, while every other non-printable / + * control byte (ESC included) is escaped with FastSync's `\NNN` octal + * convention, so a hostile server cannot inject terminal escape sequences + * through the MOTD. eight_bit_output keeps bytes >= 0x80 verbatim (matching + * --8-bit-output). Returns a malloc'd string or NULL on allocation failure. */ +char* motd_render(const char* motd, bool eight_bit_output); + +/* Send/receive the MOTD string frame. These wrap the normal string + * primitive: the MOTD is not a credential, so no redaction is used. The + * receiver additionally rejects an over-bound frame (> MOTD_MAX_BYTES) as a + * hostile input guard; the frame itself is always fully consumed first, so the + * stream stays framed. */ +bool motd_send(int file_descriptor, const char* motd); +char* motd_receive(int file_descriptor); + +#endif diff --git a/src/shared/multiprocessing.c b/src/shared/multiprocessing.c index 910330f..57af92a 100644 --- a/src/shared/multiprocessing.c +++ b/src/shared/multiprocessing.c @@ -1,4 +1,6 @@ #include "multiprocessing.h" +#include "receiver.h" + #include "array_list.h" #include "chunk.h" #include "config.h" @@ -24,23 +26,90 @@ PipelineContextSender* pipeline_context_sender_create(Config* config, Queue* que context->scanner_done = false; context->loader_done = false; context->manifest = NULL; - if (mtx_init(&context->mutex_scanner, mtx_plain) != thrd_success || - cnd_init(&context->condition_not_full_scanner) != thrd_success || - cnd_init(&context->condition_not_empty_scanner) != thrd_success || - mtx_init(&context->mutex_loader, mtx_plain) != thrd_success || - cnd_init(&context->condition_not_full_loader) != thrd_success || - cnd_init(&context->condition_not_empty_loader) != thrd_success) { - perror("Error initializing synchronization objects"); - free(context); - return NULL; + context->excluded_paths = NULL; + context->missing_args = NULL; + context->scan_had_io_error = false; + context->remove_source_files = NULL; + context->early_delete = false; + context->scan_stopped_early = false; + context->total_files = 0; + context->progress_bytes = 0; + context->total_bytes = 0; + context->sender_done = false; + atomic_init(&context->cancelled, false); + protocol_session_init(&context->allocation_session, -1, -1); + protocol_session_set_max_alloc(&context->allocation_session, config->max_alloc); + context->dir_entries = NULL; + context->dir_entries_mutex_init = false; + int init = 0; + if (config->use_metadata) { + context->dir_entries = array_list_create(file_destroy); + if (!context->dir_entries) + goto fail; } + if (mtx_init(&context->mutex_scanner, mtx_plain) != thrd_success) + goto fail; + init++; + if (cnd_init(&context->condition_not_full_scanner) != thrd_success) + goto fail; + init++; + if (cnd_init(&context->condition_not_empty_scanner) != thrd_success) + goto fail; + init++; + if (mtx_init(&context->mutex_loader, mtx_plain) != thrd_success) + goto fail; + init++; + if (cnd_init(&context->condition_not_full_loader) != thrd_success) + goto fail; + init++; + if (cnd_init(&context->condition_not_empty_loader) != thrd_success) + goto fail; + init++; + if (mtx_init(&context->mutex_progress, mtx_plain) != thrd_success) + goto fail; + // cppcheck-suppress unreadVariable + init++; + if (mtx_init(&context->dir_entries_mutex, mtx_plain) != thrd_success) + goto fail; + context->dir_entries_mutex_init = true; return context; + +fail: + log_perror("Error initializing synchronization objects"); + if (context->dir_entries_mutex_init) + mtx_destroy(&context->dir_entries_mutex); + if (context->dir_entries) + array_list_delete(context->dir_entries); + if (init >= 6) + cnd_destroy(&context->condition_not_empty_loader); + if (init >= 5) + cnd_destroy(&context->condition_not_full_loader); + if (init >= 4) + mtx_destroy(&context->mutex_loader); + if (init >= 3) + cnd_destroy(&context->condition_not_empty_scanner); + if (init >= 2) + cnd_destroy(&context->condition_not_full_scanner); + if (init >= 1) + mtx_destroy(&context->mutex_scanner); + free(context); + return NULL; } void pipeline_context_sender_destroy(PipelineContextSender* context) { if (context->manifest) { array_list_delete(context->manifest); } + if (context->excluded_paths) + array_list_delete(context->excluded_paths); + if (context->missing_args) + array_list_delete(context->missing_args); + if (context->remove_source_files) + array_list_delete(context->remove_source_files); + if (context->dir_entries) + array_list_delete(context->dir_entries); + if (context->dir_entries_mutex_init) + mtx_destroy(&context->dir_entries_mutex); config_delete(context->config); queue_destroy(context->queue_scanner); queue_destroy(context->queue_loader); @@ -50,6 +119,7 @@ void pipeline_context_sender_destroy(PipelineContextSender* context) { mtx_destroy(&context->mutex_loader); cnd_destroy(&context->condition_not_full_loader); cnd_destroy(&context->condition_not_empty_loader); + mtx_destroy(&context->mutex_progress); free(context); } @@ -62,136 +132,173 @@ PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* context->queue = queue; context->file_descriptor = file_descriptor; context->ssl = ssl; + context->outcomes.entries = NULL; + context->outcomes.count = 0; + context->outcomes.capacity = 0; + dir_time_list_init(&context->dir_times); + protocol_session_init(&context->session, file_descriptor, file_descriptor); + protocol_session_set_ssl(&context->session, ssl); context->receiver_done = false; - if (mtx_init(&context->mutex, mtx_plain) != thrd_success || - cnd_init(&context->condition_not_full) != thrd_success || - cnd_init(&context->condition_not_empty) != thrd_success) { - perror("Error initializing synchronization objects"); - free(context); - return NULL; - } + context->queued_bytes = 0; + context->max_queue_bytes = 0; + context->deferred_manifest = NULL; + atomic_init(&context->cancelled, false); + int init = 0; + if (mtx_init(&context->mutex, mtx_plain) != thrd_success) + goto fail; + init++; + if (cnd_init(&context->condition_not_full) != thrd_success) + goto fail; + init++; + if (cnd_init(&context->condition_not_empty) != thrd_success) + goto fail; + // cppcheck-suppress unreadVariable + init++; return context; + +fail: + log_perror("Error initializing synchronization objects"); + if (init >= 3) + cnd_destroy(&context->condition_not_empty); + if (init >= 2) + cnd_destroy(&context->condition_not_full); + if (init >= 1) + mtx_destroy(&context->mutex); + free(context); + return NULL; } void pipeline_context_receiver_destroy(PipelineContextReceiver* context) { config_delete(context->config); + if (context->deferred_manifest) + delete_manifest_free(context->deferred_manifest); queue_destroy(context->queue); + receiver_outcomes_destroy(&context->outcomes); + dir_time_list_free(&context->dir_times); mtx_destroy(&context->mutex); cnd_destroy(&context->condition_not_full); cnd_destroy(&context->condition_not_empty); free(context); } -static bool receive_chunk_enqueue(int file_descriptor, PipelineContextReceiver* context) { - Chunk* chunk = receive_chunk_data(file_descriptor, context->config); - if (chunk == NULL) - return false; +void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context, + size_t max_bytes) { + if (context == NULL) + return; + mtx_lock(&context->mutex); + context->max_queue_bytes = max_bytes; + context->queued_bytes = 0; + cnd_broadcast(&context->condition_not_full); + mtx_unlock(&context->mutex); +} - for (int i = 0; i < chunk->element_count; i++) { - File* file = chunk->items[i]; - chunk->items[i] = NULL; - queue_enqueue_multithreaded(context->queue, file, &context->mutex, - &context->condition_not_empty, &context->condition_not_full); +void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context, + size_t released_bytes) { + if (context == NULL || context->max_queue_bytes == 0 || released_bytes == 0) + return; + mtx_lock(&context->mutex); + if (released_bytes >= context->queued_bytes) + context->queued_bytes = 0; + else + context->queued_bytes -= released_bytes; + cnd_signal(&context->condition_not_full); + mtx_unlock(&context->mutex); +} + +bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file) { + if (context == NULL || file == NULL) + return false; + size_t file_bytes = file->data ? file->data->size : 0; + mtx_lock(&context->mutex); + while (!atomic_load(&context->cancelled)) { + bool blocked_by_count = queue_is_full(context->queue); + bool blocked_by_budget = false; + if (context->max_queue_bytes > 0) { + size_t budget = context->max_queue_bytes; + size_t used = context->queued_bytes; + if (used >= budget) { + blocked_by_budget = true; + } else if (file_bytes > budget - used) { + /* A single payload larger than the whole budget (not possible with + the per-file receive cap) is only admitted to an empty pipeline so + the wait can never deadlock. */ + blocked_by_budget = used != 0; + } + } + if (!blocked_by_count && !blocked_by_budget) + break; + cnd_wait(&context->condition_not_full, &context->mutex); } - chunk_destroy(chunk); + if (atomic_load(&context->cancelled)) { + mtx_unlock(&context->mutex); + file_destroy(file); + return false; + } + if (!queue_enqueue(context->queue, file)) { + mtx_unlock(&context->mutex); + file_destroy(file); + return false; + } + context->queued_bytes += file_bytes; + cnd_signal(&context->condition_not_empty); + mtx_unlock(&context->mutex); return true; } +static bool receiver_enqueue_file(File* file, void* context_pointer) { + PipelineContextReceiver* context = (PipelineContextReceiver*)context_pointer; + return pipeline_context_receiver_enqueue_file(context, file); +} + +static void receiver_thread_fail(PipelineContextReceiver* context) { + mtx_lock(&context->mutex); + atomic_store(&context->cancelled, true); + context->receiver_done = true; + cnd_broadcast(&context->condition_not_empty); + cnd_broadcast(&context->condition_not_full); + mtx_unlock(&context->mutex); +} + int receive_thread(void* pipeline_context) { PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context; - if (context->ssl) - io_set_ssl(context->ssl); + protocol_session_bind(&context->session); mtx_lock(&context->mutex); int file_descriptor = context->file_descriptor; const Config* config = context->config; mtx_unlock(&context->mutex); - Status status; - if (!receive_status(file_descriptor, &status)) + ReceiverSink sink = {receiver_enqueue_file, context, false, false, NULL}; + if (receiver_process_pending((Config*)config, file_descriptor, &sink, + &context->deferred_manifest) != 0) { + receiver_thread_fail(context); + protocol_session_unbind(); return thrd_error; - while (status == STATUS_NEXT || status == STATUS_CHUNK || status == STATUS_CHECK || - status == STATUS_KEEPALIVE || status == STATUS_ABORT || status == STATUS_CHECK_BATCH) { - if (status == STATUS_KEEPALIVE) { - send_status(file_descriptor, STATUS_KEEPALIVE); - goto next; - } - if (status == STATUS_ABORT) { - log_message(LOG_LEVEL_INFO, "Received abort from client, cleaning up"); - return thrd_error; - } - if (status == STATUS_CHECK) { - bool skipped; - File* file = receive_incremental_check(file_descriptor, config, &skipped); - if (!skipped) { - if (file == NULL) - return thrd_error; - queue_enqueue_multithreaded(context->queue, file, &context->mutex, - &context->condition_not_empty, &context->condition_not_full); - } - } else if (status == STATUS_CHUNK) { - if (!receive_chunk_enqueue(file_descriptor, context)) - return thrd_error; - } else if (status == STATUS_CHECK_BATCH) { - int count; - if (!receive_int(file_descriptor, &count)) - return thrd_error; - for (int i = 0; i < count; i++) { - char* check_path = receive_str(file_descriptor); - if (!check_path) - return thrd_error; - unsigned long long check_size; - long long check_mtime; - if (!receive_n_data(file_descriptor, &check_size, sizeof(check_size)) || - !receive_n_data(file_descriptor, &check_mtime, sizeof(check_mtime))) { - free(check_path); - return thrd_error; - } - char* full_path = path_cat(config->receive_root_directory, check_path); - struct stat st; - bool has_old = full_path && lstat(full_path, &st) == 0; - bool match = has_old && (unsigned long long)st.st_size == check_size && - (long long)st.st_mtime == check_mtime; - if (match) - send_status(file_descriptor, STATUS_OK); - else - send_status(file_descriptor, STATUS_NEXT); - free(full_path); - free(check_path); - } - goto next; - } else { - File* file = file_receive(config, file_descriptor); - if (file) { - queue_enqueue_multithreaded(context->queue, file, &context->mutex, - &context->condition_not_empty, &context->condition_not_full); - } else { - log_message(LOG_LEVEL_ERROR, "Failed to receive file"); - return thrd_error; - } - } - next: - if (!receive_status(file_descriptor, &status)) - return thrd_error; - } - if (status == STATUS_MANIFEST) { - if (receive_manifest(file_descriptor, config, &status) != 0) - return thrd_error; } mtx_lock(&context->mutex); context->receiver_done = true; cnd_signal(&context->condition_not_empty); mtx_unlock(&context->mutex); + protocol_session_unbind(); return thrd_success; } int write_thread(void* pipeline_context) { PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context; - if (context->ssl) - io_set_ssl(context->ssl); + protocol_session_bind(&context->session); mtx_lock(&context->mutex); bool save_to_disk = context->config->save_to_disk; char* root_directory = str_dup(context->config->receive_root_directory); mtx_unlock(&context->mutex); + if (save_to_disk && !root_directory) { + mtx_lock(&context->mutex); + atomic_store(&context->cancelled, true); + context->receiver_done = true; + cnd_broadcast(&context->condition_not_full); + cnd_broadcast(&context->condition_not_empty); + mtx_unlock(&context->mutex); + protocol_session_unbind(); + return thrd_error; + } while (true) { File* file = @@ -199,10 +306,64 @@ int write_thread(void* pipeline_context) { &context->condition_not_full, &context->receiver_done); if (file == NULL) { free(root_directory); + protocol_session_unbind(); return thrd_success; } - if (save_to_disk) - file_save_to_disk(root_directory, file, context->config); + size_t file_bytes = file->data ? file->data->size : 0; + FileSaveResult result = FILE_SAVE_SKIPPED; + if (save_to_disk) { + result = file_save_to_disk_full(root_directory, file, context->config); + if (result == FILE_SAVE_ERROR) { + file_destroy(file); + pipeline_context_receiver_note_bytes_released(context, file_bytes); + mtx_lock(&context->mutex); + atomic_store(&context->cancelled, true); + context->receiver_done = true; + cnd_broadcast(&context->condition_not_full); + cnd_broadcast(&context->condition_not_empty); + mtx_unlock(&context->mutex); + free(root_directory); + protocol_session_unbind(); + return thrd_error; + } + } + /* P7 Wave D: a directory's times are never applied inline (a later child + write would clobber them); accumulate the metadata here and let the + caller apply it once every writer has drained. */ + if (result != FILE_SAVE_ERROR && file->is_dir && file->metadata && + context->config->use_metadata && !context->config->omit_dir_times && + !dir_time_list_add(&context->dir_times, file->path, file->metadata)) { + file_destroy(file); + pipeline_context_receiver_note_bytes_released(context, file_bytes); + mtx_lock(&context->mutex); + atomic_store(&context->cancelled, true); + context->receiver_done = true; + cnd_broadcast(&context->condition_not_full); + cnd_broadcast(&context->condition_not_empty); + mtx_unlock(&context->mutex); + free(root_directory); + protocol_session_unbind(); + return thrd_error; + } + /* Record the per-file outcome so a --remove-source-files sender learns + which sources were actually written versus skipped on the receiver. + Explicit directory entries and recreated device/special nodes have no + source and are never acknowledged (mirrors receiver.c). */ + if (context->config->remove_source_files && !file->is_dir && !file->is_special && !file->skip && + !receiver_outcomes_append(&context->outcomes, (unsigned char)result)) { + file_destroy(file); + pipeline_context_receiver_note_bytes_released(context, file_bytes); + mtx_lock(&context->mutex); + atomic_store(&context->cancelled, true); + context->receiver_done = true; + cnd_broadcast(&context->condition_not_full); + cnd_broadcast(&context->condition_not_empty); + mtx_unlock(&context->mutex); + free(root_directory); + protocol_session_unbind(); + return thrd_error; + } file_destroy(file); + pipeline_context_receiver_note_bytes_released(context, file_bytes); } } diff --git a/src/shared/multiprocessing.h b/src/shared/multiprocessing.h index 5d0a39e..5cbf236 100644 --- a/src/shared/multiprocessing.h +++ b/src/shared/multiprocessing.h @@ -2,12 +2,15 @@ #define MULTIPROCESSING_H #include +#include #include "array_list.h" #include "config.h" #include "file.h" #include "protocol.h" #include "queue.h" +#include "receiver.h" +#include "stop_condition.h" #include typedef struct { @@ -23,6 +26,54 @@ typedef struct { cnd_t condition_not_empty_loader; bool loader_done; ArrayList* manifest; + /* Protected prefixes (paths the source scan excluded by user rules) sent + with the keep-set manifest so --delete leaves them alone unless + --delete-excluded is set. NULL when not collecting. Populated by the + scanner thread (parallel workers append under mutex_scanner via the + scanner's exclusion sink) or, in the early modes, by the path-only pre-scan + on the calling thread before the pipeline starts. */ + ArrayList* excluded_paths; + /* --delete-missing-args: the destination-relative mirrors of the --files-from + entries that are missing under the source. Computed by the preflight on + the calling thread before the pipeline starts; the sender thread transmits + them in the manifest frame's third section and the receiver deletes each as + an explicit request. */ + ArrayList* missing_args; + /* A source I/O error (unreadable directory) was recorded during the scan. + Set by the pre-scan (before the threads start) or by the scanner thread + under mutex_scanner; the caller turns it into a non-zero exit when + --ignore-errors kept the run going. */ + bool scan_had_io_error; + ArrayList* remove_source_files; + /* True when --delete-before/--delete-during require the keep-set manifest to + be transmitted before any file data: context->manifest is then prebuilt by + a path-only pre-scan on the calling thread and the pipeline scanner must + not append to it. Set once before the worker threads start. */ + bool early_delete; + mtx_t mutex_progress; + int total_files; + unsigned long long progress_bytes; + unsigned long long total_bytes; + bool sender_done; + atomic_bool cancelled; + ProtocolSession allocation_session; + /* Phase 6: client-only sender stop deadline, computed once before the worker + * threads start and shared read-only by the scanner and the sender thread. */ + StopCondition stop_condition; + /* Phase 6: set when the scanner/sender reached the stop deadline before the + * scan (and thus the keep-set manifest) completed naturally. When true the + * completion tail must NOT transmit the partial manifest, or the receiver + * would delete unscanned source mirrors. Written by the sender thread + * before it reads the manifest, so no additional synchronization is needed + * to suppress the manifest. */ + bool scan_stopped_early; + /* P7 Wave D: captured source directory times, filled by the scanner thread + * (and its parallel workers, guarded by dir_entries_mutex) and drained by the + * sender thread in trailing STATUS_DIR_TIMES frame(s). Owned by the + * context; NULL for non-metadata transfers. */ + ArrayList* dir_entries; + mtx_t dir_entries_mutex; + bool dir_entries_mutex_init; } PipelineContextSender; typedef struct PipelineContextReceiver { @@ -30,10 +81,33 @@ typedef struct PipelineContextReceiver { Config* config; int file_descriptor; SSL* ssl; + ProtocolSession session; + ReceiverOutcomes outcomes; mtx_t mutex; cnd_t condition_not_full; cnd_t condition_not_empty; bool receiver_done; + atomic_bool cancelled; + /* Aggregate payload bytes that have been received but not yet released by + the disk writer (queued or in the writer's hand). Guarded by `mutex`. + When `max_queue_bytes` is non-zero the receiver blocks before enqueuing + once this total would exceed it, so decompressed/copied file payloads + buffered ahead of a slow disk writer respect the per-connection memory + budget instead of growing without bound. */ + size_t queued_bytes; + size_t max_queue_bytes; + /* Keep-set manifest for the commit-style (late) deletion + (--delete/--delete-after/--delete-delay). receive_thread parses the whole + protocol stream but hands the manifest here instead of deleting while the + disk writer may still be draining; the caller (server.c) commits the + deletion after both threads have joined, so no extra is removed unless the + transfer truly succeeded. NULL in the early delete modes (which delete at + the manifest). */ + DeleteManifest* deferred_manifest; + /* P7 Wave D: directory metadata collected by write_thread from received + directory entries. Only write_thread mutates it (before it joins); the + caller (server.c) applies it after the delete/delay-updates phase. */ + DirTimeList dir_times; } PipelineContextReceiver; PipelineContextSender* pipeline_context_sender_create(Config* config, Queue* queue_scanner, @@ -42,6 +116,18 @@ void pipeline_context_sender_destroy(PipelineContextSender* context); PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue_receiver, int file_descriptor, SSL* ssl); void pipeline_context_receiver_destroy(PipelineContextReceiver* context); +/* Bound the bytes buffered ahead of the disk writer (see max_queue_bytes). */ +void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context, + size_t max_bytes); +/* Blocking enqueue used by the receive pipeline sink. Blocks while the queue + is full by element count or when adding `file` would push queued_bytes over + the configured byte limit; waits until the disk writer releases bytes. + Takes ownership of `file` on success and destroys it on failure/cancel. */ +bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file); +/* Account for `released_bytes` of payload memory that has been freed by the + disk writer, unblocking a receiver that is waiting on the byte limit. */ +void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context, + size_t released_bytes); int receive_thread(void* pipeline_context); int write_thread(void* pipeline_context); #endif diff --git a/src/shared/protocol.c b/src/shared/protocol.c index 0b32f96..3f13667 100644 --- a/src/shared/protocol.c +++ b/src/shared/protocol.c @@ -1,69 +1,211 @@ #include "protocol.h" #include "log.h" +#include "utils.h" #include +#include #include #include #include #include #include +#include #include #include -#define MAX_DATA_SIZE (100ULL * 1024 * 1024) /* 100 MB max per data message */ -#define RECEIVE_TIMEOUT_SEC 60 /* 60 second per-message timeout */ -#define MAX_CONNECTION_MEMORY (1024ULL * 1024 * 1024) /* 1 GB total per connection */ +#define RECEIVE_TIMEOUT_SEC 60 /* 60 second per-message timeout */ +#define SEND_TIMEOUT_SEC 60 static __thread int io_read_fd = -1; static __thread int io_write_fd = -1; static __thread SSL* io_ssl; +static __thread ProtocolSession* bound_session; +static __thread ProtocolSession legacy_io_session = { + .read_fd = -1, .write_fd = -1, .max_alloc = DEFAULT_MAX_ALLOC}; static unsigned long long io_bwlimit = 0; -static long long bw_tokens = 0; -static struct timespec bw_last_refill = {0, 0}; +static mtx_t bw_mutex; +static once_flag bw_mutex_once = ONCE_FLAG_INIT; -static __thread unsigned long long total_allocated_bytes = 0; +static unsigned long long global_bwlimit(void); +static bool protocol_reserve_memory(ProtocolSession* session, size_t charge) { + unsigned long long allocated = atomic_load(&session->total_allocated_bytes); + while (true) { + if (allocated > MAX_CONNECTION_MEMORY || + (unsigned long long)charge > MAX_CONNECTION_MEMORY - allocated) + return false; + if (atomic_compare_exchange_weak(&session->total_allocated_bytes, &allocated, + allocated + (unsigned long long)charge)) + return true; + } +} + +static void protocol_release_memory_for_session(ProtocolSession* session, size_t charge) { + unsigned long long allocated = atomic_load(&session->total_allocated_bytes); + while (true) { + unsigned long long remaining = (unsigned long long)charge >= allocated ? 0 : allocated - charge; + if (atomic_compare_exchange_weak(&session->total_allocated_bytes, &allocated, remaining)) + break; + } +} + +void protocol_release_memory(size_t charge) { + ProtocolSession* session = bound_session ? bound_session : &legacy_io_session; + protocol_release_memory_for_session(session, charge); +} void io_set_fds(int read_fd, int write_fd) { + bound_session = NULL; io_read_fd = read_fd; io_write_fd = write_fd; + /* A descriptor switch starts a new transport; never reuse a TLS object + belonging to a previous connection or test pipe. */ + io_ssl = NULL; + legacy_io_session.read_fd = read_fd; + legacy_io_session.write_fd = write_fd; + legacy_io_session.ssl = NULL; + legacy_io_session.eight_bit_output = false; + atomic_store(&legacy_io_session.total_allocated_bytes, 0); + legacy_io_session.max_alloc = DEFAULT_MAX_ALLOC; + protocol_session_set_bwlimit(&legacy_io_session, global_bwlimit()); +} + +void protocol_session_init(ProtocolSession* session, int read_fd, int write_fd) { + if (!session) + return; + memset(session, 0, sizeof(*session)); + session->read_fd = read_fd; + session->write_fd = write_fd; + session->max_alloc = DEFAULT_MAX_ALLOC; + atomic_init(&session->total_allocated_bytes, 0); + protocol_session_set_bwlimit(session, global_bwlimit()); +} + +void protocol_session_set_max_alloc(ProtocolSession* session, unsigned long long max_alloc) { + if (!session) + session = bound_session ? bound_session : &legacy_io_session; + session->max_alloc = max_alloc; +} + +static bool allocation_allowed(const ProtocolSession* session, size_t size) { + return (unsigned long long)size <= session->max_alloc; +} + +static void* protocol_alloc_for_session(const ProtocolSession* session, size_t size) { + if (!allocation_allowed(session, size)) + return NULL; + return malloc(size); +} + +static void* protocol_realloc_for_session(const ProtocolSession* session, void* ptr, size_t size) { + if (!allocation_allowed(session, size)) + return NULL; + return realloc(ptr, size); +} + +void* protocol_alloc(size_t size) { + const ProtocolSession* session = bound_session ? bound_session : &legacy_io_session; + return protocol_alloc_for_session(session, size); +} + +void* protocol_realloc(void* ptr, size_t size) { + const ProtocolSession* session = bound_session ? bound_session : &legacy_io_session; + return protocol_realloc_for_session(session, ptr, size); +} + +void protocol_session_bind(ProtocolSession* session) { + bound_session = session; + log_set_8_bit_output(session && session->eight_bit_output); +} + +void protocol_session_unbind(void) { + bound_session = NULL; +} + +void protocol_session_set_ssl(ProtocolSession* session, SSL* ssl) { + if (session) + session->ssl = ssl; +} + +static void bw_mutex_init(void) { + mtx_init(&bw_mutex, mtx_plain); +} + +static unsigned long long global_bwlimit(void) { + unsigned long long limit; + call_once(&bw_mutex_once, bw_mutex_init); + mtx_lock(&bw_mutex); + limit = io_bwlimit; + mtx_unlock(&bw_mutex); + return limit; } void io_set_bwlimit(unsigned long long bytes_per_sec) { - io_bwlimit = bytes_per_sec; - bw_tokens = (long long)io_bwlimit; - clock_gettime(CLOCK_MONOTONIC, &bw_last_refill); + call_once(&bw_mutex_once, bw_mutex_init); + mtx_lock(&bw_mutex); + io_bwlimit = + bytes_per_sec > (unsigned long long)LLONG_MAX ? (unsigned long long)LLONG_MAX : bytes_per_sec; + mtx_unlock(&bw_mutex); } -static void bw_throttle(size_t bytes_written) { - if (io_bwlimit == 0) +void protocol_session_set_bwlimit(ProtocolSession* session, unsigned long long bytes_per_sec) { + if (!session) + return; + session->bwlimit = + bytes_per_sec > (unsigned long long)LLONG_MAX ? (unsigned long long)LLONG_MAX : bytes_per_sec; + session->bw_tokens = (long long)session->bwlimit; + struct timespec now; + clock_gettime(CLOCK_MONOTONIC, &now); + session->bw_last_refill_sec = now.tv_sec; + session->bw_last_refill_nsec = now.tv_nsec; +} + +void protocol_session_set_8_bit_output(ProtocolSession* session, bool enabled) { + if (!session) + return; + session->eight_bit_output = enabled; + if (session == bound_session) + log_set_8_bit_output(enabled); +} + +void protocol_set_8_bit_output(bool enabled) { + ProtocolSession* session = bound_session ? bound_session : &legacy_io_session; + protocol_session_set_8_bit_output(session, enabled); +} + +static void bw_throttle_session(ProtocolSession* session, size_t bytes_written) { + if (session->bwlimit == 0) return; struct timespec now; clock_gettime(CLOCK_MONOTONIC, &now); - long long elapsed_ns = - (now.tv_sec - bw_last_refill.tv_sec) * 1000000000LL + (now.tv_nsec - bw_last_refill.tv_nsec); - bw_last_refill = now; + long long elapsed_ns = (now.tv_sec - session->bw_last_refill_sec) * 1000000000LL + + (now.tv_nsec - session->bw_last_refill_nsec); + session->bw_last_refill_sec = now.tv_sec; + session->bw_last_refill_nsec = now.tv_nsec; - long long tokens_to_add = (long long)((double)io_bwlimit * elapsed_ns / 1000000000.0); - bw_tokens += tokens_to_add; - if (bw_tokens > (long long)io_bwlimit) - bw_tokens = (long long)io_bwlimit; + long long tokens_to_add = (long long)((double)session->bwlimit * elapsed_ns / 1000000000.0); + session->bw_tokens += tokens_to_add; + if (session->bw_tokens > (long long)session->bwlimit) + session->bw_tokens = (long long)session->bwlimit; - bw_tokens -= (long long)bytes_written; + session->bw_tokens -= bytes_written; - if (bw_tokens < 0) { - long long deficit_us = (long long)((double)(-bw_tokens) / io_bwlimit * 1000000.0); + if (session->bw_tokens < 0) { + long long deficit_us = + (long long)((double)(-session->bw_tokens) / session->bwlimit * 1000000.0); if (deficit_us >= 1000) poll(NULL, 0, (int)(deficit_us / 1000)); else usleep((useconds_t)deficit_us); - bw_tokens = 0; - clock_gettime(CLOCK_MONOTONIC, &bw_last_refill); + session->bw_tokens = 0; + session->bw_last_refill_sec = now.tv_sec; + session->bw_last_refill_nsec = now.tv_nsec; } } void io_set_ssl(SSL* ssl) { + bound_session = NULL; io_ssl = ssl; } @@ -71,69 +213,149 @@ SSL* io_get_ssl(void) { return io_ssl; } -static int io_fd(int dir_fd, int file_descriptor) { - return (dir_fd != -1) ? dir_fd : file_descriptor; +static ProtocolSession* legacy_session(int read_fd, int write_fd) { + if (bound_session) + return bound_session; + int target_read_fd = io_read_fd != -1 ? io_read_fd : read_fd; + int target_write_fd = io_write_fd != -1 ? io_write_fd : write_fd; + if (legacy_io_session.read_fd != target_read_fd || + legacy_io_session.write_fd != target_write_fd) { + legacy_io_session.read_fd = target_read_fd; + legacy_io_session.write_fd = target_write_fd; + atomic_store(&legacy_io_session.total_allocated_bytes, 0); + legacy_io_session.max_alloc = DEFAULT_MAX_ALLOC; + protocol_session_set_bwlimit(&legacy_io_session, global_bwlimit()); + } else if (legacy_io_session.bwlimit != global_bwlimit()) { + protocol_session_set_bwlimit(&legacy_io_session, global_bwlimit()); + } + legacy_io_session.ssl = io_ssl; + return &legacy_io_session; } bool send_n_data(int file_descriptor, const void* data, size_t data_size) { - log_message(LOG_LEVEL_DEBUG, " Sending n Data: %zu", data_size); - int fd = io_fd(io_write_fd, file_descriptor); + return protocol_send_n_data(legacy_session(-1, file_descriptor), data, data_size); +} + +bool receive_n_data(int file_descriptor, void* data, size_t data_size) { + return protocol_receive_n_data(legacy_session(file_descriptor, -1), data, data_size); +} + +static int deadline_remaining_ms(const struct timespec* deadline) { + struct timespec now; + clock_gettime(CLOCK_MONOTONIC, &now); + long long ns = + (long long)(deadline->tv_sec - now.tv_sec) * 1000000000LL + deadline->tv_nsec - now.tv_nsec; + if (ns <= 0) + return 0; + long long ms = (ns + 999999) / 1000000; + return ms > INT_MAX ? INT_MAX : (int)ms; +} + +bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t data_size) { + if (!data && data_size != 0) + return false; + log_debug_message(LOG_DEBUG_IO, " Sending n Data: %zu", data_size); + if (!session) + return false; + int fd = session->write_fd; + struct timespec deadline; + clock_gettime(CLOCK_MONOTONIC, &deadline); + deadline.tv_sec += SEND_TIMEOUT_SEC; + short wait_events = POLLOUT; ssize_t total_bytes_send = 0; while ((size_t)total_bytes_send < data_size) { size_t chunk = data_size - total_bytes_send; - if (io_bwlimit > 0 && chunk > 65536) + if (session->bwlimit > 0 && chunk > 65536) chunk = 65536; + struct pollfd pfd = {.fd = fd, .events = wait_events}; + int poll_result = poll(&pfd, 1, deadline_remaining_ms(&deadline)); + if (poll_result == 0 || (poll_result < 0 && errno != EINTR)) { + log_message(LOG_LEVEL_ERROR, "Send timeout or poll failure"); + return false; + } + if (poll_result < 0) + continue; + if (pfd.revents & (POLLERR | POLLNVAL)) + return false; ssize_t bytes_send; - if (io_ssl) - bytes_send = SSL_write(io_ssl, (const char*)data + total_bytes_send, chunk); + if (session->ssl) + bytes_send = SSL_write(session->ssl, (const char*)data + total_bytes_send, chunk); else bytes_send = write(fd, (const char*)data + total_bytes_send, chunk); if (bytes_send <= 0) { - if (io_ssl) { - int ssl_err = SSL_get_error(io_ssl, (int)bytes_send); - if (ssl_err == SSL_ERROR_WANT_WRITE || ssl_err == SSL_ERROR_WANT_READ) + if (session->ssl) { + int ssl_err = SSL_get_error(session->ssl, (int)bytes_send); + if (ssl_err == SSL_ERROR_WANT_WRITE || ssl_err == SSL_ERROR_WANT_READ) { + wait_events = ssl_err == SSL_ERROR_WANT_WRITE ? POLLOUT : POLLIN; continue; + } } log_message(LOG_LEVEL_ERROR, "Could not send data"); return false; } - bw_throttle((size_t)bytes_send); + bw_throttle_session(session, (size_t)bytes_send); total_bytes_send += bytes_send; + if (session->ssl) + wait_events = POLLOUT; } - log_message(LOG_LEVEL_DEBUG, " Send n Data: %zu", total_bytes_send); + log_debug_message(LOG_DEBUG_IO, " Send n Data: %zu", total_bytes_send); return true; } -bool receive_n_data(int file_descriptor, void* data, size_t data_size) { - log_message(LOG_LEVEL_DEBUG, " Receiving n Data: %zu", data_size); - int fd = io_fd(io_read_fd, file_descriptor); +bool protocol_receive_n_data_timed(ProtocolSession* session, void* data, size_t data_size, + int timeout_sec); + +bool protocol_receive_n_data(ProtocolSession* session, void* data, size_t data_size) { + return protocol_receive_n_data_timed(session, data, data_size, RECEIVE_TIMEOUT_SEC); +} + +bool protocol_receive_n_data_timed(ProtocolSession* session, void* data, size_t data_size, + int timeout_sec) { + log_debug_message(LOG_DEBUG_IO, " Receiving n Data: %zu", data_size); + if (!session) + return false; + int fd = session->read_fd; + if (timeout_sec <= 0) + timeout_sec = RECEIVE_TIMEOUT_SEC; struct timespec deadline; clock_gettime(CLOCK_MONOTONIC, &deadline); - deadline.tv_sec += RECEIVE_TIMEOUT_SEC; + deadline.tv_sec += timeout_sec; size_t total_bytes_received = 0; + short wait_events = POLLIN; while (total_bytes_received < data_size) { - struct timespec now; - clock_gettime(CLOCK_MONOTONIC, &now); - if (now.tv_sec > deadline.tv_sec || - (now.tv_sec == deadline.tv_sec && now.tv_nsec > deadline.tv_nsec)) { - log_message(LOG_LEVEL_ERROR, "Receive timeout after %ds", RECEIVE_TIMEOUT_SEC); - return false; + if (!session->ssl || SSL_pending(session->ssl) == 0) { + struct pollfd pfd = {.fd = fd, .events = wait_events}; + int poll_result = poll(&pfd, 1, deadline_remaining_ms(&deadline)); + if (poll_result == 0) { + log_message(LOG_LEVEL_ERROR, "Receive timeout after %ds", timeout_sec); + return false; + } + if (poll_result < 0) { + if (errno == EINTR) + continue; + return false; + } + /* POLLHUP may accompany the final readable bytes on pipes/sockets. */ + if (pfd.revents & (POLLERR | POLLNVAL)) + return false; } ssize_t bytes_received; - if (io_ssl) - bytes_received = - SSL_read(io_ssl, (char*)data + total_bytes_received, data_size - total_bytes_received); + if (session->ssl) + bytes_received = SSL_read(session->ssl, (char*)data + total_bytes_received, + data_size - total_bytes_received); else bytes_received = read(fd, (char*)data + total_bytes_received, data_size - total_bytes_received); if (bytes_received <= 0) { - if (io_ssl) { - int ssl_err = SSL_get_error(io_ssl, (int)bytes_received); - if (ssl_err == SSL_ERROR_WANT_WRITE || ssl_err == SSL_ERROR_WANT_READ) + if (session->ssl) { + int ssl_err = SSL_get_error(session->ssl, (int)bytes_received); + if (ssl_err == SSL_ERROR_WANT_WRITE || ssl_err == SSL_ERROR_WANT_READ) { + wait_events = ssl_err == SSL_ERROR_WANT_WRITE ? POLLOUT : POLLIN; continue; + } } if (bytes_received == 0) log_message(LOG_LEVEL_ERROR, "Connection closed while receiving data"); @@ -141,9 +363,11 @@ bool receive_n_data(int file_descriptor, void* data, size_t data_size) { log_message(LOG_LEVEL_ERROR, "Could not receive bytes"); return false; } - total_bytes_received += bytes_received; + total_bytes_received += (size_t)bytes_received; + if (session->ssl) + wait_events = POLLIN; } - log_message(LOG_LEVEL_DEBUG, " Received n Data: %zu", total_bytes_received); + log_debug_message(LOG_DEBUG_IO, " Received n Data: %zu", total_bytes_received); return true; } @@ -171,103 +395,243 @@ static const char* status_to_string(Status status) { return "ABORT"; case STATUS_CHECK_BATCH: return "CHECK_BATCH"; + case STATUS_MKDIR: + return "MKDIR"; + case STATUS_APPEND: + return "APPEND"; + case STATUS_APPEND_SIG: + return "APPEND_SIG"; + case STATUS_APPEND_OK: + return "APPEND_OK"; + case STATUS_APPEND_DATA: + return "APPEND_DATA"; + case STATUS_HARDLINK: + return "HARDLINK"; + case STATUS_SYMLINK: + return "SYMLINK"; + case STATUS_SPECIAL: + return "SPECIAL"; + case STATUS_DIR_TIMES: + return "DIR_TIMES"; + case STATUS_AUTH_CHALLENGE: + return "AUTH_CHALLENGE"; + case STATUS_AUTH_RESPONSE: + return "AUTH_RESPONSE"; + case STATUS_AUTH_OK: + return "AUTH_OK"; + case STATUS_AUTH_FAILED: + return "AUTH_FAILED"; default: return "UNKNOWN"; } } -bool send_str(int file_descriptor, const char* data) { +/* Shared string send/receive implementation. `redact` selects whether the + * payload body is written to the LOG_DEBUG_PROTO debug log: daemon auth material + * (the username and the proof/signature fields) sets it so a --verbose log never + * captures a replayable credential, while every other string keeps its normal + * debug trace. */ +static bool protocol_send_str_impl(ProtocolSession* session, const char* data, bool redact) { + if (data == NULL) + return false; size_t size = strlen(data); - if (!send_n_data(file_descriptor, &size, sizeof(size_t))) + if (!protocol_send_n_data(session, &size, sizeof(size_t))) return false; - if (!send_n_data(file_descriptor, data, size)) + if (!protocol_send_n_data(session, data, size)) return false; - log_message(LOG_LEVEL_DEBUG, "Send String: %s", data); + if (redact) { + log_debug_message(LOG_DEBUG_PROTO, "Send String: "); + } else if (log_debug_enabled(LOG_DEBUG_PROTO)) { + char* escaped_data = output_escape(data, log_get_8_bit_output()); + log_debug_message(LOG_DEBUG_PROTO, "Send String: %s", + escaped_data ? escaped_data : ""); + free(escaped_data); + } return true; } -char* receive_str(int file_descriptor) { +static char* protocol_receive_str_impl(ProtocolSession* session, bool redact) { size_t size; - if (!receive_n_data(file_descriptor, &size, sizeof(size_t))) + if (!protocol_receive_n_data(session, &size, sizeof(size_t))) return NULL; - if (size > MAX_STRING_SIZE) { + if (size > MAX_STRING_SIZE || size > SIZE_MAX - 1) { log_message(LOG_LEVEL_ERROR, "String size %zu exceeds maximum %llu", size, (unsigned long long)MAX_STRING_SIZE); return NULL; } - char* data = (char*)malloc(size + 1); + char* data = (char*)protocol_alloc_for_session(session, size + 1); if (data == NULL) return NULL; - if (!receive_n_data(file_descriptor, data, size)) { + if (!protocol_receive_n_data(session, data, size)) { free(data); return NULL; } + if (memchr(data, '\0', size) != NULL) { + free(data); + log_message(LOG_LEVEL_ERROR, "Received string contains an embedded NUL"); + return NULL; + } data[size] = '\0'; - log_message(LOG_LEVEL_DEBUG, "Received String: %s", data); + if (redact) { + log_debug_message(LOG_DEBUG_PROTO, "Received String: "); + } else if (log_debug_enabled(LOG_DEBUG_PROTO)) { + char* escaped_data = output_escape(data, log_get_8_bit_output()); + log_debug_message(LOG_DEBUG_PROTO, "Received String: %s", + escaped_data ? escaped_data : ""); + free(escaped_data); + } return data; } -bool send_data(int file_descriptor, const Data* data) { +bool protocol_send_str(ProtocolSession* session, const char* data) { + return protocol_send_str_impl(session, data, false); +} + +bool protocol_send_str_redacted(ProtocolSession* session, const char* data) { + return protocol_send_str_impl(session, data, true); +} + +char* protocol_receive_str(ProtocolSession* session) { + return protocol_receive_str_impl(session, false); +} + +char* protocol_receive_str_redacted(ProtocolSession* session) { + return protocol_receive_str_impl(session, true); +} + +bool protocol_send_data(ProtocolSession* session, const Data* data) { + if (!data || (!data->data && data->size != 0)) + return false; + if (!session) + return false; unsigned long long data_size = data->size; - if (!send_n_data(file_descriptor, &data_size, sizeof(unsigned long long))) + if (!protocol_send_n_data(session, &data_size, sizeof(unsigned long long))) return false; - if (!send_n_data(file_descriptor, data->data, data_size)) + if (!protocol_send_n_data(session, data->data, data_size)) return false; - log_message(LOG_LEVEL_DEBUG, "Send %lld data", data_size); + log_debug_message(LOG_DEBUG_PROTO, "Send %lld data", data_size); return true; } -Data* receive_data(int file_descriptor) { - unsigned long long size = 0; - if (!receive_n_data(file_descriptor, &size, sizeof(unsigned long long))) +Data* protocol_receive_data_limited(ProtocolSession* session, unsigned long long maximum_size) { + if (!session) return NULL; - if (size > MAX_DATA_SIZE) { + unsigned long long size = 0; + if (!protocol_receive_n_data(session, &size, sizeof(unsigned long long))) + return NULL; + if (size > MAX_DATA_PAYLOAD_SIZE || size > maximum_size) { log_message(LOG_LEVEL_ERROR, "Data size %llu exceeds maximum %llu", size, - (unsigned long long)MAX_DATA_SIZE); + (unsigned long long)MAX_DATA_PAYLOAD_SIZE); return NULL; } - if (total_allocated_bytes + size > MAX_CONNECTION_MEMORY) { + if (size > SIZE_MAX) + return NULL; + size_t allocation_size = size == 0 ? 1 : (size_t)size; + if (!protocol_reserve_memory(session, allocation_size)) { log_message(LOG_LEVEL_ERROR, "Per-connection memory limit exceeded (%llu + %llu > %llu)", - (unsigned long long)total_allocated_bytes, size, + (unsigned long long)atomic_load(&session->total_allocated_bytes), size, (unsigned long long)MAX_CONNECTION_MEMORY); return NULL; } - void* data = malloc((size_t)size); - if (data == NULL) - return NULL; - if (!receive_n_data(file_descriptor, data, (size_t)size)) { - free(data); + void* data = protocol_alloc_for_session(session, allocation_size); + if (data == NULL) { + protocol_release_memory_for_session(session, allocation_size); return NULL; } - total_allocated_bytes += size; - log_message(LOG_LEVEL_DEBUG, "Received %lld data", size); - return data_create(data, (size_t)size); + if (!protocol_receive_n_data(session, data, (size_t)size)) { + free(data); + protocol_release_memory_for_session(session, allocation_size); + return NULL; + } + log_debug_message(LOG_DEBUG_PROTO, "Received %lld data", size); + Data* result = data_create(data, (size_t)size); + if (!result) { + protocol_release_memory_for_session(session, allocation_size); + return NULL; + } + result->protocol_charge = allocation_size; + return result; } -bool send_int(int file_descriptor, int data) { - if (!send_n_data(file_descriptor, &data, sizeof(int))) +Data* protocol_receive_data(ProtocolSession* session) { + return protocol_receive_data_limited(session, MAX_DATA_PAYLOAD_SIZE); +} + +bool protocol_send_int(ProtocolSession* session, int data) { + if (!protocol_send_n_data(session, &data, sizeof(int))) return false; - log_message(LOG_LEVEL_DEBUG, "Send Int: %d", data); + log_debug_message(LOG_DEBUG_PROTO, "Send Int: %d", data); return true; } -bool receive_int(int file_descriptor, int* data) { - if (!receive_n_data(file_descriptor, data, sizeof(int))) +bool protocol_receive_int(ProtocolSession* session, int* data) { + if (!protocol_receive_n_data(session, data, sizeof(int))) return false; - log_message(LOG_LEVEL_DEBUG, "Received Int: %d", *data); + log_debug_message(LOG_DEBUG_PROTO, "Received Int: %d", *data); return true; } -bool send_status(int file_descriptor, Status status) { - if (!send_n_data(file_descriptor, &status, sizeof(Status))) +bool protocol_send_status(ProtocolSession* session, Status status) { + if (!protocol_send_n_data(session, &status, sizeof(Status))) return false; - log_message(LOG_LEVEL_DEBUG, "Send Status: %s", status_to_string(status)); + log_debug_message(LOG_DEBUG_PROTO, "Send Status: %s", status_to_string(status)); return true; } -bool receive_status(int file_descriptor, Status* status) { - if (!receive_n_data(file_descriptor, status, sizeof(Status))) +bool protocol_receive_status(ProtocolSession* session, Status* status) { + if (!protocol_receive_n_data(session, status, sizeof(Status))) return false; - log_message(LOG_LEVEL_DEBUG, "Received Status: %s", status_to_string(*status)); + log_debug_message(LOG_DEBUG_PROTO, "Received Status: %s", status_to_string(*status)); return true; } + +/* protocol_receive_status with an explicit per-message deadline (seconds). + Used where a single reply may legitimately take far longer than the default + 60 s receive window - e.g. the sender waiting for the early-delete ACK after + the receiver committed a large (up to MAX_SERVER_DELETE_COUNT) deletion. */ +bool protocol_receive_status_timed(ProtocolSession* session, Status* status, int timeout_sec) { + if (!protocol_receive_n_data_timed(session, status, sizeof(Status), timeout_sec)) + return false; + log_debug_message(LOG_DEBUG_PROTO, "Received Status: %s", status_to_string(*status)); + return true; +} + +bool send_str(int fd, const char* data) { + return protocol_send_str(legacy_session(-1, fd), data); +} +char* receive_str(int fd) { + return protocol_receive_str(legacy_session(fd, -1)); +} +/* Redacted variants: identical framing, but the string body is never written to + the debug protocol log. Used for daemon auth material (username, proof, + signature). */ +bool send_str_redacted(int fd, const char* data) { + return protocol_send_str_redacted(legacy_session(-1, fd), data); +} +char* receive_str_redacted(int fd) { + return protocol_receive_str_redacted(legacy_session(fd, -1)); +} +bool send_data(int fd, const Data* data) { + return protocol_send_data(legacy_session(-1, fd), data); +} +Data* receive_data(int fd) { + return protocol_receive_data_limited(legacy_session(fd, -1), MAX_DATA_PAYLOAD_SIZE); +} +Data* receive_data_limited(int fd, unsigned long long maximum_size) { + return protocol_receive_data_limited(legacy_session(fd, -1), maximum_size); +} +bool send_int(int fd, int data) { + return protocol_send_int(legacy_session(-1, fd), data); +} +bool receive_int(int fd, int* data) { + return protocol_receive_int(legacy_session(fd, -1), data); +} +bool send_status(int fd, Status status) { + return protocol_send_status(legacy_session(-1, fd), status); +} +bool receive_status(int fd, Status* status) { + return protocol_receive_status(legacy_session(fd, -1), status); +} +bool receive_status_timed(int fd, Status* status, int timeout_sec) { + return protocol_receive_status_timed(legacy_session(fd, -1), status, timeout_sec); +} diff --git a/src/shared/protocol.h b/src/shared/protocol.h index b65584d..e23a68f 100644 --- a/src/shared/protocol.h +++ b/src/shared/protocol.h @@ -4,15 +4,56 @@ #include "data.h" #include #include +#include /* Maximum allowed string size for receive_str (64 KB) */ #define MAX_STRING_SIZE (64 * 1024) -/* Maximum allowed data payload size for receive_data (100 MB) */ -#define MAX_DATA_PAYLOAD_SIZE (100ULL * 1024 * 1024) +/* Maximum uncompressed file payload accepted by the receiver's whole-file + * paths. A single whole file is charged against the per-connection memory + * reservation (MAX_CONNECTION_MEMORY) and against the server allocation + * ceiling (MAX_SERVER_ALLOC), so this mirrors those 256 MB bounds rather than + * the older 64 MB chunk-era cap. Chunk-serialized payloads keep their own + * 64 MB cap (MAX_CHUNK_SIZE). */ +#define MAX_RECEIVE_WHOLE_FILE_SIZE (256ULL * 1024 * 1024) + +/* Maximum allowed data payload size for receive_data (whole-file bound) */ +#define MAX_DATA_PAYLOAD_SIZE MAX_RECEIVE_WHOLE_FILE_SIZE + +/* Maximum chunk size (64 MB) — prevents unbounded allocation from the wire */ +#define MAX_CHUNK_SIZE (64ULL * 1024 * 1024) +#define MAX_MANIFEST_ENTRIES (1024 * 1024) +/* Aggregate bytes retained by one received deletion manifest. */ +#define MAX_MANIFEST_BYTES (16ULL * 1024 * 1024) +#define DEFAULT_MAX_ALLOC (1ULL * 1024 * 1024 * 1024) +/* Server policy ceiling for a client-provided allocation limit. */ +#define MAX_SERVER_ALLOC (256ULL * 1024 * 1024) +/* Bounded cumulative per-connection receive budget. In-flight wire buffers, + decompression buffers and queued (not yet written) file payloads for a + connection must stay within this ceiling. */ +#define MAX_CONNECTION_MEMORY (256ULL * 1024 * 1024) typedef struct ssl_st SSL; +/* + * Explicit owner of protocol I/O. A session does not own the descriptors or + * SSL object; it only describes the transport used by a transfer. This makes + * it safe to pass the transport to a worker without relying on inherited + * thread-local state. + */ +typedef struct ProtocolSession { + int read_fd; + int write_fd; + SSL* ssl; + unsigned long long bwlimit; + long long bw_tokens; + long long bw_last_refill_sec; + long bw_last_refill_nsec; + atomic_ullong total_allocated_bytes; + bool eight_bit_output; + unsigned long long max_alloc; +} ProtocolSession; + typedef int Status; enum NET_STATUS { STATUS_OK, @@ -26,23 +67,121 @@ enum NET_STATUS { STATUS_DELTA_DATA, STATUS_KEEPALIVE, STATUS_ABORT, - STATUS_CHECK_BATCH + STATUS_CHECK_BATCH, + /* An explicit directory entry (--dirs): the sender transmits only the path; + * the receiver creates the directory below the receive root. */ + STATUS_MKDIR, + /* --append / --append-verify tail resume. STATUS_APPEND is sent by the + * receiver after a per-file STATUS_CHECK when the existing destination file + * is SHORTER than the source and an append mode is negotiated: its payload is + * the resume offset (the number of prefix bytes already present), after which + * the sender answers either directly with STATUS_APPEND_DATA (plain --append, + * prefix not verified) or, for --append-verify, first with STATUS_APPEND_SIG + * carrying the xxHash64 of the source prefix; the receiver then replies + * STATUS_APPEND_OK (prefix matched -> sender transmits the tail) or + * STATUS_NEXT (prefix mismatch -> sender falls back to a full transfer). + * STATUS_APPEND_DATA carries the tail bytes (compressed data frame). */ + STATUS_APPEND, + STATUS_APPEND_SIG, + STATUS_APPEND_OK, + STATUS_APPEND_DATA, + /* --hard-links/-H: a sibling (later member) of a source hard-link group. + * The sender transmits only the path, the run-local link-group id, and the + * first (data-carrying) member's destination-relative wire path; the receiver + * creates this entry as a hard link to the first member's installed file + * (falling back to a byte-identical copy if link() fails). Protocol 2.12.0. */ + STATUS_HARDLINK, + /* A symlink-type entry (-l/--links, -k/--copy-dirlinks' keep-as-symlink + * branch). The sender transmits the destination path, the (sender-munged, + * if --munge-links) symlink target, and optional metadata; the receiver + * creates a symlink to the unmunged target beneath the receive root (see + * file_receive_symlink). Protocol 2.13.0. */ + STATUS_SYMLINK, + /* --devices / --specials (-D): a device or special node the sender wants + * recreated (not written from content). Payload: destination path, the + * metadata frame (whose mode's S_IFMT bits carry the node kind), and two + * int32 rdev major/minor fields. The receiver validates the kind and rdev, + * confines the node below the receive root, and recreates it (mknod/mkfifo), + * privilege-gating the mknod. Protocol 2.13.0. */ + STATUS_SPECIAL, + /* Directory-time superstructure (P7 Wave D, protocol 2.17.0): one or more + * trailing frames sent after all file data (and after the optional delete + * manifest) carrying the source directories' captured metadata so the + * receiver can apply directory mtimes/atimes AFTER all of a directory's + * children have been written. Payload per frame: an int count, then count + * repetitions of (wire path string, metadata frame); an entry count larger + * than MAX_MANIFEST_ENTRIES is split across repeated frames. The receiver + * defers the actual utimensat until its own delete/publish phase has + * committed, then skips the whole set when -O/--omit-dir-times is set. */ + STATUS_DIR_TIMES, + /* Daemon SCRAM-SHA-256 authentication (A7 remediation, protocol 2.19.0). + * STATUS_AUTH_CHALLENGE: the server requires auth and is about to send the + * iteration count, the base64 salt and the base64 server nonce. + * STATUS_AUTH_RESPONSE: the client's reply, followed by the base64 client + * nonce and the base64 ClientProof. STATUS_AUTH_OK: the client proof + * verified, followed by the base64 ServerSignature. STATUS_AUTH_FAILED: + * a single generic refusal (unknown user, off-list user, wrong proof, + * missing/malformed credentials) after which the server closes without + * writing any data. */ + STATUS_AUTH_CHALLENGE, + STATUS_AUTH_RESPONSE, + STATUS_AUTH_OK, + STATUS_AUTH_FAILED }; void io_set_fds(int read_fd, int write_fd); void io_set_bwlimit(unsigned long long bytes_per_sec); void io_set_ssl(SSL* ssl); SSL* io_get_ssl(void); + +void protocol_session_init(ProtocolSession* session, int read_fd, int write_fd); +/* Transitional bridge for helpers whose signatures still carry only an fd. */ +void protocol_session_bind(ProtocolSession* session); +void protocol_session_unbind(void); +void protocol_session_set_ssl(ProtocolSession* session, SSL* ssl); +void protocol_session_set_bwlimit(ProtocolSession* session, unsigned long long bytes_per_sec); +void protocol_session_set_max_alloc(ProtocolSession* session, unsigned long long max_alloc); +void* protocol_alloc(size_t size); +void* protocol_realloc(void* ptr, size_t size); +void protocol_session_set_8_bit_output(ProtocolSession* session, bool enabled); +void protocol_set_8_bit_output(bool enabled); +bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t data_size); +bool protocol_receive_n_data(ProtocolSession* session, void* data, size_t data_size); +bool protocol_send_str(ProtocolSession* session, const char* data); +char* protocol_receive_str(ProtocolSession* session); +/* Redacted string variants: identical wire framing to protocol_send_str / + * protocol_receive_str, but the payload body is replaced by `` in the + * LOG_DEBUG_PROTO debug log. Used for daemon auth material (the username and + * the proof/signature fields) so a --verbose log can never capture a credential + * that could be replayed. */ +bool protocol_send_str_redacted(ProtocolSession* session, const char* data); +char* protocol_receive_str_redacted(ProtocolSession* session); +bool protocol_send_data(ProtocolSession* session, const Data* data); +Data* protocol_receive_data(ProtocolSession* session); +Data* protocol_receive_data_limited(ProtocolSession* session, unsigned long long maximum_size); +bool protocol_send_int(ProtocolSession* session, int data); +bool protocol_receive_int(ProtocolSession* session, int* data); +bool protocol_send_status(ProtocolSession* session, Status status); +bool protocol_receive_status(ProtocolSession* session, Status* status); bool send_n_data(int file_descriptor, const void* data, size_t data_size); bool receive_n_data(int file_descriptor, void* data, size_t data_size); bool send_str(int file_descriptor, const char* data); char* receive_str(int file_descriptor); +/* Redacted fd-level string variants (see protocol_send_str_redacted). */ +bool send_str_redacted(int file_descriptor, const char* data); +char* receive_str_redacted(int file_descriptor); bool send_data(int file_descriptor, const Data* data); Data* receive_data(int file_descriptor); +Data* receive_data_limited(int file_descriptor, unsigned long long maximum_size); bool send_int(int file_descriptor, int data); bool receive_int(int file_descriptor, int* data); bool send_status(int file_descriptor, Status status); bool receive_status(int file_descriptor, Status* status); +/* receive_status with an explicit per-message deadline in seconds, instead of + the default RECEIVE_TIMEOUT_SEC. A reply that may legitimately take longer + (e.g. the early-delete ACK after a large receiver-side deletion) must use + this so the sender does not abort after the deletion already committed. */ +bool receive_status_timed(int file_descriptor, Status* status, int timeout_sec); #endif diff --git a/src/shared/queue.c b/src/shared/queue.c index 14d901d..e4e5f09 100644 --- a/src/shared/queue.c +++ b/src/shared/queue.c @@ -1,132 +1,155 @@ -#include -#include -#include -#include -#include - -#include "queue.h" - -Queue* queue_create(int capacity, void (*destroyer)(void* item)) { - Queue* queue = (Queue*)malloc(sizeof(Queue)); - if (queue == NULL) { - perror("ERROR: Could not allocate memory for queue structure"); - return NULL; - } - - queue->items = malloc(capacity * sizeof(void*)); - if (queue->items == NULL) { - free(queue); - return NULL; - } - - for (int i = 0; i < capacity; ++i) { - queue->items[i] = NULL; - } - - queue->capacity = capacity; - queue->front = 0; - queue->rear = 0; - queue->size = 0; - queue->item_destroyer = destroyer; - - return queue; -} - -void queue_destroy(Queue* queue) { - if (queue == NULL) - return; - - if (queue->item_destroyer != NULL) { - for (int i = 0; i < queue->size; ++i) { - int index = (queue->front + i) % queue->capacity; - queue->item_destroyer(queue->items[index]); - } - } - free(queue->items); - free(queue); -} - -bool queue_is_empty(const Queue* queue) { - if (queue == NULL) - return true; - return queue->size == 0; -} - -bool queue_is_full(const Queue* queue) { - if (queue == NULL) - return false; - return queue->size == queue->capacity; -} - -static bool queue_double_capacity(Queue* queue) { - if (queue == NULL) - return false; - unsigned int new_capacity = queue->capacity * 2; - if (new_capacity <= 1) - new_capacity = 100; - void** new_items = malloc(new_capacity * sizeof(void*)); - if (new_items == NULL) { - perror("ERROR: Could not allocate memory for doubling capacity of queue."); - return false; - } - for (int i = 0; i < queue->size; i++) - new_items[i] = queue->items[(i + queue->front) % queue->capacity]; - free(queue->items); - queue->items = new_items; - queue->front = 0; - queue->rear = queue->size; - queue->capacity = new_capacity; - return true; -} - -bool queue_enqueue(Queue* queue, void* item) { - if (queue == NULL || item == NULL) - return false; - if (queue_is_full(queue)) { - if (!queue_double_capacity(queue)) - return false; - } - queue->items[queue->rear] = item; - queue->rear = (queue->rear + 1) % queue->capacity; - queue->size++; - return true; -} - -bool queue_enqueue_multithreaded(Queue* queue, void* item, mtx_t* mutex, cnd_t* condition_not_empty, - cnd_t* condition_not_full) { - mtx_lock(mutex); - while (queue_is_full(queue)) - cnd_wait(condition_not_full, mutex); - bool ok = queue_enqueue(queue, item); - cnd_signal(condition_not_empty); - mtx_unlock(mutex); - return ok; -} - -void* queue_dequeue(Queue* queue) { - if (queue == NULL || queue_is_empty(queue)) { - perror("ERROR: Could not dequeue from null or empty queue."); - return NULL; - } - - void* item = queue->items[queue->front]; - queue->items[queue->front] = NULL; - queue->front = (queue->front + 1) % queue->capacity; - queue->size--; - return item; -} - -void* queue_dequeue_multithreaded(Queue* queue, mtx_t* mutex, cnd_t* condition_not_empty, - cnd_t* condition_not_full, const bool* other_thread_done) { - mtx_lock(mutex); - while (queue_is_empty(queue) && !*other_thread_done) - cnd_wait(condition_not_empty, mutex); - if (queue_is_empty(queue) && *other_thread_done) { - mtx_unlock(mutex); - return NULL; - } - void* item = queue_dequeue(queue); - cnd_signal(condition_not_full); - mtx_unlock(mutex); - return item; -} +#include "log.h" +#include +#include +#include +#include +#include +#include + +#include "queue.h" + +Queue* queue_create(int capacity, void (*destroyer)(void* item)) { + if (capacity <= 0) + return NULL; + + Queue* queue = (Queue*)malloc(sizeof(Queue)); + if (queue == NULL) { + log_perror("ERROR: Could not allocate memory for queue structure"); + return NULL; + } + + queue->items = malloc(capacity * sizeof(void*)); + if (queue->items == NULL) { + free(queue); + return NULL; + } + + for (int i = 0; i < capacity; ++i) { + queue->items[i] = NULL; + } + + queue->capacity = capacity; + queue->front = 0; + queue->rear = 0; + queue->size = 0; + queue->item_destroyer = destroyer; + + return queue; +} + +void queue_destroy(Queue* queue) { + if (queue == NULL) + return; + + if (queue->item_destroyer != NULL) { + for (int i = 0; i < queue->size; ++i) { + int index = (queue->front + i) % queue->capacity; + queue->item_destroyer(queue->items[index]); + } + } + free(queue->items); + free(queue); +} + +bool queue_is_empty(const Queue* queue) { + if (queue == NULL) + return true; + return queue->size == 0; +} + +bool queue_is_full(const Queue* queue) { + if (queue == NULL) + return false; + return queue->size == queue->capacity; +} + +static bool queue_double_capacity(Queue* queue) { + if (queue == NULL) + return false; + if (queue->capacity > INT_MAX / 2) + return false; + int new_capacity = queue->capacity * 2; + if (new_capacity <= 1) + new_capacity = 100; + void** new_items = malloc(new_capacity * sizeof(void*)); + if (new_items == NULL) { + log_perror("ERROR: Could not allocate memory for doubling capacity of queue."); + return false; + } + for (int i = 0; i < queue->size; i++) + new_items[i] = queue->items[(i + queue->front) % queue->capacity]; + free(queue->items); + queue->items = new_items; + queue->front = 0; + queue->rear = queue->size; + queue->capacity = new_capacity; + return true; +} + +bool queue_enqueue(Queue* queue, void* item) { + if (queue == NULL || item == NULL) + return false; + if (queue_is_full(queue)) { + if (!queue_double_capacity(queue)) + return false; + } + queue->items[queue->rear] = item; + queue->rear = (queue->rear + 1) % queue->capacity; + queue->size++; + return true; +} + +bool queue_enqueue_multithreaded(Queue* queue, void* item, mtx_t* mutex, cnd_t* condition_not_empty, + cnd_t* condition_not_full) { + mtx_lock(mutex); + while (queue_is_full(queue)) + cnd_wait(condition_not_full, mutex); + bool ok = queue_enqueue(queue, item); + cnd_signal(condition_not_empty); + mtx_unlock(mutex); + return ok; +} + +bool queue_enqueue_multithreaded_cancel(Queue* queue, void* item, mtx_t* mutex, + cnd_t* condition_not_empty, cnd_t* condition_not_full, + const atomic_bool* cancelled) { + mtx_lock(mutex); + while (queue_is_full(queue) && (cancelled == NULL || !atomic_load(cancelled))) + cnd_wait(condition_not_full, mutex); + if (cancelled != NULL && atomic_load(cancelled)) { + mtx_unlock(mutex); + return false; + } + bool ok = queue_enqueue(queue, item); + cnd_signal(condition_not_empty); + mtx_unlock(mutex); + return ok; +} + +void* queue_dequeue(Queue* queue) { + if (queue == NULL || queue_is_empty(queue)) { + log_perror("ERROR: Could not dequeue from null or empty queue."); + return NULL; + } + + void* item = queue->items[queue->front]; + queue->items[queue->front] = NULL; + queue->front = (queue->front + 1) % queue->capacity; + queue->size--; + return item; +} + +void* queue_dequeue_multithreaded(Queue* queue, mtx_t* mutex, cnd_t* condition_not_empty, + cnd_t* condition_not_full, const bool* other_thread_done) { + mtx_lock(mutex); + while (queue_is_empty(queue) && !*other_thread_done) + cnd_wait(condition_not_empty, mutex); + if (queue_is_empty(queue) && *other_thread_done) { + mtx_unlock(mutex); + return NULL; + } + void* item = queue_dequeue(queue); + cnd_signal(condition_not_full); + mtx_unlock(mutex); + return item; +} diff --git a/src/shared/queue.h b/src/shared/queue.h index 8bd6ba4..7a8bd98 100644 --- a/src/shared/queue.h +++ b/src/shared/queue.h @@ -1,27 +1,31 @@ -#ifndef QUEUE_H -#define QUEUE_H - -#include -#include - -typedef struct Queue { - void** items; - int front; - int rear; - int size; - int capacity; - void (*item_destroyer)(void* item); -} Queue; - -Queue* queue_create(int capacity, void (*destroyer)(void* item)); -void queue_destroy(Queue* queue); -bool queue_is_empty(const Queue* queue); -bool queue_is_full(const Queue* queue); -bool queue_enqueue(Queue* queue, void* item); -bool queue_enqueue_multithreaded(Queue* queue, void* item, mtx_t* mutex, cnd_t* condition_not_empty, - cnd_t* condition_not_full); -void* queue_dequeue(Queue* queue); -void* queue_dequeue_multithreaded(Queue* queue, mtx_t* mutex, cnd_t* condition_not_empty, - cnd_t* condition_not_full, const bool* other_thread_done); - -#endif +#ifndef QUEUE_H +#define QUEUE_H + +#include +#include +#include + +typedef struct Queue { + void** items; + int front; + int rear; + int size; + int capacity; + void (*item_destroyer)(void* item); +} Queue; + +Queue* queue_create(int capacity, void (*destroyer)(void* item)); +void queue_destroy(Queue* queue); +bool queue_is_empty(const Queue* queue); +bool queue_is_full(const Queue* queue); +bool queue_enqueue(Queue* queue, void* item); +bool queue_enqueue_multithreaded(Queue* queue, void* item, mtx_t* mutex, cnd_t* condition_not_empty, + cnd_t* condition_not_full); +bool queue_enqueue_multithreaded_cancel(Queue* queue, void* item, mtx_t* mutex, + cnd_t* condition_not_empty, cnd_t* condition_not_full, + const atomic_bool* cancelled); +void* queue_dequeue(Queue* queue); +void* queue_dequeue_multithreaded(Queue* queue, mtx_t* mutex, cnd_t* condition_not_empty, + cnd_t* condition_not_full, const bool* other_thread_done); + +#endif diff --git a/src/shared/stop_condition.c b/src/shared/stop_condition.c new file mode 100644 index 0000000..b49053c --- /dev/null +++ b/src/shared/stop_condition.c @@ -0,0 +1,157 @@ +#include "stop_condition.h" +#include +#include +#include +#include + +/* Parse a strictly positive decimal integer: only ASCII digits, no leading + * whitespace, sign or trailing garbage. */ +static bool parse_positive_minutes(const char* value, long* out) { + if (!value || *value == '\0') + return false; + if (*value < '0' || *value > '9') + return false; + long v = 0; + for (const char* p = value; *p != '\0'; p++) { + if (*p < '0' || *p > '9') + return false; + int digit = *p - '0'; + if (v > (LONG_MAX - digit) / 10) + return false; + v = v * 10 + digit; + } + if (v <= 0 || v > INT_MAX) + return false; + *out = v; + return true; +} + +bool stop_parse_after_minutes(const char* value, int* out_minutes) { + if (!out_minutes) + return false; + long minutes = 0; + if (!parse_positive_minutes(value, &minutes)) + return false; + *out_minutes = (int)minutes; + return true; +} + +/* Two consecutive ASCII digits -> 0..99. */ +static bool parse_two_digits(const char* s, int* out) { + if (s[0] < '0' || s[0] > '9' || s[1] < '0' || s[1] > '9') + return false; + *out = (s[0] - '0') * 10 + (s[1] - '0'); + return true; +} + +bool stop_parse_at_time(const char* value, time_t now, time_t* out_deadline) { + if (!value || !out_deadline) + return false; + + /* now+N[smhd]: N whole units from the current wall clock. */ + if (strncmp(value, "now+", 4) == 0) { + const char* p = value + 4; + /* The count must be a bare non-negative digit run: reject leading + whitespace ('now+ 5s') and a leading sign ('now++5s'). */ + if (*p < '0' || *p > '9') + return false; + errno = 0; + char* end = NULL; + long amount = strtol(p, &end, 10); + if (errno != 0 || end == p || amount < 0) + return false; + long unit_seconds; + switch (*end) { + case 's': + unit_seconds = 1; + break; + case 'm': + unit_seconds = 60; + break; + case 'h': + unit_seconds = 3600; + break; + case 'd': + unit_seconds = 86400; + break; + default: + return false; + } + if (end[1] != '\0') + return false; + if (amount > LONG_MAX / unit_seconds) + return false; + long long delta = (long long)amount * unit_seconds; + /* Guard against signed overflow of now + delta. */ + if ((long long)now > 0 && delta > (long long)LLONG_MAX - (long long)now) + return false; + if ((long long)now < 0 && delta < (long long)LLONG_MIN - (long long)now) + return false; + *out_deadline = now + (time_t)delta; + return true; + } + + /* HH:MM or HH:MM:SS on the current local day. */ + size_t len = strlen(value); + if (len != 5 && len != 8) + return false; + if (value[2] != ':' || (len == 8 && value[5] != ':')) + return false; + int hh, mm, ss = 0; + if (!parse_two_digits(value, &hh) || !parse_two_digits(value + 3, &mm)) + return false; + if (len == 8 && !parse_two_digits(value + 6, &ss)) + return false; + if (hh > 23 || mm > 59 || ss > 59) + return false; + + struct tm today; + if (!localtime_r(&now, &today)) + return false; + today.tm_hour = hh; + today.tm_min = mm; + today.tm_sec = ss; + today.tm_isdst = -1; + time_t deadline = mktime(&today); + if (deadline == (time_t)-1) + return false; + *out_deadline = deadline; + return true; +} + +StopCondition stop_condition_make(bool has_after, int after_minutes, bool has_at, time_t at_time, + struct timespec now_mono) { + StopCondition condition; + condition.has_monotonic = false; + condition.monotonic_deadline.tv_sec = 0; + condition.monotonic_deadline.tv_nsec = 0; + condition.has_wall = false; + condition.wall_deadline = 0; + if (has_after && after_minutes > 0) { + condition.has_monotonic = true; + condition.monotonic_deadline.tv_sec = now_mono.tv_sec + (time_t)after_minutes * 60; + condition.monotonic_deadline.tv_nsec = now_mono.tv_nsec; + } + if (has_at) { + condition.has_wall = true; + condition.wall_deadline = at_time; + } + return condition; +} + +bool stop_condition_reached(const StopCondition* condition) { + if (!condition) + return false; + if (condition->has_wall && time(NULL) >= condition->wall_deadline) + return true; + if (condition->has_monotonic) { + struct timespec now; + if (clock_gettime(CLOCK_MONOTONIC, &now) != 0) + return false; + if (now.tv_sec > condition->monotonic_deadline.tv_sec || + (now.tv_sec == condition->monotonic_deadline.tv_sec && + now.tv_nsec >= condition->monotonic_deadline.tv_nsec)) + return true; + } + return false; +} \ No newline at end of file diff --git a/src/shared/stop_condition.h b/src/shared/stop_condition.h new file mode 100644 index 0000000..f1ceb7a --- /dev/null +++ b/src/shared/stop_condition.h @@ -0,0 +1,46 @@ +#ifndef STOP_CONDITION_H +#define STOP_CONDITION_H + +#include +#include + +/* Client-only transfer stop conditions (--stop-after=MINS / --stop-at=TIME). + * Both are local sender-side deadlines: they are never serialized into the + * config frame and never bump PROTOCOL_VERSION. A transfer checks the + * condition at natural chunk/file boundaries and, once reached, stops + * elegantly (everything already sent is finalized normally, exit 0). + * + * A condition combines an optional CLOCK_MONOTONIC instant (the relative + * --stop-after duration, immune to wall-clock changes) with an optional + * wall-clock instant (the absolute --stop-at form). Either one being reached + * ends the transfer. */ +typedef struct StopCondition { + bool has_monotonic; + struct timespec monotonic_deadline; + bool has_wall; + time_t wall_deadline; +} StopCondition; + +/* Parse --stop-after=MINS: a positive integer count of minutes. Zero, + * negative, empty and non-numeric values are rejected. Returns true when + * accepted and stores the value in *out_minutes. */ +bool stop_parse_after_minutes(const char* value, int* out_minutes); + +/* Parse --stop-at=TIME. Accepted forms are HH:MM, HH:MM:SS and + * now+N[smhd] (seconds/minutes/hours/days from now). The absolute forms are + * resolved against `now` (local wall clock) and written to *out_deadline; a + * time already in the past yields a deadline <= now ("stop immediately"). + * Returns false on any malformed value. */ +bool stop_parse_at_time(const char* value, time_t now, time_t* out_deadline); + +/* Build the runtime condition at transfer start. after_minutes is the + * relative --stop-after duration (<= 0 disables it); at_time is the absolute + * --stop-at deadline (only consulted when has_at is true); now_mono is the + * CLOCK_MONOTONIC reading at start. */ +StopCondition stop_condition_make(bool has_after, int after_minutes, bool has_at, time_t at_time, + struct timespec now_mono); + +/* True once either deadline has passed (wall clock first, then monotonic). */ +bool stop_condition_reached(const StopCondition* condition); + +#endif diff --git a/src/shared/transport_ssh.c b/src/shared/transport_ssh.c index 7b6dff8..b5010ac 100644 --- a/src/shared/transport_ssh.c +++ b/src/shared/transport_ssh.c @@ -1,10 +1,13 @@ +#include "log.h" #include "transport_ssh.h" #include "utils.h" #include #include +#include #include #include #include +#include #include #include @@ -14,6 +17,14 @@ typedef struct { char* remote_path; } RemoteDest; +/* Writes the exec-failure marker and exits the child. Marked noreturn so + * static analyzers prove the caller's error path never falls through. */ +__attribute__((noreturn)) static void ssh_child_setup_failed(int status_fd) { + ssize_t wret = write(status_fd, "x", 1); + (void)wret; + _exit(1); +} + static void remote_dest_destroy(RemoteDest* r) { free(r->user); free(r->host); @@ -67,16 +78,221 @@ static int parse_remote_dest(const char* dest, RemoteDest* r) { return 0; } -Client* client_connect_ssh(const char* destination, int port, const char* server_path) { +char* ssh_build_remote_command(const char* server_path, bool old_args, char* const* remote_options, + int remote_option_count) { + const char* path = server_path ? server_path : "fastsync-server"; + const char* suffix = " --stdio"; + + /* Each --remote-option=OPT is appended after " --stdio" as one shell word, + escaped with the SAME single-quote boundary used for the server path, so a + value containing shell metacharacters (; & | ` $ ()) can never break out of + the quoting to inject an unrelated remote command. Values are already + validated at CLI parse time (non-empty, no control characters); this layer + only adds the escaping boundary. */ + size_t path_len = strlen(path); + size_t suffix_len = strlen(suffix); + + /* The base command: the server path is ALWAYS quoted as one single-quoted + shell word (remote options below reuse the same escaping), then + " --stdio". Quoting the path is the only injection-safe construction: an + unquoted path would carry shell metacharacters straight into the remote + shell command. --old-args is kept for CLI/ABI compatibility but no longer + disables that protection. */ + (void)old_args; + size_t quote_count = 0; + for (const char* p = path; *p; p++) + if (*p == '\'') + quote_count++; + if (path_len > SIZE_MAX - suffix_len - 4 || + quote_count > (SIZE_MAX - path_len - suffix_len - 4) / 4) + return NULL; + size_t command_len = path_len + quote_count * 4 + suffix_len + 4; + + /* Add each remote option, escaped as one single-quoted word: + " ''", i.e. 1 leading space + 1 open quote + body (len + 3 per + embedded single quote) + 1 close quote = len + q*3 + 3 bytes. + Defense-in-depth against a non-conforming caller: never forward an empty + or control-character value, independent of the CLI validation. */ + for (int i = 0; i < remote_option_count; i++) { + const char* opt = remote_options[i]; + if (!opt || opt[0] == '\0') + return NULL; + size_t len = 0, q = 0; + for (const char* p = opt; *p; p++) { + /* Defense-in-depth: never forward a control character (newline/CR/etc.) + that could break the single-quoted shell word regardless of the remote + shell, independent of the CLI validation. */ + if ((unsigned char)*p < 0x20 || (unsigned char)*p == 0x7f) + return NULL; + if (*p == '\'') + q++; + len++; + } + if (len > SIZE_MAX - q * 3 || len + q * 3 + 3 > SIZE_MAX - command_len) + return NULL; + command_len += len + q * 3 + 3; + } + command_len += 1; /* NUL */ + + char* command = malloc(command_len); + if (!command) + return NULL; + char* out = command; + *out++ = '\''; + for (const char* p = path; *p; p++) { + if (*p == '\'') { + memcpy(out, "'\\''", 4); + out += 4; + } else { + *out++ = *p; + } + } + *out++ = '\''; + memcpy(out, suffix, suffix_len + 1); + out += suffix_len; + for (int i = 0; i < remote_option_count; i++) { + const char* opt = remote_options[i]; + *out++ = ' '; + *out++ = '\''; + for (const char* p = opt; *p; p++) { + if (*p == '\'') { + memcpy(out, "'\\''", 4); + out += 4; + } else { + *out++ = *p; + } + } + *out++ = '\''; + } + *out = '\0'; + return command; +} + +/* A heap-owned, NULL-terminated argv whose every string is separately malloc'd + * (str_dup'd) so a caller can free arbitrary slots, including argv[0]. */ +char** ssh_build_client_argv(const char* rsh_command, int port, const char* userhost, + const char* remote_command) { + const char* rsh = (rsh_command && *rsh_command) ? rsh_command : "ssh"; + + /* Whitespace-split the remote-shell command into the leading argv words so + * "-e 'ssh -p 2222'" (or "--rsh=ssh -p 2222") works like rsync's rsh. A + * blank command falls back to the default "ssh". */ + char* copy = str_dup(rsh); + if (!copy) + return NULL; + char* save = NULL; + int nwords = 0; + char** words = NULL; + for (char* tok = strtok_r(copy, " \t", &save); tok; tok = strtok_r(NULL, " \t", &save)) { + char** grown = realloc(words, (size_t)(nwords + 1) * sizeof(char*)); + if (!grown) { + for (int i = 0; i < nwords; i++) + free(words[i]); + free(words); + free(copy); + return NULL; + } + words = grown; + words[nwords] = str_dup(tok); + if (!words[nwords]) { + for (int i = 0; i < nwords; i++) + free(words[i]); + free(words); + free(copy); + return NULL; + } + nwords++; + } + free(copy); + if (nwords == 0) { + words = malloc(sizeof(char*)); + if (!words) + return NULL; + words[0] = str_dup("ssh"); + if (!words[0]) { + free(words); + return NULL; + } + nwords = 1; + } + + /* Fixed tail: three -o pairs (6) + optional -p/value (2) + user@host + + * remote command + terminating NULL. */ + int port_extra = (port > 0 && port != 22) ? 2 : 0; + size_t total = (size_t)nwords + 6 + (size_t)port_extra + 3; + char** argv = calloc(total, sizeof(char*)); + if (!argv) { + for (int i = 0; i < nwords; i++) + free(words[i]); + free(words); + return NULL; + } + int ac = 0; + for (int i = 0; i < nwords; i++) + argv[ac++] = words[i]; + free(words); + + char* tail[] = {"-o", "Compression=no", + "-o", "ControlMaster=auto", + "-o", "ControlPath=~/.cache/fastsync-%r@%h:%p"}; + for (size_t i = 0; i < sizeof(tail) / sizeof(tail[0]); i++) { + argv[ac] = str_dup(tail[i]); + if (!argv[ac]) + goto fail_argv; + ac++; + } + if (port_extra) { + char port_str[16]; + snprintf(port_str, sizeof(port_str), "%d", port); + argv[ac] = str_dup("-p"); + if (!argv[ac]) + goto fail_argv; + ac++; + argv[ac] = str_dup(port_str); + if (!argv[ac]) + goto fail_argv; + ac++; + } + argv[ac] = str_dup(userhost); + if (!argv[ac]) + goto fail_argv; + ac++; + argv[ac] = str_dup(remote_command); + if (!argv[ac]) + goto fail_argv; + ac++; + argv[ac] = NULL; + return argv; + +fail_argv: + for (int i = 0; i < ac; i++) + free(argv[i]); + free(argv); + return NULL; +} + +void ssh_free_client_argv(char** argv) { + if (!argv) + return; + for (int i = 0; argv[i]; i++) + free(argv[i]); + free(argv); +} + +Client* client_connect_ssh(const char* destination, int port, const char* server_path, + bool old_args, const char* rsh_command, bool blocking_io, + char* const* remote_options, int remote_option_count) { RemoteDest r; if (parse_remote_dest(destination, &r) != 0) { - fprintf(stderr, "Invalid remote destination: %s\n", destination); + char* escaped = output_escape(destination, false); + fprintf(stderr, "Invalid remote destination: %s\n", escaped ? escaped : ""); + free(escaped); return NULL; } int sv[2]; if (socketpair(AF_UNIX, SOCK_STREAM, 0, sv) < 0) { - perror("socketpair failed"); + log_perror("socketpair failed"); remote_dest_destroy(&r); return NULL; } @@ -87,9 +303,20 @@ Client* client_connect_ssh(const char* destination, int port, const char* server setsockopt(sv[1], SOL_SOCKET, SO_SNDBUF, &buf_size, sizeof(buf_size)); setsockopt(sv[1], SOL_SOCKET, SO_RCVBUF, &buf_size, sizeof(buf_size)); + /* By default the SSH transport socket gets the same read/write timeout as + * the TCP transport so a wedged remote shell cannot hang forever. With + * --blocking-io the timeouts are skipped and the socket blocks naturally. */ + if (!blocking_io) { + struct timeval tv; + tv.tv_sec = tcp_get_timeout_sec(); + tv.tv_usec = 0; + setsockopt(sv[0], SOL_SOCKET, SO_RCVTIMEO, &tv, sizeof(tv)); + setsockopt(sv[0], SOL_SOCKET, SO_SNDTIMEO, &tv, sizeof(tv)); + } + int exec_pipe[2]; if (pipe(exec_pipe) < 0) { - perror("pipe failed"); + log_perror("pipe failed"); close(sv[0]); close(sv[1]); remote_dest_destroy(&r); @@ -98,7 +325,7 @@ Client* client_connect_ssh(const char* destination, int port, const char* server pid_t pid = fork(); if (pid < 0) { - perror("fork failed"); + log_perror("fork failed"); close(sv[0]); close(sv[1]); close(exec_pipe[0]); @@ -110,11 +337,12 @@ Client* client_connect_ssh(const char* destination, int port, const char* server if (pid == 0) { close(sv[0]); close(exec_pipe[0]); - fcntl(exec_pipe[1], F_SETFD, FD_CLOEXEC); - if (sv[1] != STDIN_FILENO) - dup2(sv[1], STDIN_FILENO); - if (sv[1] != STDOUT_FILENO) - dup2(sv[1], STDOUT_FILENO); + if (fcntl(exec_pipe[1], F_SETFD, FD_CLOEXEC) < 0) + ssh_child_setup_failed(exec_pipe[1]); + if (sv[1] != STDIN_FILENO && dup2(sv[1], STDIN_FILENO) < 0) + ssh_child_setup_failed(exec_pipe[1]); + if (sv[1] != STDOUT_FILENO && dup2(sv[1], STDOUT_FILENO) < 0) + ssh_child_setup_failed(exec_pipe[1]); if (sv[1] > 1) close(sv[1]); @@ -125,36 +353,25 @@ Client* client_connect_ssh(const char* destination, int port, const char* server ssh_user_len = strlen(r.host) + 1; char* ssh_user = malloc(ssh_user_len); if (!ssh_user) - _exit(1); + ssh_child_setup_failed(exec_pipe[1]); if (r.user && r.user[0] != '\0') snprintf(ssh_user, ssh_user_len, "%s@%s", r.user, r.host); else snprintf(ssh_user, ssh_user_len, "%s", r.host); - char* ssh_argv[16]; - int ac = 0; - char port_str[16]; - ssh_argv[ac++] = "ssh"; - ssh_argv[ac++] = "-o"; - ssh_argv[ac++] = "Compression=no"; - ssh_argv[ac++] = "-o"; - ssh_argv[ac++] = "ControlMaster=auto"; - ssh_argv[ac++] = "-o"; - ssh_argv[ac++] = "ControlPath=~/.cache/fastsync-%r@%h:%p"; - if (port > 0 && port != 22) { - ssh_argv[ac++] = "-p"; - snprintf(port_str, sizeof(port_str), "%d", port); - ssh_argv[ac++] = port_str; - } - ssh_argv[ac++] = ssh_user; - ssh_argv[ac++] = (char*)(server_path ? server_path : "fastsync-server"); - ssh_argv[ac++] = "--stdio"; - ssh_argv[ac] = NULL; - execvp("ssh", ssh_argv); - perror("exec of ssh failed"); - ssize_t wret = write(exec_pipe[1], "x", 1); - (void)wret; - _exit(1); + char* remote_command = + ssh_build_remote_command(server_path, old_args, remote_options, remote_option_count); + if (!remote_command) + ssh_child_setup_failed(exec_pipe[1]); + char** ssh_argv = ssh_build_client_argv(rsh_command, port, ssh_user, remote_command); + free(ssh_user); + free(remote_command); + if (!ssh_argv) + ssh_child_setup_failed(exec_pipe[1]); + execvp(ssh_argv[0], ssh_argv); + log_perror("exec of remote shell failed"); + ssh_free_client_argv(ssh_argv); + ssh_child_setup_failed(exec_pipe[1]); } close(sv[1]); @@ -164,12 +381,15 @@ Client* client_connect_ssh(const char* destination, int port, const char* server ssize_t n = read(exec_pipe[0], &exec_status, 1); close(exec_pipe[0]); - if (n > 0) { + if (n != 0) { close(sv[0]); waitpid(pid, NULL, 0); remote_dest_destroy(&r); + const char* path = server_path ? server_path : "fastsync-server"; + char* escaped = output_escape(path, false); fprintf(stderr, "Error: could not launch '%s --stdio' on remote\n", - server_path ? server_path : "fastsync-server"); + escaped ? escaped : ""); + free(escaped); return NULL; } @@ -182,7 +402,7 @@ Client* client_connect_ssh(const char* destination, int port, const char* server return NULL; } client->file_descriptor = sv[0]; - client->address.sin_family = AF_UNIX; + client->address.ss_family = AF_UNIX; client->address_length = 0; client->ssh_child_pid = pid; client->ssl = NULL; diff --git a/src/shared/transport_ssh.h b/src/shared/transport_ssh.h index 315f37d..08908ce 100644 --- a/src/shared/transport_ssh.h +++ b/src/shared/transport_ssh.h @@ -3,6 +3,26 @@ #include "transport_tcp.h" -Client* client_connect_ssh(const char* destination, int port, const char* server_path); +Client* client_connect_ssh(const char* destination, int port, const char* server_path, + bool old_args, const char* rsh_command, bool blocking_io, + char* const* remote_options, int remote_option_count); +/* Build the escaped remote-shell command string (the server program path always + * quoted as one remote-shell word, followed by ` --stdio` and each + * --remote-option value appended as an individually single-quoted shell word). + * `old_args` is accepted for CLI/ABI compatibility but no longer disables + * quoting: the path is always escaped so a metacharacter-bearing + * --rsync-path can never be interpreted by the remote shell. Every + * --remote-option value is individually escaped with the '\'' sequence and + * values with empty/control characters are rejected at the CLI parse layer. */ +char* ssh_build_remote_command(const char* server_path, bool old_args, char* const* remote_options, + int remote_option_count); +/* Build the NULL-terminated child argv for the remote-shell client (argv[0] is + * the exec/execvp program). rsh_command is whitespace-split into leading argv + * words (NULL or "" selects the default "ssh"); the standard -o family, the + * optional -p port, the user@host and the remote command are appended. Every + * string (including argv[0]) is heap-owned; free with ssh_free_client_argv. */ +char** ssh_build_client_argv(const char* rsh_command, int port, const char* userhost, + const char* remote_command); +void ssh_free_client_argv(char** argv); #endif diff --git a/src/shared/transport_tcp.c b/src/shared/transport_tcp.c index 6aec063..b26cb8e 100644 --- a/src/shared/transport_tcp.c +++ b/src/shared/transport_tcp.c @@ -1,8 +1,12 @@ #include "transport_tcp.h" #include "log.h" #include "protocol.h" +#include "utils.h" #include #include +#include +#include +#include #include #include #include @@ -14,6 +18,8 @@ static volatile sig_atomic_t g_active_connections = 0; +static void tcp_apply_socket_timeout(int fd); + static void sigchld_handler(int sig) { (void)sig; int saved_errno = errno; @@ -24,47 +30,91 @@ static void sigchld_handler(int sig) { errno = saved_errno; } -Server* server_create(int port) { +/* Map a listen socket's address to its numeric port for logging, independent + * of whether it is an IPv4 or IPv6 sockaddr. */ +static unsigned short server_address_port(const struct sockaddr_storage* addr) { + if (addr->ss_family == AF_INET6) + return ntohs(((const struct sockaddr_in6*)addr)->sin6_port); + if (addr->ss_family == AF_INET) + return ntohs(((const struct sockaddr_in*)addr)->sin_port); + return 0; +} + +Server* server_create_ex(int port, const ServerBindOptions* bind_opts) { Server* server = (Server*)malloc(sizeof(Server)); if (server == NULL) { - perror("Could not allocate space for Server"); + log_perror("Could not allocate space for Server"); return NULL; } - int file_descriptor = socket(AF_INET, SOCK_STREAM, 0); + /* Effective address family. preserve the historical default (IPv4 wildcard) + * when neither --address nor -4/-6 were given. */ + int family = (bind_opts && bind_opts->family != AF_UNSPEC) ? bind_opts->family : AF_INET; + const char* bind_address = bind_opts ? bind_opts->bind_address : NULL; + + struct addrinfo hints; + memset(&hints, 0, sizeof(hints)); + hints.ai_family = family; + hints.ai_socktype = SOCK_STREAM; + hints.ai_protocol = IPPROTO_TCP; + hints.ai_flags = AI_PASSIVE; + + char port_str[16]; + snprintf(port_str, sizeof(port_str), "%d", port); + + struct addrinfo* result = NULL; + int err = getaddrinfo(bind_address, port_str, &hints, &result); + if (err != 0 || result == NULL) { + char* escaped = bind_address ? output_escape(bind_address, false) : NULL; + fprintf(stderr, "Could not resolve bind address %s (%s)\n", escaped ? escaped : "(wildcard)", + gai_strerror(err)); + free(escaped); + free(server); + return NULL; + } + + int file_descriptor = -1; + struct addrinfo* rp; + for (rp = result; rp != NULL; rp = rp->ai_next) { + file_descriptor = socket(rp->ai_family, rp->ai_socktype, rp->ai_protocol); + if (file_descriptor < 0) + continue; + int opt = 1; + if (setsockopt(file_descriptor, SOL_SOCKET, SO_REUSEADDR, &opt, sizeof(opt)) != 0) { + log_perror("Error setting a socket option!"); + close(file_descriptor); + file_descriptor = -1; + continue; + } + if (bind(file_descriptor, rp->ai_addr, (socklen_t)rp->ai_addrlen) == 0) { + memset(&server->address, 0, sizeof(server->address)); + memcpy(&server->address, rp->ai_addr, rp->ai_addrlen); + server->address_length = rp->ai_addrlen; + break; + } + log_perror("Could not bind server address"); + close(file_descriptor); + file_descriptor = -1; + } + freeaddrinfo(result); + if (file_descriptor < 0) { - perror("Could not create Socket!"); - free(server); - return NULL; - } - server->file_descriptor = file_descriptor; - int opt = 1; - if (setsockopt(server->file_descriptor, SOL_SOCKET, SO_REUSEADDR, &opt, sizeof(opt))) { - perror("Error setting a socket option!"); - close(server->file_descriptor); free(server); return NULL; } - server->address.sin_family = AF_INET; - server->address.sin_addr.s_addr = INADDR_ANY; - server->address.sin_port = htons(port); - server->address_length = sizeof(server->address); + server->file_descriptor = file_descriptor; server->ssl_ctx = NULL; server->max_connections = 100; server->active_connections = 0; - if (bind(server->file_descriptor, (struct sockaddr*)&server->address, server->address_length) < - 0) { - perror("Could not bind server"); - close(server->file_descriptor); - free(server); - return NULL; - } - return server; } +Server* server_create(int port) { + return server_create_ex(port, NULL); +} + void server_delete(Server** server) { if (server == NULL || *server == NULL) return; @@ -80,7 +130,7 @@ void server_delete(Server** server) { static void accept_loop(Server* server, void (*child_fn)(int, void*), void* child_ctx, const char* log_fmt) { if (listen(server->file_descriptor, SOMAXCONN) < 0) { - perror("Could not listen on port!"); + log_perror("Could not listen on port!"); return; } signal(SIGCHLD, sigchld_handler); @@ -89,9 +139,10 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil socklen_t client_len = sizeof(client_addr); int fd = accept(server->file_descriptor, (struct sockaddr*)&client_addr, &client_len); if (fd < 0) { - perror("Could not accept the connection"); + log_perror("Could not accept the connection"); continue; } + tcp_apply_socket_timeout(fd); if ((unsigned int)g_active_connections >= server->max_connections) { log_message(LOG_LEVEL_WARNING, "Max connections (%u) reached, rejecting", server->max_connections); @@ -121,7 +172,7 @@ static void plain_child_fn(int fd, void* ctx) { } bool server_listen(Server* server, void (*handler)(int file_descriptor)) { - log_message(LOG_LEVEL_INFO, "Start Listening on Port: %d", ntohs(server->address.sin_port)); + log_message(LOG_LEVEL_INFO, "Start Listening on Port: %d", server_address_port(&server->address)); struct plain_ctx ctx = {handler}; accept_loop(server, plain_child_fn, &ctx, "Received Connection"); return true; @@ -129,7 +180,8 @@ bool server_listen(Server* server, void (*handler)(int file_descriptor)) { void server_accept_loop(Server* server, void (*child_fn)(int, void*), void* child_ctx, const char* log_fmt) { - log_message(LOG_LEVEL_INFO, "Start TLS Listening on Port: %d", ntohs(server->address.sin_port)); + log_message(LOG_LEVEL_INFO, "Start TLS Listening on Port: %d", + server_address_port(&server->address)); accept_loop(server, child_fn, child_ctx, log_fmt); } @@ -143,6 +195,14 @@ void tcp_set_timeouts(int timeout_sec, int contimeout_sec) { g_contimeout_sec = contimeout_sec; } +int tcp_get_contimeout_sec(void) { + return g_contimeout_sec; +} + +int tcp_get_timeout_sec(void) { + return g_timeout_sec; +} + static void tcp_apply_socket_timeout(int fd) { struct timeval tv; tv.tv_sec = g_timeout_sec; @@ -152,19 +212,12 @@ static void tcp_apply_socket_timeout(int fd) { } Client* client_create() { - int file_descriptor = socket(AF_INET, SOCK_STREAM, 0); - if (file_descriptor < 0) { - perror("Could not create Socket!"); - return NULL; - } - Client* client = (Client*)malloc(sizeof(Client)); if (client == NULL) { - close(file_descriptor); return NULL; } - client->file_descriptor = file_descriptor; - client->address.sin_family = AF_INET; + client->file_descriptor = -1; + memset(&client->address, 0, sizeof(client->address)); client->address_length = sizeof(client->address); client->ssh_child_pid = -1; client->ssl = NULL; @@ -172,28 +225,197 @@ Client* client_create() { return client; } -bool client_connect(Client* client, char* host, int port) { - client->address.sin_port = htons(port); - client->address.sin_family = AF_INET; - client->address_length = sizeof(client->address); +int tcp_connect_family(bool ipv4, bool ipv6) { + if (ipv4) + return AF_INET; + if (ipv6) + return AF_INET6; + return AF_UNSPEC; +} - if (inet_pton(AF_INET, host, &client->address.sin_addr) <= 0) { - perror("Could not convert host address!"); +/* The socket-option apply layer maps an allowlist SockOptId to the concrete + * level/optname pair and applies it with the correct (int) value type. The + * allowlist bounds what can ever reach this point, so the id-to-name mapping + * is total for every SOCKOPT_* value. */ +static int tcp_sockopt_level(SockOptId id) { + return id == SOCKOPT_TCP_NODELAY ? IPPROTO_TCP : SOL_SOCKET; +} + +static int tcp_sockopt_name(SockOptId id) { + switch (id) { + case SOCKOPT_TCP_NODELAY: + return TCP_NODELAY; + case SOCKOPT_SO_KEEPALIVE: + return SO_KEEPALIVE; + case SOCKOPT_SO_RCVBUF: + return SO_RCVBUF; + case SOCKOPT_SO_SNDBUF: + return SO_SNDBUF; + case SOCKOPT_SO_REUSEADDR: + return SO_REUSEADDR; + default: + return -1; + } +} + +static bool tcp_apply_sockopts(int fd, const SockOptEntry* sockopts, int sockopt_count) { + for (int i = 0; i < sockopt_count; i++) { + int name = tcp_sockopt_name(sockopts[i].id); + if (name < 0) { /* unreachable for a validated allowlist, but stay defensive */ + log_message(LOG_LEVEL_ERROR, "Unsupported socket option requested"); + return false; + } + int value = sockopts[i].value; + if (setsockopt(fd, tcp_sockopt_level(sockopts[i].id), name, &value, sizeof(value)) != 0) { + log_perror("Could not apply socket option"); + return false; + } + } + return true; +} + +/* Resolve an explicit --address source/bind address into a sockaddr once, so + * the per-candidate connect loop can bind() the outgoing socket to it. The + * family follows the same -4/-6 hints as the destination resolution, so a + * forced family selects a matching local address; returns 0 on success. */ +static int resolve_bind_address(const char* addr, int family, struct sockaddr_storage* out, + socklen_t* out_len, int* out_family) { + struct addrinfo hints; + memset(&hints, 0, sizeof(hints)); + hints.ai_family = family; /* AF_UNSPEC when no -4/-6 */ + hints.ai_socktype = SOCK_STREAM; + hints.ai_protocol = IPPROTO_TCP; + + struct addrinfo* result = NULL; + int err = getaddrinfo(addr, NULL, &hints, &result); + if (err != 0 || result == NULL) { + char* escaped = output_escape(addr, false); + fprintf(stderr, "Could not resolve --address %s (%s)\n", + escaped ? escaped : "", gai_strerror(err)); + free(escaped); + return -1; + } + struct addrinfo* rp; + bool found = false; + for (rp = result; rp != NULL; rp = rp->ai_next) { + if (family != AF_UNSPEC && rp->ai_family != family) + continue; + memcpy(out, rp->ai_addr, rp->ai_addrlen); + *out_len = (socklen_t)rp->ai_addrlen; + *out_family = rp->ai_family; + found = true; + break; + } + freeaddrinfo(result); + return found ? 0 : -1; +} + +bool tcp_connect_socket_ex(Client* client, const char* host, int port, + const TcpConnectOptions* opts) { + struct addrinfo hints; + struct addrinfo* result; + memset(&hints, 0, sizeof(hints)); + /* TcpConnectOptions.family already encodes -4/-6 (or AF_UNSPEC); feed it + * straight into the getaddrinfo hints so the destination resolution is + * (optionally) pinned to one address family. */ + hints.ai_family = opts ? opts->family : AF_UNSPEC; + hints.ai_socktype = SOCK_STREAM; + hints.ai_protocol = IPPROTO_TCP; + + char port_str[16]; + snprintf(port_str, sizeof(port_str), "%d", port); + + int err = getaddrinfo(host, port_str, &hints, &result); + if (err != 0 || result == NULL) { + char* escaped_host = output_escape(host, false); + fprintf(stderr, "Could not resolve host: %s (%s)\n", + escaped_host ? escaped_host : "", gai_strerror(err)); + free(escaped_host); return false; } - struct timeval ct; - ct.tv_sec = g_contimeout_sec; - ct.tv_usec = 0; - setsockopt(client->file_descriptor, SOL_SOCKET, SO_RCVTIMEO, &ct, sizeof(ct)); - setsockopt(client->file_descriptor, SOL_SOCKET, SO_SNDTIMEO, &ct, sizeof(ct)); + /* Resolve the optional --address source address once up front. */ + struct sockaddr_storage bind_addr; + socklen_t bind_addr_len = 0; + int bind_addr_family = 0; + if (opts && opts->bind_address) { + if (resolve_bind_address(opts->bind_address, hints.ai_family, &bind_addr, &bind_addr_len, + &bind_addr_family) != 0) { + freeaddrinfo(result); + return false; + } + } - if (connect(client->file_descriptor, (struct sockaddr*)&client->address, client->address_length) < - 0) { - perror("Could not connect to Server!"); + struct addrinfo* rp; + bool connected = false; + for (rp = result; rp != NULL; rp = rp->ai_next) { + if (client->file_descriptor >= 0) + close(client->file_descriptor); + + client->file_descriptor = socket(rp->ai_family, rp->ai_socktype, rp->ai_protocol); + if (client->file_descriptor < 0) + continue; + + if (opts && opts->sockopt_count > 0 && + !tcp_apply_sockopts(client->file_descriptor, opts->sockopts, opts->sockopt_count)) { + close(client->file_descriptor); + client->file_descriptor = -1; + break; + } + + struct timeval ct; + ct.tv_sec = g_contimeout_sec; + ct.tv_usec = 0; + setsockopt(client->file_descriptor, SOL_SOCKET, SO_RCVTIMEO, &ct, sizeof(ct)); + setsockopt(client->file_descriptor, SOL_SOCKET, SO_SNDTIMEO, &ct, sizeof(ct)); + + if (bind_addr_family != 0) { + if (rp->ai_family != bind_addr_family) { + close(client->file_descriptor); + client->file_descriptor = -1; + continue; + } + if (bind(client->file_descriptor, (struct sockaddr*)&bind_addr, bind_addr_len) != 0) { + log_perror("Could not bind outgoing socket to --address"); + close(client->file_descriptor); + client->file_descriptor = -1; + break; + } + } + + memcpy(&client->address, rp->ai_addr, rp->ai_addrlen); + client->address_length = rp->ai_addrlen; + + if (connect(client->file_descriptor, (struct sockaddr*)&client->address, + client->address_length) == 0) { + connected = true; + break; + } + } + freeaddrinfo(result); + + if (!connected) { + log_perror("Could not connect to Server!"); return false; } + return true; +} + +bool tcp_connect_socket(Client* client, const char* host, int port) { + return tcp_connect_socket_ex(client, host, port, NULL); +} + +bool client_connect_ex(Client* client, const char* host, int port, const TcpConnectOptions* opts) { + if (!tcp_connect_socket_ex(client, host, port, opts)) + return false; + tcp_apply_socket_timeout(client->file_descriptor); + return true; +} + +bool client_connect(Client* client, const char* host, int port) { + if (!tcp_connect_socket_ex(client, host, port, NULL)) + return false; tcp_apply_socket_timeout(client->file_descriptor); return true; } @@ -205,7 +427,10 @@ void client_disconnect(Client* client) { client->ssl = NULL; io_set_ssl(NULL); } - close(client->file_descriptor); + if (client->file_descriptor >= 0) { + close(client->file_descriptor); + client->file_descriptor = -1; + } if (client->ssh_child_pid > 0) { int status; waitpid(client->ssh_child_pid, &status, 0); diff --git a/src/shared/transport_tcp.h b/src/shared/transport_tcp.h index 71b03a2..b5937e6 100644 --- a/src/shared/transport_tcp.h +++ b/src/shared/transport_tcp.h @@ -1,12 +1,14 @@ #ifndef TRANSPORT_TCP_H #define TRANSPORT_TCP_H +#include "config.h" +#include #include #include #include typedef struct Server { - struct sockaddr_in address; + struct sockaddr_storage address; unsigned int address_length; int file_descriptor; void* ssl_ctx; @@ -15,7 +17,7 @@ typedef struct Server { } Server; typedef struct Client { - struct sockaddr_in address; + struct sockaddr_storage address; unsigned int address_length; int file_descriptor; pid_t ssh_child_pid; @@ -23,15 +25,46 @@ typedef struct Client { void* ssl_ctx; } Client; +/* Options controlling the server's listening bind (/--address, -4/-6). When + * bind_address is NULL and family is AF_UNSPEC the existing default is used: + * an IPv4 wildcard (INADDR_ANY). */ +typedef struct { + const char* bind_address; /* explicit address to bind, or NULL for wildcard */ + int family; /* AF_INET / AF_INET6, or AF_UNSPEC to use the default */ +} ServerBindOptions; + +/* Options controlling an outgoing client connect (--address, -4/-6, + * --sockopts). All fields are client/connection-level and never cross the + * wire config frame. */ +typedef struct { + const char* bind_address; /* --address: local source address to bind, or NULL */ + int family; /* AF_INET / AF_INET6 / AF_UNSPEC (from -4 / -6) */ + const SockOptEntry* sockopts; /* --sockopts allowlist entries */ + int sockopt_count; +} TcpConnectOptions; + +Server* server_create_ex(int port, const ServerBindOptions* bind_opts); Server* server_create(int port); bool server_listen(Server* server, void (*handler)(int file_descriptor)); void server_accept_loop(Server* server, void (*child_fn)(int, void*), void* child_ctx, const char* log_fmt); void server_delete(Server** server); Client* client_create(); -bool client_connect(Client* client, char* host, int port); +bool client_connect_ex(Client* client, const char* host, int port, const TcpConnectOptions* opts); +bool client_connect(Client* client, const char* host, int port); +bool tcp_connect_socket_ex(Client* client, const char* host, int port, + const TcpConnectOptions* opts); +bool tcp_connect_socket(Client* client, const char* host, int port); void client_disconnect(Client* client); void client_delete(Client* client); void tcp_set_timeouts(int timeout_sec, int contimeout_sec); +int tcp_get_contimeout_sec(void); +int tcp_get_timeout_sec(void); + +/* Resolve -4/-6 flags to a getaddrinfo ai_family value. ipv4 wins over ipv6; + * when neither is set it returns AF_UNSPEC. 0 means "no preference" and is + * therefore never returned; callers that need the "no explicit flag" sentinel + * compare the flags directly. */ +int tcp_connect_family(bool ipv4, bool ipv6); #endif diff --git a/src/shared/transport_tls.c b/src/shared/transport_tls.c index f245a68..85bb5a1 100644 --- a/src/shared/transport_tls.c +++ b/src/shared/transport_tls.c @@ -2,6 +2,7 @@ #include "log.h" #include "protocol.h" #include "transport_tcp.h" +#include "utils.h" #include #include #include @@ -10,7 +11,9 @@ #include #include #include +#include #include +#include #include bool tls_global_init(void) { @@ -33,6 +36,10 @@ static void log_ssl_errors(void) { static SSL_CTX* create_ssl_ctx(bool is_server, const char* cert, const char* key, const char* ca_path) { + if (!is_server && !ca_path) { + log_message(LOG_LEVEL_ERROR, "TLS clients require a CA certificate path"); + return NULL; + } const SSL_METHOD* method = is_server ? TLS_server_method() : TLS_client_method(); SSL_CTX* ctx = SSL_CTX_new(method); if (!ctx) { @@ -41,17 +48,46 @@ static SSL_CTX* create_ssl_ctx(bool is_server, const char* cert, const char* key return NULL; } - SSL_CTX_set_min_proto_version(ctx, TLS1_2_VERSION); + /* Harden the context: never negotiate TLS compression (the CRIME attack + * vector) and never honour a post-handshake renegotiation request. + * SSL_OP_NO_RENEGOTIATION is only available from OpenSSL 1.1.1, so it is + * guarded to keep older headers building. */ + SSL_CTX_set_options(ctx, SSL_OP_NO_COMPRESSION); +#ifdef SSL_OP_NO_RENEGOTIATION + SSL_CTX_set_options(ctx, SSL_OP_NO_RENEGOTIATION); +#endif + + if (SSL_CTX_set_min_proto_version(ctx, TLS1_2_VERSION) != 1) { + SSL_CTX_free(ctx); + return NULL; + } + if (SSL_CTX_set_cipher_list(ctx, "HIGH:!aNULL:!eNULL:!MD5:!RC4:!3DES") != 1) { + SSL_CTX_free(ctx); + return NULL; + } if (cert && key) { + struct stat key_stat; + if (stat(key, &key_stat) != 0 || !S_ISREG(key_stat.st_mode) || key_stat.st_uid != geteuid() || + (key_stat.st_mode & (S_IRGRP | S_IWGRP | S_IROTH | S_IWOTH))) { + log_message(LOG_LEVEL_ERROR, "TLS private key must be owned by the current user and private"); + SSL_CTX_free(ctx); + return NULL; + } if (SSL_CTX_use_certificate_file(ctx, cert, SSL_FILETYPE_PEM) <= 0) { - log_message(LOG_LEVEL_ERROR, "Failed to load certificate: %s", cert); + char* escaped = output_escape(cert, false); + log_message(LOG_LEVEL_ERROR, "Failed to load certificate: %s", + escaped ? escaped : ""); + free(escaped); log_ssl_errors(); SSL_CTX_free(ctx); return NULL; } if (SSL_CTX_use_PrivateKey_file(ctx, key, SSL_FILETYPE_PEM) <= 0) { - log_message(LOG_LEVEL_ERROR, "Failed to load private key: %s", key); + char* escaped = output_escape(key, false); + log_message(LOG_LEVEL_ERROR, "Failed to load private key: %s", + escaped ? escaped : ""); + free(escaped); log_ssl_errors(); SSL_CTX_free(ctx); return NULL; @@ -65,12 +101,15 @@ static SSL_CTX* create_ssl_ctx(bool is_server, const char* cert, const char* key if (ca_path) { if (!SSL_CTX_load_verify_locations(ctx, ca_path, NULL)) { - log_message(LOG_LEVEL_ERROR, "Failed to load CA: %s", ca_path); + char* escaped = output_escape(ca_path, false); + log_message(LOG_LEVEL_ERROR, "Failed to load CA: %s", + escaped ? escaped : ""); + free(escaped); log_ssl_errors(); SSL_CTX_free(ctx); return NULL; } - SSL_CTX_set_verify(ctx, SSL_VERIFY_PEER, NULL); + SSL_CTX_set_verify(ctx, SSL_VERIFY_PEER | SSL_VERIFY_FAIL_IF_NO_PEER_CERT, NULL); SSL_CTX_set_verify_depth(ctx, 4); } else { SSL_CTX_set_verify(ctx, SSL_VERIFY_NONE, NULL); @@ -85,15 +124,22 @@ static SSL* wrap_fd_with_ssl(int fd, SSL_CTX* ctx, bool is_server, const char* h log_message(LOG_LEVEL_ERROR, "Failed to create SSL object"); return NULL; } - SSL_set_fd(ssl, fd); + if (SSL_set_fd(ssl, fd) != 1) { + SSL_free(ssl); + return NULL; + } // Enable hostname verification for client connections when a hostname is provided. // Must be done before SSL_connect to take effect during the handshake. if (!is_server && hostname) { - SSL_set1_host(ssl, hostname); + if (SSL_set1_host(ssl, hostname) != 1) { + SSL_free(ssl); + return NULL; + } } // Retry SSL_accept/SSL_connect on WANT_READ/WANT_WRITE (non-blocking handshake) + time_t deadline = time(NULL) + (is_server ? tcp_get_timeout_sec() : tcp_get_contimeout_sec()); int ret; do { if (is_server) @@ -103,7 +149,8 @@ static SSL* wrap_fd_with_ssl(int fd, SSL_CTX* ctx, bool is_server, const char* h if (ret <= 0) { int ssl_err = SSL_get_error(ssl, ret); - if (ssl_err == SSL_ERROR_WANT_READ || ssl_err == SSL_ERROR_WANT_WRITE) + if ((ssl_err == SSL_ERROR_WANT_READ || ssl_err == SSL_ERROR_WANT_WRITE) && + time(NULL) < deadline) continue; log_message(LOG_LEVEL_ERROR, "SSL %s failed", is_server ? "accept" : "connect"); log_ssl_errors(); @@ -131,8 +178,10 @@ struct tls_child_ctx { static void tls_child_fn(int fd, void* arg) { struct tls_child_ctx* ctx = (struct tls_child_ctx*)arg; SSL* ssl = wrap_fd_with_ssl(fd, ctx->ssl_ctx, true, NULL); - if (!ssl) + if (!ssl) { + io_set_ssl(NULL); return; + } io_set_ssl(ssl); ctx->handler(fd); SSL_shutdown(ssl); @@ -146,22 +195,22 @@ bool server_listen_tls(Server* server, void (*handler)(int file_descriptor)) { return true; } -bool client_connect_tls(Client* client, char* host, int port, const char* cert_path, - const char* key_path, const char* ca_path) { - client->address.sin_port = htons(port); - if (inet_pton(AF_INET, host, &client->address.sin_addr) <= 0) { - perror("Could not convert host address!"); - return false; - } - if (connect(client->file_descriptor, (struct sockaddr*)&client->address, client->address_length) < - 0) { - perror("Could not connect to Server!"); +bool client_connect_tls_ex(Client* client, const char* host, int port, const char* cert_path, + const char* key_path, const char* ca_path, + const TcpConnectOptions* opts) { + if (!tcp_connect_socket_ex(client, host, port, opts)) { + if (client->file_descriptor >= 0) + close(client->file_descriptor); + client->file_descriptor = -1; return false; } SSL_CTX* ctx = create_ssl_ctx(false, cert_path, key_path, ca_path); - if (!ctx) + if (!ctx) { + close(client->file_descriptor); + client->file_descriptor = -1; return false; + } client->ssl_ctx = ctx; // Pass the server hostname for TLS hostname verification (SSL_set1_host @@ -171,6 +220,8 @@ bool client_connect_tls(Client* client, char* host, int port, const char* cert_p if (!ssl) { SSL_CTX_free(ctx); client->ssl_ctx = NULL; + close(client->file_descriptor); + client->file_descriptor = -1; return false; } @@ -178,3 +229,8 @@ bool client_connect_tls(Client* client, char* host, int port, const char* cert_p io_set_ssl(ssl); return true; } + +bool client_connect_tls(Client* client, const char* host, int port, const char* cert_path, + const char* key_path, const char* ca_path) { + return client_connect_tls_ex(client, host, port, cert_path, key_path, ca_path, NULL); +} diff --git a/src/shared/transport_tls.h b/src/shared/transport_tls.h index b9d59e7..fd1f9d5 100644 --- a/src/shared/transport_tls.h +++ b/src/shared/transport_tls.h @@ -9,7 +9,10 @@ bool tls_global_init(void); bool server_create_tls(Server* server, const char* cert_path, const char* key_path, const char* ca_path); bool server_listen_tls(Server* server, void (*handler)(int file_descriptor)); -bool client_connect_tls(Client* client, char* host, int port, const char* cert_path, +bool client_connect_tls_ex(Client* client, const char* host, int port, const char* cert_path, + const char* key_path, const char* ca_path, + const TcpConnectOptions* opts); +bool client_connect_tls(Client* client, const char* host, int port, const char* cert_path, const char* key_path, const char* ca_path); #endif diff --git a/src/shared/utils.c b/src/shared/utils.c index 755d695..16ea671 100644 --- a/src/shared/utils.c +++ b/src/shared/utils.c @@ -1,67 +1,139 @@ #include "utils.h" #include "array_list.h" -#include "libgen.h" +#include "log.h" +#include #include #include +#include +#include #include #include #include +#include +#include #include #include -bool mkdir_r(const char* path) { - char* path_duplicate = malloc(strlen(path) + 1); - if (!path_duplicate) - return false; - strcpy(path_duplicate, path); - char* path_current = (char*)malloc((strlen(path) + 2) * sizeof(char)); - if (!path_current) { - free(path_duplicate); +static int authorized_root_fd = -1; +static char* authorized_root_path; + +bool utils_set_authorized_root(int fd, const char* canonical_path) { + char* path_copy = canonical_path ? str_dup(canonical_path) : NULL; + if (canonical_path && !path_copy) { + authorized_root_fd = -1; + free(authorized_root_path); + authorized_root_path = NULL; return false; } - char* path_current_position = path_current; - if (path[0] == '/') { - strcpy(path_current, "/"); - path_current_position += 1; - } else { - path_current[0] = '\0'; + authorized_root_fd = fd; + free(authorized_root_path); + authorized_root_path = path_copy; + return true; +} + +void utils_set_authorized_root_fd(int fd) { + (void)utils_set_authorized_root(fd, NULL); +} + +static bool path_is_within_root(const char* root, const char* path) { + size_t root_len = strlen(root); + return strncmp(root, path, root_len) == 0 && (path[root_len] == '\0' || path[root_len] == '/'); +} + +static int open_authorized_destination(const char* dest_root) { + if (authorized_root_fd < 0 || !authorized_root_path || !dest_root || + !path_is_within_root(authorized_root_path, dest_root)) + return -1; + + int dirfd = dup(authorized_root_fd); + if (dirfd < 0) + return -1; + + const char* relative_path = dest_root + strlen(authorized_root_path); + while (*relative_path == '/') + relative_path++; + char* relative = str_dup(*relative_path ? relative_path : "."); + if (!relative) { + close(dirfd); + return -1; } - const char* delimiter = "/"; - char* saveptr; - const char* part = strtok_r(path_duplicate, delimiter, &saveptr); - bool ok = true; - while (part != NULL) { - strcpy(path_current_position, part); - path_current_position += strlen(part) * sizeof(char); - strcpy(path_current_position, "/"); - path_current_position += sizeof(char); - struct stat st; - if (stat(path_current, &st) != 0) { - if (mkdir(path_current, 0755) != 0) { - perror("Could not create directory"); - ok = false; - break; - } + + char* saveptr = NULL; + char* component = strtok_r(relative, "/", &saveptr); + while (component) { + if (strcmp(component, "..") == 0) { + free(relative); + close(dirfd); + return -1; } - part = strtok_r(NULL, delimiter, &saveptr); + if (strcmp(component, ".") == 0) { + component = strtok_r(NULL, "/", &saveptr); + continue; + } + int next = openat(dirfd, component, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + if (next < 0) { + free(relative); + close(dirfd); + return -1; + } + close(dirfd); + dirfd = next; + component = strtok_r(NULL, "/", &saveptr); } - free(path_duplicate); - free(path_current); - return ok; + + free(relative); + return dirfd; } char* str_dup(const char* string) { if (string == NULL) return NULL; - char* new_string = (char*)malloc(strlen(string) + 1); - strcpy(new_string, string); + size_t str_len = strlen(string); + char* new_string = (char*)malloc(str_len + 1); + if (new_string == NULL) + return NULL; + memcpy(new_string, string, str_len + 1); return new_string; } +char* output_escape(const char* string, bool eight_bit_output) { + if (!string) + return NULL; + size_t length = strlen(string); + if (length > (SIZE_MAX - 1) / 5) + return NULL; + char* escaped = malloc(length * 5 + 1); + if (!escaped) + return NULL; + size_t out = 0; + for (size_t i = 0; i < length; i++) { + unsigned char byte = (unsigned char)string[i]; + if ((byte >= 32 && byte <= 126) || (eight_bit_output && byte >= 128)) { + escaped[out++] = (char)byte; + } else { + escaped[out++] = '\\'; + escaped[out++] = '#'; + escaped[out++] = (char)('0' + ((byte >> 6) & 7)); + escaped[out++] = (char)('0' + ((byte >> 3) & 7)); + escaped[out++] = (char)('0' + (byte & 7)); + } + } + escaped[out] = '\0'; + return escaped; +} + +/* Match a glob pattern against a string. Supported wildcards: + * ? matches any single character except '/'. + * * matches any sequence of characters within one path component (no '/'). + * ** matches any sequence of characters, including '/' (cross-directory). + * slash-star-star-slash is treated as a cross-directory wildcard when it appears between + * literals. + */ bool glob_match(const char* pattern, const char* str) { while (*pattern) { if (*pattern == '*') { if (*(pattern + 1) == '*') { + /* globstar: match across directories */ pattern += 2; if (*pattern == '\0') return true; @@ -74,6 +146,7 @@ bool glob_match(const char* pattern, const char* str) { } return glob_match(pattern, str); } + /* single *: match within one path component */ pattern++; while (*str && *str != '/') { if (glob_match(pattern, str)) @@ -88,6 +161,7 @@ bool glob_match(const char* pattern, const char* str) { str++; } else { if (*pattern != *str) { + /* allow literal / ** / rest to match any number of directories */ if (*pattern == '/' && *(pattern + 1) == '*' && *(pattern + 2) == '*') { const char* rest = pattern + 3; if (*rest == '/') @@ -103,6 +177,25 @@ bool glob_match(const char* pattern, const char* str) { return *str == '\0'; } +bool format_human_bytes(unsigned long long bytes, char* buffer, size_t buffer_size) { + static const char* const units[] = {"B", "KB", "MB", "GB", "TB", "PB", "EB"}; + double value = (double)bytes; + size_t unit = 0; + int written; + + if (!buffer || buffer_size == 0) + return false; + while (value >= 1024.0 && unit < sizeof(units) / sizeof(units[0]) - 1) { + value /= 1024.0; + unit++; + } + if (unit == 0) + written = snprintf(buffer, buffer_size, "%llu %s", bytes, units[unit]); + else + written = snprintf(buffer, buffer_size, "%.1f %s", value, units[unit]); + return written >= 0 && (size_t)written < buffer_size; +} + static bool is_dir_in_manifest(const char* rel_path, ArrayList* manifest) { size_t len = strlen(rel_path); for (int i = 0; i < manifest->size; i++) { @@ -114,35 +207,202 @@ static bool is_dir_in_manifest(const char* rel_path, ArrayList* manifest) { return false; } -static void delete_extras_walk(const char* abs_path, const char* rel_path, ArrayList* manifest) { - DIR* dir = opendir(abs_path); - if (!dir) - return; - bool all_removed = true; +/* True when child_rel is, or lies below, a protected entry. A prefix "a" + therefore protects "a" and "a/b/c" but not "ab". Entries with top_level_only + set only protect DIRECT children of the receive root (at_root); nested + directories that share such a name stay ordinary destination content. */ +bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips, + int skip_count) { + for (int i = 0; i < skip_count; i++) { + if (skips[i].top_level_only && !at_root) + continue; + size_t prefix_len = strlen(skips[i].prefix); + if (strncmp(child_rel, skips[i].prefix, prefix_len) == 0 && + (child_rel[prefix_len] == '\0' || child_rel[prefix_len] == '/')) + return true; + } + return false; +} + +/* All-or-nothing max-delete needs to know BEFORE any unlink whether the run + would delete more than max_delete entries. This rehearsal pass walks the + destination with the same decisions as the delete pass but never touches the + filesystem: it counts every regular file the delete pass would unlink and + every directory it would rmdir (a directory is removed only once every entry + below it has been removed and nothing the walker leaves in place survives). + Entries the walker never removes (symlinks, manifest-listed files, protected + prefixes) mark the enclosing directory as surviving, exactly as they would + make a real rmdir fail with ENOTEMPTY. Stops early once *count reaches the + cap (sets *exceeds). Returns false on a traversal error. */ +static bool count_extras_fd(int dirfd, const char* rel_path, ArrayList* manifest, size_t cap, + size_t* count, bool* exceeds, const DeleteSkipEntry* skips, + int skip_count, bool* survives) { + /* openat(dirfd, ".") opens an independent file description: a dup() would + share dirfd's file offset, and a prior rehearsal pass must not have drained + this directory's stream before the delete pass reads it again. */ + int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + if (scanfd < 0) + return false; + DIR* dir = fdopendir(scanfd); + if (!dir) { + close(scanfd); + return false; + } + bool operation_ok = true; + bool local_survives = false; + bool at_root = rel_path[0] == '\0'; const struct dirent* entry; while ((entry = readdir(dir)) != NULL) { if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) continue; - char* child_abs = path_cat((char*)abs_path, entry->d_name); + if (*exceeds) + break; char* child_rel = path_cat((char*)rel_path, entry->d_name); + if (!child_rel) { + operation_ok = false; + continue; + } + if (path_under_skip_prefix(child_rel, at_root, skips, skip_count)) { + local_survives = true; + free(child_rel); + continue; + } struct stat st; - if (lstat(child_abs, &st) != 0) { - free(child_abs); + if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) { + if (errno != ENOENT) + operation_ok = false; + free(child_rel); + continue; + } + if (S_ISLNK(st.st_mode)) { + local_survives = true; + free(child_rel); + continue; + } + if (S_ISDIR(st.st_mode)) { + int childfd = openat(dirfd, entry->d_name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + bool child_ok = true; + bool child_survives = true; + if (childfd >= 0) { + child_ok = count_extras_fd(childfd, child_rel, manifest, cap, count, exceeds, skips, + skip_count, &child_survives); + close(childfd); + } else if (errno != ENOENT) { + operation_ok = false; + } + if (!child_ok) + operation_ok = false; + if (is_dir_in_manifest(child_rel, manifest)) { + /* A directory with kept content below it is never removed. */ + local_survives = true; + } else if (child_survives) { + /* The directory still holds entries the walker leaves in place, so an + rmdir would fail with ENOTEMPTY; the delete pass leaves it behind + rather than reporting an error (matching rsync). */ + local_survives = true; + } else { + if (*count >= cap) { + *exceeds = true; + } else { + (*count)++; + } + } + } else { + bool found = false; + for (int i = 0; i < manifest->size; i++) { + if (strcmp((char*)manifest->items[i], child_rel) == 0) { + found = true; + break; + } + } + if (!found) { + if (*count >= cap) { + *exceeds = true; + } else { + (*count)++; + } + } + } + free(child_rel); + } + closedir(dir); + *survives = local_survives; + return operation_ok; +} + +static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifest, + size_t max_delete, size_t* deleted_count, const DeleteSkipEntry* skips, + int skip_count) { + /* Independent file description (see count_extras_fd). */ + int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + if (scanfd < 0) + return false; + DIR* dir = fdopendir(scanfd); + if (!dir) { + close(scanfd); + return false; + } + bool operation_ok = true; + const struct dirent* entry; + while ((entry = readdir(dir)) != NULL) { + if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) + continue; + char* child_rel = path_cat((char*)rel_path, entry->d_name); + if (!child_rel) { + operation_ok = false; + continue; + } + /* A --delay-updates run keeps its staging directory as a direct child of + the receive root, and basis-dir snapshots live below it too. Their + contents are not manifest entries, so descending into them would delete + every staged / basis file as an "extra". Only the staging name (a + top-level-only prefix) and the basis prefixes are protected: a nested + destination directory that happens to be called .fastsync-stage is + ordinary content. */ + if (path_under_skip_prefix(child_rel, rel_path[0] == '\0', skips, skip_count)) { + free(child_rel); + continue; + } + struct stat st; + if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) { + if (errno != ENOENT) + operation_ok = false; free(child_rel); continue; } // Skip symlinks to prevent following them outside the destination tree if (S_ISLNK(st.st_mode)) { - free(child_abs); free(child_rel); continue; } if (S_ISDIR(st.st_mode)) { - delete_extras_walk(child_abs, child_rel, manifest); - // After recursion, try to remove the subdirectory if it's now empty. - // Ignore ENOENT: the recursive call may have already removed it. - if (rmdir(child_abs) != 0 && errno != ENOENT) { - all_removed = false; + int childfd = openat(dirfd, entry->d_name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + bool child_removed = false; + if (childfd >= 0) { + child_removed = delete_extras_fd(childfd, child_rel, manifest, max_delete, deleted_count, + skips, skip_count); + if (!child_removed) + operation_ok = false; + close(childfd); + } else if (errno != ENOENT) { + operation_ok = false; + } + if (child_removed && !is_dir_in_manifest(child_rel, manifest)) { + if (*deleted_count >= max_delete) { + operation_ok = false; + } else { + if (unlinkat(dirfd, entry->d_name, AT_REMOVEDIR) != 0) { + /* ENOENT: already gone (fine). ENOTEMPTY/EEXIST: the directory + still holds entries the walker leaves in place (a protected + excluded prefix, a kept file the manifest protects, a symlink); + rsync leaves such a directory behind, so this is not an error. + Only genuine I/O failures abort the deletion. */ + if (errno != ENOENT && errno != ENOTEMPTY && errno != EEXIST) + operation_ok = false; + } else { + (*deleted_count)++; + } + } } } else { // Check if relative path is in manifest @@ -154,33 +414,84 @@ static void delete_extras_walk(const char* abs_path, const char* rel_path, Array } } if (!found) { - unlink(child_abs); - fprintf(stderr, " Deleted: %s\n", child_rel); - } else { - all_removed = false; + if (*deleted_count >= max_delete) { + operation_ok = false; + free(child_rel); + continue; + } + if (unlinkat(dirfd, entry->d_name, 0) != 0) { + if (errno != ENOENT) + operation_ok = false; + } else { + (*deleted_count)++; + } + char* escaped_path = output_escape(child_rel, log_get_8_bit_output()); + fprintf(stderr, " Deleted: %s\n", escaped_path ? escaped_path : ""); + free(escaped_path); } } - free(child_abs); free(child_rel); } closedir(dir); - // Only remove the directory itself if it is not in the manifest - // and contained no kept entries. - if (all_removed && rel_path[0] != '\0' && !is_dir_in_manifest(rel_path, manifest)) { - rmdir(abs_path); - } + return operation_ok; } -void delete_extras(const char* dest_root, ArrayList* manifest) { - delete_extras_walk(dest_root, "", manifest); +DeleteWalkResult delete_extras_limited(const char* dest_root, ArrayList* manifest, + size_t max_delete, const DeleteSkipEntry* skips, + int skip_count, size_t* deleted_out) { + if (deleted_out) + *deleted_out = 0; + if (!manifest) + return DELETE_WALK_ERROR; + int rootfd; + if (authorized_root_fd >= 0) { + if (authorized_root_path) + rootfd = open_authorized_destination(dest_root); + else if (dest_root == NULL) + rootfd = dup(authorized_root_fd); + else + rootfd = -1; + } else { + rootfd = open(dest_root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + } + if (rootfd < 0) + return DELETE_WALK_ERROR; + if (max_delete != SIZE_MAX) { + /* Rehearse the deletion first so a run that would exceed the cap removes + nothing (rsync's all-or-nothing --max-delete contract). */ + size_t count = 0; + bool exceeds = false; + bool survives = false; + bool counted_ok = count_extras_fd(rootfd, "", manifest, max_delete, &count, &exceeds, skips, + skip_count, &survives); + if (!counted_ok) { + close(rootfd); + return DELETE_WALK_ERROR; + } + if (exceeds) { + close(rootfd); + return DELETE_WALK_LIMIT_EXCEEDED; + } + } + size_t deleted_count = 0; + bool ok = delete_extras_fd(rootfd, "", manifest, max_delete, &deleted_count, skips, skip_count); + if (close(rootfd) != 0) + ok = false; + if (deleted_out) + *deleted_out = deleted_count; + return ok ? DELETE_WALK_OK : DELETE_WALK_ERROR; +} + +bool delete_extras(const char* dest_root, ArrayList* manifest) { + return delete_extras_limited(dest_root, manifest, SIZE_MAX, NULL, 0, NULL) == DELETE_WALK_OK; } bool has_path_traversal(const char* path) { if (!path) - return false; + return true; char* dup = str_dup(path); if (!dup) - return false; + return true; char* saveptr; const char* part = strtok_r(dup, "/", &saveptr); while (part) { @@ -194,6 +505,10 @@ bool has_path_traversal(const char* path) { return false; } +bool utils_valid_batch_path(const char* path) { + return path && path[0] != '\0' && path[0] != '/' && !has_path_traversal(path); +} + char* path_cat(const char* path1, const char* path2) { if (path1 == NULL || *path1 == '\0') return str_dup(path2); @@ -208,6 +523,8 @@ char* path_cat(const char* path1, const char* path2) { offset = 1; path2_len -= 1; } + if (path1_len > SIZE_MAX - path2_len - 2) + return NULL; char* new_path = malloc(path1_len + path2_len + 2); if (new_path == NULL) return NULL; @@ -217,3 +534,81 @@ char* path_cat(const char* path1, const char* path2) { new_path[path1_len + path2_len + 1] = '\0'; return new_path; } + +bool append_resume_eligible(unsigned long long old_size, unsigned long long check_size) { + return old_size < check_size; +} + +bool append_tail_length(unsigned long long old_size, unsigned long long check_size, + unsigned long long* tail_out) { + if (!tail_out || !append_resume_eligible(old_size, check_size)) + return false; + *tail_out = check_size - old_size; + return true; +} + +/* True when a bound/peer socket address is on the loopback interface: any + 127.0.0.0/8 IPv4 address, IPv6 ::1, or an IPv4-mapped ::ffff:127.x.x.x. This + is the transport-local test the daemon auth gate uses to decide whether a + plaintext connection is a trustworthy local/SSH channel. */ +bool utils_sockaddr_is_loopback(const struct sockaddr* addr) { + if (!addr) + return false; + if (addr->sa_family == AF_INET) { + const struct sockaddr_in* v4 = (const struct sockaddr_in*)addr; + uint32_t host = ntohl(v4->sin_addr.s_addr); + return (host & 0xff000000u) == 0x7f000000u; + } + if (addr->sa_family == AF_INET6) { + const struct sockaddr_in6* v6 = (const struct sockaddr_in6*)addr; + if (IN6_IS_ADDR_LOOPBACK(&v6->sin6_addr)) + return true; + /* An IPv4-mapped ::ffff:127.x.x.x is loopback too. */ + if (IN6_IS_ADDR_V4MAPPED(&v6->sin6_addr) && v6->sin6_addr.s6_addr[12] == 127) + return true; + return false; + } + return false; +} + +/* True when the fd's peer is provably a loopback TCP peer: getpeername must + succeed AND the returned address must classify as loopback. Everything else + is NOT local, including a non-socket descriptor (pipe/socketpair): a failed + getpeername (ENOTSOCK, ENOTCONN, ...) fails closed. The daemon auth gate + must not treat "I cannot tell" as "trusted", and daemon auth modules are + daemon-only anyway (the --stdio path never loads a daemon config). */ +bool utils_fd_peer_is_local(int fd) { + if (fd < 0) + return false; + struct sockaddr_storage peer; + socklen_t length = sizeof(peer); + if (getpeername(fd, (struct sockaddr*)&peer, &length) != 0) + return false; + return utils_sockaddr_is_loopback((const struct sockaddr*)&peer); +} + +/* True when a client-supplied host string names a loopback destination: + "localhost", any 127.0.0.0/8 literal, "::1", or "[::1]". */ +bool utils_host_is_loopback(const char* host) { + if (!host || host[0] == '\0') + return false; + if (strcmp(host, "localhost") == 0) + return true; + struct in_addr v4; + if (inet_pton(AF_INET, host, &v4) == 1) + return (ntohl(v4.s_addr) & 0xff000000u) == 0x7f000000u; + struct in6_addr addr6; + if (host[0] == '[') { + size_t len = strlen(host); + if (len < 3 || host[len - 1] != ']') + return false; + /* inet_pton needs the bare address, not the bracketed form. */ + char bare[INET6_ADDRSTRLEN]; + if (len - 2 >= sizeof(bare)) + return false; + memcpy(bare, host + 1, len - 2); + bare[len - 2] = '\0'; + return inet_pton(AF_INET6, bare, &addr6) == 1 && IN6_IS_ADDR_LOOPBACK(&addr6); + } + return inet_pton(AF_INET6, host, &addr6) == 1 && IN6_IS_ADDR_LOOPBACK(&addr6); +} diff --git a/src/shared/utils.h b/src/shared/utils.h index 1d1d085..4d09c67 100644 --- a/src/shared/utils.h +++ b/src/shared/utils.h @@ -2,13 +2,80 @@ #define UTILS_H #include "array_list.h" +#include #include +#include -bool mkdir_r(const char* path); char* str_dup(const char* string); +char* output_escape(const char* string, bool eight_bit_output); char* path_cat(const char* path1, const char* path2); bool glob_match(const char* pattern, const char* str); -void delete_extras(const char* dest_root, ArrayList* manifest); +/* Result of a bounded extra-file deletion run. */ +typedef enum { + /* Every extra entry was removed (or there were none). */ + DELETE_WALK_OK = 0, + /* The destination holds more extras than the numeric cap for this run. With + the all-or-nothing max-delete semantics NOTHING was removed (the walker + counts first and refuses to start when the run would exceed the limit). */ + DELETE_WALK_LIMIT_EXCEEDED, + /* A traversal or unlink failure aborted the deletion (partial removal is + possible, mirroring the delete pass). */ + DELETE_WALK_ERROR +} DeleteWalkResult; +/* One protected entry for the delete walker. When top_level_only is true the + prefix is skipped only as a DIRECT child of dest_root (the --delay-updates + staging directory, which must not hide genuine extras inside a nested + destination directory that happens to share the staging name); otherwise the + prefix is skipped at any depth (the --compare-dest/--copy-dest/--link-dest + basis trees, and the sender-side protected filter-excluded prefixes, which + are never destination content). */ +typedef struct { + const char* prefix; + bool top_level_only; +} DeleteSkipEntry; +/* True when child_rel is, or lies below, one of the protected entries (a prefix + "a" protects "a" and "a/b/c" but not "ab"; top_level_only entries protect + only DIRECT children of the destination root, i.e. child_rel has no '/'). */ +bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips, + int skip_count); +/* Remove files/dirs under dest_root that are not listed in manifest without + ever descending into a protected prefix (see DeleteSkipEntry). When + max_delete is not SIZE_MAX the run is all-or-nothing: extras are counted + first and DELETE_WALK_LIMIT_EXCEEDED is returned (with nothing removed) when + the count would exceed the cap. `deleted_out` optionally receives the number + of entries actually removed. The all-or-nothing guarantee holds only while + the destination tree is not being concurrently modified: the rehearsal pass + and the delete pass are two separate walks, so a concurrent change between + them (another process adding/removing entries) can make the second pass + delete a different set than the first one counted. */ +DeleteWalkResult delete_extras_limited(const char* dest_root, ArrayList* manifest, + size_t max_delete, const DeleteSkipEntry* skips, + int skip_count, size_t* deleted_out); +bool delete_extras(const char* dest_root, ArrayList* manifest); +bool utils_set_authorized_root(int fd, const char* canonical_path); +/* The fd-only compatibility form is fail-closed for path-based operations; + * callers should use utils_set_authorized_root with the canonical identity. */ +void utils_set_authorized_root_fd(int fd); bool has_path_traversal(const char* path); +bool utils_valid_batch_path(const char* path); +bool format_human_bytes(unsigned long long bytes, char* buffer, size_t buffer_size); +/* --append / --append-verify tail-resume math (pure). A resume is eligible only + when an existing destination file is SHORTER than the source; the tail length + is then the difference. append_resume_eligible answers whether the shorter + file makes a resume possible; append_tail_length additionally returns that + tail length, refusing (false) the degenerate old_size >= check_size case. */ +bool append_resume_eligible(unsigned long long old_size, unsigned long long check_size); +bool append_tail_length(unsigned long long old_size, unsigned long long check_size, + unsigned long long* tail_out); +/* Loopback / local-transport classification for the daemon auth gate and the + client credential rule. utils_sockaddr_is_loopback accepts 127.0.0.0/8, + IPv6 ::1 and IPv4-mapped ::ffff:127.x.x.x; utils_host_is_loopback additionally + accepts the literal "localhost". utils_fd_peer_is_local is fail-closed: it is + true only when getpeername SUCCEEDS and reports a loopback peer -- a non-socket + descriptor (pipe/socketpair) or any getpeername error yields false. See + utils.c for the exact accepted forms. */ +bool utils_sockaddr_is_loopback(const struct sockaddr* addr); +bool utils_fd_peer_is_local(int fd); +bool utils_host_is_loopback(const char* host); #endif diff --git a/src/shared/xattr.c b/src/shared/xattr.c new file mode 100644 index 0000000..d01a38b --- /dev/null +++ b/src/shared/xattr.c @@ -0,0 +1,395 @@ +#define _GNU_SOURCE +#include "xattr.h" +#include "identity.h" +#include "log.h" +#include "protocol.h" +#include "utils.h" +#include "file_types.h" +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/* ---- lifecycle ---- */ + +FileXattrList* xattr_list_new(void) { + FileXattrList* list = protocol_alloc(sizeof(FileXattrList)); + if (!list) + return NULL; + list->items = NULL; + list->count = 0; + return list; +} + +void xattr_list_free(FileXattrList* list) { + if (!list) + return; + for (int i = 0; i < list->count; i++) { + free(list->items[i].name); + free(list->items[i].value); + } + free(list->items); + free(list); +} + +bool xattr_list_append(FileXattrList* list, const char* name, const void* value, size_t value_len) { + if (!list || !name || (!value && value_len != 0)) + return false; + if (list->count >= XATTR_MAX_COUNT) + return false; + FileXattr* grown = realloc(list->items, ((size_t)list->count + 1) * sizeof(FileXattr)); + if (!grown) + return false; + list->items = grown; + size_t name_len = strlen(name); + char* name_copy = malloc(name_len + 1); + if (!name_copy) { + return false; + } + unsigned char* value_copy = NULL; + if (value_len > 0) { + value_copy = malloc(value_len); + if (!value_copy) { + free(name_copy); + return false; + } + memcpy(value_copy, value, value_len); + } + memcpy(name_copy, name, name_len); + name_copy[name_len] = '\0'; + list->items[list->count].name = name_copy; + list->items[list->count].value = value_copy; + list->items[list->count].value_len = value_len; + list->count++; + return true; +} + +/* ---- namespace / length validation ---- */ + +/* A Linux xattr name is "namespace.name" with an optional leading "trusted.", + * "system.", "security.", "user.", or "trusted." prefix. We only ever touch + * the unprivileged "user.*" namespace and the two POSIX ACL xattrs carried in + * the "system." namespace. Everything else -- especially "security.*" (ACLs, + * capabilities, SELinux labels) and "trusted.*" -- is refused so a client can + * never compel the receiver to apply a privileged attribute it would not + * otherwise be able to set (and which would be a local privilege escalation if + * it could). */ +bool xattr_name_appliable(const char* name) { + if (!name || name[0] == '\0') + return false; + size_t len = strlen(name); + if (len > XATTR_NAME_MAX) + return false; + /* The reserved --fake-super key is exclusively the RECEIVER's: it records the + * source stat for a later privileged restore. A plain -X run must never + * forward a source file that already carries this key (spoofable) onto the + * destination, so it is excluded from capture AND from application. Only + * fake_super_store_fd() writes it. */ + if (strcmp(name, FAKESUPER_XATTR) == 0) + return false; + if (strncmp(name, "user.", 5) == 0) + return name[5] != '\0'; + if (strcmp(name, "system.posix_acl_access") == 0) + return true; + if (strcmp(name, "system.posix_acl_default") == 0) + return true; + return false; +} + +/* ---- SENDER: capture ---- */ + +FileXattrList* xattr_capture_path(const char* path) { + if (!path) + return NULL; + ssize_t list_size = listxattr(path, NULL, 0); + if (list_size <= 0) + return NULL; /* no xattrs, ENOTSUP, or error: nothing appliable */ + char* names = malloc((size_t)list_size); + if (!names) + return NULL; + ssize_t got = listxattr(path, names, (size_t)list_size); + if (got < 0) { + free(names); + return NULL; + } + FileXattrList* list = xattr_list_new(); + if (!list) { + free(names); + return NULL; + } + size_t budget = 0; + ssize_t offset = 0; + while (offset < got) { + const char* name = names + offset; + size_t name_len = strlen(name); + if (name_len == 0) + break; /* trailing double NUL not expected; stop */ + offset += (ssize_t)name_len + 1; + if (!xattr_name_appliable(name)) + continue; + ssize_t value_size = getxattr(path, name, NULL, 0); + if (value_size < 0) + continue; + if (value_size > XATTR_VALUE_MAX) + continue; /* oversize value is refused up front (bounded capture) */ + if (name_len + (size_t)value_size > XATTR_TOTAL_MAX - budget) + continue; /* would exceed the per-file budget: skip, keep the rest */ + unsigned char* buffer = malloc(value_size > 0 ? (size_t)value_size : 1); + if (!buffer) { + xattr_list_free(list); + free(names); + return NULL; + } + ssize_t read_len = getxattr(path, name, buffer, (size_t)value_size); + if (read_len < 0 || read_len != value_size) { + free(buffer); + continue; + } + if (!xattr_list_append(list, name, buffer, (size_t)value_size)) { + free(buffer); + xattr_list_free(list); + free(names); + return NULL; + } + free(buffer); + budget += name_len + (size_t)value_size; + } + free(names); + if (list->count == 0) { + xattr_list_free(list); + return NULL; + } + return list; +} + +/* ---- WIRE ---- */ + +bool xattr_send(int fd, const FileXattrList* list) { + int count = list ? list->count : 0; + if (!send_int(fd, count)) + return false; + for (int i = 0; i < count; i++) { + const FileXattr* xa = &list->items[i]; + size_t name_len = strlen(xa->name); + if (name_len > INT32_MAX) + return false; + int32_t name_len32 = (int32_t)name_len; + if (xa->value_len > INT32_MAX) + return false; + int32_t value_len32 = (int32_t)xa->value_len; + if (!send_n_data(fd, &name_len32, sizeof(name_len32)) || !send_n_data(fd, xa->name, name_len) || + !send_n_data(fd, &value_len32, sizeof(value_len32)) || + (value_len32 > 0 && !send_n_data(fd, xa->value, (size_t)value_len32))) + return false; + } + return true; +} + +FileXattrList* xattr_receive(int fd, int* ok) { + if (ok) + *ok = 0; + int count; + if (!receive_int(fd, &count)) + return NULL; + if (count < 0 || count > XATTR_MAX_COUNT) { + log_message(LOG_LEVEL_ERROR, "rejected xattr block: invalid attribute count %d", count); + return NULL; + } + FileXattrList* list = xattr_list_new(); + if (!list) + return NULL; + size_t budget = 0; + for (int i = 0; i < count; i++) { + int32_t name_len32; + if (!receive_n_data(fd, &name_len32, sizeof(name_len32))) { + xattr_list_free(list); + return NULL; + } + if (name_len32 <= 0 || name_len32 > XATTR_NAME_MAX) { + log_message(LOG_LEVEL_ERROR, "rejected xattr block: invalid name length %d", name_len32); + xattr_list_free(list); + return NULL; + } + char* name = protocol_alloc((size_t)name_len32 + 1); + if (!name) { + xattr_list_free(list); + return NULL; + } + if (!receive_n_data(fd, name, (size_t)name_len32)) { + free(name); + xattr_list_free(list); + return NULL; + } + name[name_len32] = '\0'; + if (memchr(name, '\0', (size_t)name_len32) != NULL) { + /* embedded NUL in the name: malformed, reject */ + free(name); + xattr_list_free(list); + return NULL; + } + if (!xattr_name_appliable(name)) { + log_message(LOG_LEVEL_ERROR, "rejected xattr block: disallowed namespace for '%s'", name); + free(name); + xattr_list_free(list); + return NULL; + } + int32_t value_len32; + if (!receive_n_data(fd, &value_len32, sizeof(value_len32))) { + free(name); + xattr_list_free(list); + return NULL; + } + if (value_len32 < 0 || value_len32 > XATTR_VALUE_MAX) { + log_message(LOG_LEVEL_ERROR, "rejected xattr block: invalid value length %d for '%s'", + value_len32, name); + free(name); + xattr_list_free(list); + return NULL; + } + if ((size_t)name_len32 + (size_t)value_len32 > XATTR_TOTAL_MAX - budget) { + log_message(LOG_LEVEL_ERROR, "rejected xattr block: total size budget exceeded for '%s'", + name); + free(name); + xattr_list_free(list); + return NULL; + } + unsigned char* value = NULL; + if (value_len32 > 0) { + value = protocol_alloc((size_t)value_len32); + if (!value) { + free(name); + xattr_list_free(list); + return NULL; + } + if (!receive_n_data(fd, value, (size_t)value_len32)) { + free(value); + free(name); + xattr_list_free(list); + return NULL; + } + } + if (!xattr_list_append(list, name, value, (size_t)value_len32)) { + free(value); + free(name); + xattr_list_free(list); + return NULL; + } + free(value); + free(name); + budget += (size_t)name_len32 + (size_t)value_len32; + } + if (ok) + *ok = 1; + return list; +} + +/* ---- RECEIVER: apply (fd-relative, best-effort) ---- */ + +bool xattr_apply_fd(int fd, const FileXattrList* list) { + if (fd < 0 || !list) + return false; + bool warned = false; + int first_errno = 0; + for (int i = 0; i < list->count; i++) { + const FileXattr* xa = &list->items[i]; + /* Defense in depth: even a hand-crafted list can never apply the reserved + --fake-super key (only fake_super_store_fd may write it). */ + if (strcmp(xa->name, FAKESUPER_XATTR) == 0) + continue; + if (fsetxattr(fd, xa->name, xa->value, xa->value_len, 0) != 0) { + if (!warned) { + warned = true; + first_errno = errno; + } + } + } + /* Collapse potentially many per-attribute failures into one per-file warning + so a run with many unsettable attributes does not spam the log. */ + if (warned) + log_message(LOG_LEVEL_WARNING, "could not set one or more xattrs on the destination file: %s", + strerror(first_errno)); + return true; +} + +/* ---- --fake-super: park ownership/mode/mtime in a reserved xattr ---- */ + +void fake_super_store_fd(int fd, uint32_t uid, uint32_t gid, uint32_t mode, int64_t mtime_sec, + int64_t mtime_nsec) { + if (fd < 0) + return; + char record[128]; + int len = + snprintf(record, sizeof(record), "%lu:%lu:%03o:%lld:%ld", (unsigned long)uid, + (unsigned long)gid, (unsigned)mode & 0777U, (long long)mtime_sec, (long)mtime_nsec); + if (len <= 0 || (size_t)len >= sizeof(record)) + return; + if (fsetxattr(fd, FAKESUPER_XATTR, record, (size_t)len, 0) != 0) { + log_message(LOG_LEVEL_WARNING, "--fake-super: could not store %s on destination file: %s", + FAKESUPER_XATTR, strerror(errno)); + } +} + +/* --fake-super replay: read the freshly-stored record and re-apply the source + * stat fd-relative. A privileged (root) run can actually change the owner; + * a non-root run silently skips the fchown on EPERM/EACCES (never fatal, + * mirroring the normal metadata identity path; other errors are logged) and + * still applies mode/mtime where permitted. + * + * The OWNER leg additionally honors three policies: + * - an explicit ownership identity policy must be active (numeric-ids / + * chown / usermap / groupmap / copy-as). --fake-super on its own only + * RECORDS the source owner; replaying that owner as a live chown without an + * explicit ownership opt-in would be an un-gated client-chosen-ownership + * primitive. + * - --no-super (privilege_super_permitted() false) suppresses it even for a + * root receiver, exactly like the normal metadata identity path. + * - an active --copy-as is AUTHORITATIVE: the identity path already forced the + * target owner, so replaying the recorded source owner here would silently + * override it. The xattr record is still stored/replayed for a later + * privileged restore; only the live chown is skipped. Mode/mtime remain + * applied either way so unprivileged --fake-super still works. */ +bool fake_super_restore_fd(int fd) { + if (fd < 0) + return false; + char record[128]; + ssize_t len = fgetxattr(fd, FAKESUPER_XATTR, record, sizeof(record) - 1); + if (len < 0) + return false; /* absent or filesystem without xattrs: silent no-op */ + record[len] = '\0'; + unsigned long ul_uid, ul_gid, ul_mode; + long long mtime_sec; + long mtime_nsec; + if (sscanf(record, "%lu:%lu:%lo:%lld:%ld", &ul_uid, &ul_gid, &ul_mode, &mtime_sec, &mtime_nsec) != + 5) + return false; /* malformed record: skip, never fatal */ + + /* Owner is applied best-effort only: a non-root process cannot chown and + must not abort the transfer for that reason (FastSync identity philosophy). + EPERM/EACCES (expected for a non-root receiver) are skipped silently; a + genuine EINVAL (an impossible stored id) is logged so the corruption is + not hidden. --no-super suppresses the owner leg even for root, and an + active --copy-as is authoritative so its forced owner must not be + overwritten by the recorded source owner. */ + if (identity_active_enabled() && privilege_super_permitted() && !identity_copy_as_active() && + fchown(fd, (uid_t)ul_uid, (gid_t)ul_gid) != 0 && errno != EPERM && errno != EACCES) + log_message(LOG_LEVEL_WARNING, "--fake-super: could not restore owner on destination file: %s", + strerror(errno)); + /* Mode is applied through the same sanitization the normal metadata path + uses (metadata_mode): group/other write bits are never granted, so a + recorded source mode of 0666 restores as 0644 — identical to a non-fake- + super --preserve run, never a privilege-granting regression. */ + if (fchmod(fd, (mode_t)(ul_mode & 0777U & ~(S_IWGRP | S_IWOTH))) != 0) + log_message(LOG_LEVEL_WARNING, "--fake-super: could not restore mode on destination file: %s", + strerror(errno)); + struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT}, + {.tv_sec = (time_t)mtime_sec, .tv_nsec = mtime_nsec}}; + if (futimens(fd, times) != 0) + log_message(LOG_LEVEL_WARNING, "--fake-super: could not restore mtime on destination file: %s", + strerror(errno)); + return true; +} \ No newline at end of file diff --git a/src/shared/xattr.h b/src/shared/xattr.h new file mode 100644 index 0000000..55f22dd --- /dev/null +++ b/src/shared/xattr.h @@ -0,0 +1,101 @@ +#ifndef XATTR_H +#define XATTR_H + +#include +#include +#include + +/* + * Portable extended-attribute (xattr) and POSIX-ACL preservation (Phase 4, + * protocol 2.13.0). --xattrs/-X and --acls/-A are implemented on top of the + * xattr machinery: the SENDER captures a bounded, namespace-whitelisted set of + * `name = value` pairs per file, transmits them in a per-file wire block, and + * the RECEIVER re-applies them fd-relative on the just-written file. Linux + * xattr syscalls are used; libacl is NOT required (ACLs travel as the + * system.posix_acl_access / system.posix_acl_default xattrs). + * + * Security model: + * * A client can never force a `security.*` / privileged xattr onto the + * destination: both capture (sender) and apply (receiver) are restricted to + * the unprivileged `user.*` namespace and the two POSIX ACL xattrs. The + * receiver independently re-validates every incoming name against this + * whitelist, so a malicious sender's `security.capability` payload is + * rejected, not applied. + * * Payloads are bounded (per-name length, per-value length, per-file count + * and total bytes) on BOTH ends to prevent OOM/memory abuse; an oversized + * or malformed frame is a clean protocol rejection, never an allocation + * blowup. + * * Application is confined to the exact destination file descriptor + * (fsetxattr on the just-written fd), never a caller-controlled path. + */ + +/* Reserved key used by --fake-super to park the source's privileged ownership + * / mode / mtime on the destination file as an unprivileged user.* xattr, so a + * later privileged restore could re-apply them. Exact documented format: + * uid:gid:mode:mtime_sec:mtime_nsec (decimal, decimal, octal, dec, dec) + * e.g. "1000:1000:644:1765238400:0". */ +#define FAKESUPER_XATTR "user.fastsync.stat" + +/* --- bounds --- */ +#define XATTR_NAME_MAX 255 /* xattr names are limited to 255 bytes */ +#define XATTR_VALUE_MAX (1024 * 1024) /* per-value cap (1 MiB) */ +#define XATTR_TOTAL_MAX (4 * 1024 * 1024) /* per-file total name+value bytes */ +#define XATTR_MAX_COUNT 256 + +typedef struct { + char* name; /* owned, NUL-terminated */ + unsigned char* value; /* owned, may hold embedded NULs */ + size_t value_len; +} FileXattr; + +typedef struct { + FileXattr* items; + int count; +} FileXattrList; + +FileXattrList* xattr_list_new(void); +void xattr_list_free(FileXattrList* list); +/* Append one entry (deep copy). Returns false on allocation failure. */ +bool xattr_list_append(FileXattrList* list, const char* name, const void* value, size_t value_len); + +/* True when `name` is a well-formed xattr name AND belongs to a namespace this + * build is authorized to apply (user.* or the two POSIX ACL xattrs). Used for + * both capture and receiver-side validation. */ +bool xattr_name_appliable(const char* name); + +/* Sender: read the whitelisted xattrs of `path` into a new list. Returns NULL + * when the path has no appliable xattrs (or the filesystem has no xattr + * support); an empty-but-valid list is never returned distinct from NULL. */ +FileXattrList* xattr_capture_path(const char* path); + +/* Wire: bounded serialization. xattr_send returns false on write failure; an + * empty/NULL list transmits a zero-count block. xattr_receive returns NULL and + * sets *ok = 0 on any malformed / oversized / non-whitelisted entry. */ +bool xattr_send(int fd, const FileXattrList* list); +FileXattrList* xattr_receive(int fd, int* ok); + +/* Receiver: apply every entry fd-relative (fsetxattr) to the just-written file + * descriptor. A per-attribute failure (e.g. ACL set refused for non-root on a + * file the process does not own) is logged and skipped, never fatal. Returns + * true when apply was attempted (allowing callers to treat it as best-effort). */ +bool xattr_apply_fd(int fd, const FileXattrList* list); + +/* --fake-super: write the source uid/gid/mode/mtime record into the reserved + * FAKESUPER_XATTR on `fd`. Best-effort (logged, never fatal). Only meaningful + * when metadata was transmitted so the values exist. */ +void fake_super_store_fd(int fd, uint32_t uid, uint32_t gid, uint32_t mode, int64_t mtime_sec, + int64_t mtime_nsec); + +/* --fake-super replay: parse the FAKESUPER_XATTR record previously written on + * `fd` by fake_super_store_fd and re-apply uid/gid/mode/mtime fd-relative. + * Best-effort: absence of the xattr or a malformed record is a silent no-op + * that never fails the transfer. The OWNER leg is applied only when an explicit + * ownership identity policy is active (numeric-ids/chown/usermap/groupmap/ + * copy-as), when super-user activities are permitted, and when --copy-as is not + * authoritative; a non-root EPERM/EACCES is skipped silently, matching + * FastSync's identity philosophy. The mode is sanitized exactly like the normal + * metadata path (group/other write bits never granted). Returns true when the + * xattr was present and parsed. */ +bool fake_super_restore_fd(int fd); + +#endif \ No newline at end of file diff --git a/tests/conftest.py b/tests/conftest.py index 51d2ded..d4dd89d 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -1,17 +1,31 @@ """Shared pytest configuration for integration tests.""" import os +import shutil import sys import pytest sys.path.insert(0, os.path.join(os.path.dirname(__file__), "integration")) -from common import ServerManager +from common import ServerManager, TEST_DATA_DIR @pytest.fixture(scope="session") def shared_server(): - """One server for the entire test session. Avoids 27+ server start/stop cycles.""" + """One server for the entire test session. Avoids 27+ server start/stop cycles. + + Under pytest-xdist this session fixture is instantiated once per worker + process, so each worker gets its own server on an ephemeral port.""" server = ServerManager() server.start() yield server server.stop() + + +@pytest.fixture(scope="session", autouse=True) +def _cleanup_worker_test_data(): + """Remove this (worker-keyed) TEST_DATA_DIR at the end of the session. + + Modules clean only their own rows; this final pass ensures the per-worker + directory never lingers in the working tree.""" + yield + shutil.rmtree(TEST_DATA_DIR, ignore_errors=True) diff --git a/tests/fuzz/fuzz_config_receive.c b/tests/fuzz/fuzz_config_receive.c new file mode 100644 index 0000000..12c9c61 --- /dev/null +++ b/tests/fuzz/fuzz_config_receive.c @@ -0,0 +1,346 @@ +/* + * Fuzz the binary config-frame receive path: Config* config_receive(int fd). + * + * The frame is a length-prefixed stream of strings/ints/bools, so the receiver + * stops at the first malformed field. Feeding raw fuzz bytes alone therefore + * almost never reaches the deep P8 trailing blocks (--super / --copy-as) or the + * identity-map block, because every preceding wire bool must be exactly 0 or 1. + * + * To exercise those paths we first build one canonical, fully-valid frame with + * the production sender and then feed the receiver four shapes: + * + * 1. raw : the raw fuzz bytes as the whole frame (version gate included). + * 2. general : the valid version-string prefix + the raw fuzz bytes, so the + * fuzzer can walk the early/core/selection blocks from arbitrary + * input while staying past the version gate. + * 3. tail : the valid frame up to its last P8_TAIL_BYTES (super_mode + + * copy-as presence/uid/gid) + the raw fuzz bytes, so the fuzzer + * directly mutates super_mode and the copy-as ids and truncates + * the tail at any byte. + * 4. map : the valid frame up to the --usermap count + the raw fuzz bytes, + * so the fuzzer directly drives the map count (huge/extreme) and + * the map entries. + * + * The canonical frame is captured by running config_send once, writing the + * frame into a pipe whose read end is drained afterwards; the STATUS_OK ack is + * pre-loaded into a second pipe so a single thread suffices. + */ +#include "config.h" +#include "credentials.h" +#include "protocol.h" +#include "utils.h" +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/* super_mode (4) + copy-as presence (4) + uid (4) + gid (4) = the P8 tail. */ +#define P8_TAIL_BYTES 16 + +/* Distinctive --usermap entry used to locate the map-count field in the + * canonical frame without duplicating the wire layout here. */ +#define MAP_FROM 0x11223344 +#define MAP_TO 0x55667788 + +static unsigned char* g_frame; +static size_t g_frame_len; +static size_t g_version_len; /* length of the leading version-string frame */ +static size_t g_usermap_count_off; /* offset of the usermap count int, 0 = unknown */ +static size_t g_auth_off; /* offset of the auth presence int, 0 = unknown */ +static bool g_auth_found; /* whether g_auth_off is valid */ +static bool g_frame_ready; + +/* Read the canonical frame from the send peer. The producer shuts down its + * write half first, so a blocking read drains the frame and then sees EOF. */ +static unsigned char* drain_frame(int fd, size_t* out_len) { + size_t cap = 4096; + size_t len = 0; + unsigned char* buf = malloc(cap); + if (!buf) + return NULL; + for (;;) { + if (len == cap) { + size_t grown = cap * 2; + unsigned char* bigger = realloc(buf, grown); + if (!bigger) { + free(buf); + return NULL; + } + buf = bigger; + cap = grown; + } + ssize_t n = read(fd, buf + len, cap - len); + if (n > 0) { + len += (size_t)n; + continue; + } + if (n < 0 && errno == EINTR) + continue; + break; /* 0 (EOF) or error */ + } + *out_len = len; + return buf; +} + +/* Serialize a valid Config with the real sender. The frame is written into a + * pipe (64 KiB kernel buffer, far larger than one config frame) whose read end + * is drained afterwards; the STATUS_OK ack is pre-loaded into a second pipe so + * a single thread suffices (config_send writes the whole frame before it reads + * the ack). */ +static void build_canonical_frame(void) { + g_frame_ready = true; + + Config* cfg = config_create(); + if (!cfg) + return; + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + /* Force the shortened auth block (`[present][username]`) to be present so the + * fuzzer can mutate it. */ + cfg->auth_user = str_dup("alice"); + cfg->auth_password = str_dup("alice-s3cret"); + /* Force the three P8 tail fields to be present (copy-as requires metadata). */ + cfg->copy_as_set = true; + cfg->copy_as_uid = 0; + cfg->copy_as_gid = 0; + cfg->use_metadata = true; + /* Force one usermap entry with a locatable sentinel. */ + cfg->usermap = malloc(sizeof(IdentityMap)); + if (cfg->usermap) { + cfg->usermap_count = 1; + cfg->usermap[0].from = MAP_FROM; + cfg->usermap[0].to = MAP_TO; + } + if (!cfg->send_directory || !cfg->receive_root_directory || !cfg->usermap) { + config_delete(cfg); + return; + } + + int frame_pipe[2] = {-1, -1}; + int status_pipe[2] = {-1, -1}; + if (pipe(frame_pipe) != 0 || pipe(status_pipe) != 0) + goto out; + + int ack = STATUS_OK; + if (write(status_pipe[1], &ack, sizeof(ack)) != (ssize_t)sizeof(ack)) + goto out; + + io_set_fds(status_pipe[0], frame_pipe[1]); + io_set_bwlimit(0); + bool sent = config_send(frame_pipe[1], cfg); + close(frame_pipe[1]); + frame_pipe[1] = -1; + close(status_pipe[0]); + status_pipe[0] = -1; + close(status_pipe[1]); + status_pipe[1] = -1; + + if (sent) + g_frame = drain_frame(frame_pipe[0], &g_frame_len); + +out: + if (frame_pipe[0] != -1) + close(frame_pipe[0]); + if (frame_pipe[1] != -1) + close(frame_pipe[1]); + if (status_pipe[0] != -1) + close(status_pipe[0]); + if (status_pipe[1] != -1) + close(status_pipe[1]); + config_delete(cfg); + if (!g_frame || g_frame_len == 0) { + free(g_frame); + g_frame = NULL; + g_frame_len = 0; + return; + } + + g_version_len = sizeof(size_t) + strlen(PROTOCOL_VERSION); + if (g_version_len > g_frame_len) + g_version_len = g_frame_len; + + /* Locate the usermap entry sentinel; its count int sits 4 bytes before it. */ + int32_t from = MAP_FROM; + int32_t to = MAP_TO; + unsigned char pattern[8]; + memcpy(pattern, &from, sizeof(from)); + memcpy(pattern + sizeof(from), &to, sizeof(to)); + if (g_frame_len >= sizeof(pattern)) { + for (size_t i = 4; i + sizeof(pattern) <= g_frame_len; i++) { + if (memcmp(g_frame + i, pattern, sizeof(pattern)) == 0) { + g_usermap_count_off = i - sizeof(int32_t); + break; + } + } + } + + /* Locate the auth username string (a size_t length followed by its bytes); + * the presence int sits one int before the length. The username bytes cannot + * start before sizeof(size_t)+sizeof(int) without the presence-int offset + * underflowing, so begin the scan there. */ + const char* auth_name = "alice"; + size_t auth_name_len = strlen(auth_name); + if (g_frame_len >= sizeof(size_t) + auth_name_len + sizeof(int)) { + for (size_t i = sizeof(size_t) + sizeof(int); i + auth_name_len <= g_frame_len; i++) { + if (memcmp(g_frame + i, auth_name, auth_name_len) != 0) + continue; + size_t found_len = 0; + memcpy(&found_len, g_frame + i - sizeof(size_t), sizeof(size_t)); + if (found_len == auth_name_len) { + g_auth_off = i - sizeof(size_t) - sizeof(int); + g_auth_found = true; + break; + } + } + } +} + +/* Fuzz the A7 auth crypto primitives directly: arbitrary bytes through the + * base64 decoder, plus a self-consistent SCRAM property (a proof built from a + * chosen client key must verify, while a tampered proof, a proof replayed + * against a different nonce, and a not-found verifier must all be refused). */ +static uint8_t pick_byte(const uint8_t* data, size_t size, size_t index) { + return size ? data[index % size] : 0; +} + +static void fuzz_credentials(const uint8_t* data, size_t size) { + char b64[300]; + size_t n = size < sizeof(b64) - 1 ? size : sizeof(b64) - 1; + memcpy(b64, data, n); + b64[n] = '\0'; + uint8_t decoded[64]; + size_t decoded_len = 0; + (void)credentials_b64_decode(b64, decoded, sizeof(decoded), &decoded_len); + + uint8_t client_key[CREDENTIAL_KEY_LEN]; + uint8_t stored_key[CREDENTIAL_KEY_LEN]; + uint8_t server_key[CREDENTIAL_KEY_LEN]; + uint8_t snonce[CREDENTIAL_NONCE_LEN]; + uint8_t cnonce[CREDENTIAL_NONCE_LEN]; + for (size_t i = 0; i < CREDENTIAL_KEY_LEN; i++) { + client_key[i] = pick_byte(data, size, i); + server_key[i] = pick_byte(data, size, i + CREDENTIAL_KEY_LEN); + } + for (size_t i = 0; i < CREDENTIAL_NONCE_LEN; i++) { + snonce[i] = pick_byte(data, size, i + 2 * CREDENTIAL_KEY_LEN); + cnonce[i] = pick_byte(data, size, i + 2 * CREDENTIAL_KEY_LEN + CREDENTIAL_NONCE_LEN); + } + unsigned int stored_len = 0; + if (EVP_Digest(client_key, sizeof(client_key), stored_key, &stored_len, EVP_sha256(), NULL) != + 1 || + stored_len != CREDENTIAL_KEY_LEN) + return; + uint8_t auth_msg[CREDENTIAL_AUTH_MESSAGE_MAX]; + size_t msg_len = 0; + if (!credentials_build_auth_message("alice", snonce, cnonce, auth_msg, sizeof(auth_msg), + &msg_len)) + return; + uint8_t proof[CREDENTIAL_KEY_LEN]; + uint8_t server_sig[CREDENTIAL_KEY_LEN]; + if (!credentials_client_proof(client_key, stored_key, server_key, auth_msg, msg_len, proof, + server_sig)) + return; + CredentialVerifier verifier; + memset(&verifier, 0, sizeof(verifier)); + verifier.found = true; + verifier.iters = CREDENTIAL_DEFAULT_ITERS; + memcpy(verifier.stored_key, stored_key, CREDENTIAL_KEY_LEN); + memcpy(verifier.server_key, server_key, CREDENTIAL_KEY_LEN); + uint8_t out_sig[CREDENTIAL_KEY_LEN]; + if (!credentials_verify_response(&verifier, "alice", snonce, cnonce, proof, out_sig)) + abort(); + if (memcmp(out_sig, server_sig, CREDENTIAL_KEY_LEN) != 0) + abort(); + uint8_t bad_proof[CREDENTIAL_KEY_LEN]; + memcpy(bad_proof, proof, CREDENTIAL_KEY_LEN); + bad_proof[pick_byte(data, size, 0) % CREDENTIAL_KEY_LEN] ^= 0x01; + if (credentials_verify_response(&verifier, "alice", snonce, cnonce, bad_proof, out_sig)) + abort(); + uint8_t other_cnonce[CREDENTIAL_NONCE_LEN]; + memcpy(other_cnonce, cnonce, CREDENTIAL_NONCE_LEN); + other_cnonce[pick_byte(data, size, 1) % CREDENTIAL_NONCE_LEN] ^= 0x80; + if (credentials_verify_response(&verifier, "alice", snonce, other_cnonce, proof, out_sig)) + abort(); + verifier.found = false; + if (credentials_verify_response(&verifier, "alice", snonce, cnonce, proof, out_sig)) + abort(); +} + +/* Best-effort non-blocking write: an oversized fuzz input is truncated rather + * than stalling the harness. */ +static void write_best_effort(int fd, const void* data, size_t size) { + const unsigned char* p = data; + size_t off = 0; + while (off < size) { + ssize_t n = write(fd, p + off, size - off); + if (n > 0) { + off += (size_t)n; + continue; + } + if (n < 0 && errno == EINTR) + continue; + break; + } +} + +/* Build prefix ++ data as a stream and drive config_receive over it. */ +static void receive_stream(const unsigned char* prefix, size_t prefix_len, const uint8_t* data, + size_t size) { + int sv[2]; + if (socketpair(AF_UNIX, SOCK_STREAM, 0, sv) != 0) + return; + + int flags = fcntl(sv[0], F_GETFL, 0); + if (flags != -1) + (void)fcntl(sv[0], F_SETFL, flags | O_NONBLOCK); + + if (prefix_len > 0) + write_best_effort(sv[0], prefix, prefix_len); + if (size > 0) + write_best_effort(sv[0], data, size); + /* Signal EOF without closing the read half, so the receiver's STATUS_ERROR + * replies do not hit EPIPE. */ + shutdown(sv[0], SHUT_WR); + + io_set_fds(sv[1], sv[1]); + io_set_bwlimit(0); + Config* cfg = config_receive(sv[1]); + config_delete(cfg); + + close(sv[0]); + close(sv[1]); +} + +int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { + fuzz_credentials(data, size); + + if (!g_frame_ready) + build_canonical_frame(); + + /* Raw bytes as the whole frame (version gate and all). */ + receive_stream(NULL, 0, data, size); + + if (g_frame) { + /* Keep the valid version prefix, fuzz everything after it. */ + receive_stream(g_frame, g_version_len, data, size); + + /* Keep the valid frame up to the shortened auth block, fuzz it. */ + if (g_auth_found) + receive_stream(g_frame, g_auth_off, data, size); + + /* Keep the valid frame up to the P8 tail, fuzz super_mode + copy-as. */ + if (g_frame_len > P8_TAIL_BYTES) + receive_stream(g_frame, g_frame_len - P8_TAIL_BYTES, data, size); + + /* Keep the valid frame up to the usermap count, fuzz the count + entries. */ + if (g_usermap_count_off > 0) + receive_stream(g_frame, g_usermap_count_off, data, size); + } + return 0; +} diff --git a/tests/fuzz/fuzz_identity_parse.c b/tests/fuzz/fuzz_identity_parse.c new file mode 100644 index 0000000..0b9f961 --- /dev/null +++ b/tests/fuzz/fuzz_identity_parse.c @@ -0,0 +1,58 @@ +/* + * Fuzz the CLI-time identity parsers (identity.h): + * - identity_parse_copy_as + * - identity_parse_map (user and group variants) + * - identity_parse_chown + * + * Each parser mutates a Config, so every input gets a fresh config_create() + * (freed afterwards). After a successful parse the shared wire validator and + * the ownership predicate are also exercised on the mutated config. The input + * is NUL-terminated; embedded NULs simply shorten the effective spec, which is + * fine for a parser fuzzer. + */ +#include "config.h" +#include "identity.h" +#include +#include +#include + +static void exercise(Config* c, const char* spec, int which) { + if (!c) + return; + switch (which) { + case 0: + (void)identity_parse_copy_as(c, spec); + break; + case 1: + (void)identity_parse_map(c, spec, false); + break; + case 2: + (void)identity_parse_map(c, spec, true); + break; + default: + (void)identity_parse_chown(c, spec); + break; + } + (void)identity_wire_valid(c); + (void)identity_ownership_requested(c); + config_delete(c); +} + +int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { + if (size == 0) + return 0; + + char* spec = malloc(size + 1); + if (!spec) + return 0; + memcpy(spec, data, size); + spec[size] = '\0'; + + exercise(config_create(), spec, 0); + exercise(config_create(), spec, 1); + exercise(config_create(), spec, 2); + exercise(config_create(), spec, 3); + + free(spec); + return 0; +} diff --git a/tests/integration/common.py b/tests/integration/common.py index 08cc9c9..c47e6dd 100644 --- a/tests/integration/common.py +++ b/tests/integration/common.py @@ -6,13 +6,19 @@ import socket import subprocess import sys import tempfile +import threading import time PROJECT_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..")) BUILD_DIR = os.path.join(PROJECT_ROOT, "build") SERVER_CMD = [os.path.join(BUILD_DIR, "server")] CLIENT_CMD = [os.path.join(BUILD_DIR, "client")] -TEST_DATA_DIR = os.path.join(PROJECT_ROOT, "test_data") +# Under pytest-xdist each worker process gets its own PYTEST_XDIST_WORKER id +# ('gw0', 'gw1', ...). Worker-key the transient working dir so concurrent +# workers on the shared filesystem never collide on fixtures. Outside xdist +# (or with -n1) this stays the historical 'test_data' path. +_WORKER = os.environ.get("PYTEST_XDIST_WORKER") +TEST_DATA_DIR = os.path.join(PROJECT_ROOT, f"test_data-{_WORKER}" if _WORKER else "test_data") class ServerManager: @@ -25,7 +31,9 @@ class ServerManager: def start(self, extra_args=None): self.stop() self._port = _find_free_port() - cmd = SERVER_CMD + ["-p", str(self._port)] + # Plain TCP is intentionally explicit in the server; integration tests + # exercise that opt-in mode rather than relying on the secure default. + cmd = SERVER_CMD + ["-p", str(self._port), "--allow-unauthenticated"] if extra_args: cmd += extra_args self._proc = subprocess.Popen(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) @@ -51,6 +59,78 @@ class ServerManager: self.stop() +class CountingProxy: + """One-shot TCP forwarder that counts the bytes flowing in each direction + between one client and the real server. + + Client output and --stats report SOURCE lengths, so a delta/fuzzy transfer + that moves only a few percent of the file is invisible in normal output. + Routing the client through this proxy makes the actual wire usage + observable: client_to_server counts every byte the client sent (config, + paths, and file/delta payloads), server_to_client counts the reply bytes + (including the receiver's delta signatures). + """ + + def __init__(self, target_port): + self.target_port = target_port + self._listener = socket.socket() + self._listener.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + self._listener.bind(("127.0.0.1", 0)) + self._listener.listen(1) + self._listener.settimeout(30) + self.port = self._listener.getsockname()[1] + self.client_to_server = 0 + self.server_to_client = 0 + + @staticmethod + def _pump(src, dst, counter): + while True: + try: + data = src.recv(65536) + except OSError: + return + if not data: + try: + dst.shutdown(socket.SHUT_WR) + except OSError: + pass + return + try: + dst.sendall(data) + except OSError: + return + counter[0] += len(data) + + def run(self, cmd): + """Forward one client run (the full command list) to the real server and + return the CompletedProcess after the counts have settled.""" + + def serve(): + try: + client_sock, _ = self._listener.accept() + server_sock = socket.create_connection(("127.0.0.1", self.target_port), + timeout=10) + except OSError: + self._listener.close() + return + c2s, s2c = [0], [0] + a = threading.Thread(target=self._pump, args=(client_sock, server_sock, c2s)) + b = threading.Thread(target=self._pump, args=(server_sock, client_sock, s2c)) + a.start() + b.start() + a.join() + b.join() + self.client_to_server = c2s[0] + self.server_to_client = s2c[0] + self._listener.close() + + thread = threading.Thread(target=serve) + thread.start() + result = subprocess.run(cmd, capture_output=True, text=True, timeout=180) + thread.join(20) + return result + + def run_client(source_dir, dest_dir, flags=None, port=None, extra_args=None): """Run the client and return (result, duration).""" cmd = CLIENT_CMD + ["--source-dir", source_dir, "--dest-dir", dest_dir, "--save-to-disk"] diff --git a/tests/integration/test_append.py b/tests/integration/test_append.py new file mode 100644 index 0000000..685db4c --- /dev/null +++ b/tests/integration/test_append.py @@ -0,0 +1,144 @@ +"""--append / --append-verify tail-resume integration tests. + +A shorter existing destination file is resumed by transferring only the tail: +--append sends it without verifying the retained prefix (rsync parity: a wrong +prefix is kept, so the result can differ from the source), while --append-verify +checksums the retained prefix against the source and, on a mismatch, falls back +to a clean full transfer so the result is always a byte-identical source copy. +""" +import os +import random +import shutil + +import pytest + +from common import ( + TEST_DATA_DIR, + run_client, CountingProxy, clean_dir, + get_dest_received_dir, CLIENT_CMD, +) + +REL = "sub/grow.dat" + + +def _grow_payload(prefix_size, added_size, seed=99): + r = random.Random(seed) + return bytes(r.randbytes(prefix_size)), bytes(r.randbytes(added_size)) + + +class TestAppend: + def _make(self, tag): + source = os.path.join(TEST_DATA_DIR, f"append_{tag}_src") + dest = os.path.join(TEST_DATA_DIR, f"append_{tag}_dst") + clean_dir(source) + shutil.rmtree(dest, ignore_errors=True) + return source, dest + + def _place(self, root, rel, data): + p = os.path.join(root, rel) + os.makedirs(os.path.dirname(p), exist_ok=True) + with open(p, "wb") as fh: + fh.write(data) + return p + + def _read(self, root, rel): + with open(os.path.join(root, rel), "rb") as fh: + return fh.read() + + def _dest_file(self, source, dest, rel): + return os.path.join(get_dest_received_dir(dest, source), rel) + + @pytest.mark.ci + def test_append_resumes_short_dest_atomically(self, shared_server): + """A shorter dest with a MATCHING prefix is resumed; the reconstructed + file is byte-identical to the source.""" + source, dest = self._make("atomic") + prefix, added = _grow_payload(1 * 1024 * 1024, 64 * 1024) + self._place(source, REL, prefix + added) + self._place(self._dest_file(source, dest, ""), REL, prefix) + + result, _ = run_client(source, dest, flags=["--append"], port=shared_server.port) + assert result.returncode == 0, \ + f"--append failed: {(result.stderr or result.stdout)[:400]}" + assert self._read(self._dest_file(source, dest, ""), REL) == prefix + added + + def test_append_verify_matching_prefix_succeeds(self, shared_server): + source, dest = self._make("verify_ok") + prefix, added = _grow_payload(512 * 1024, 32 * 1024) + self._place(source, REL, prefix + added) + self._place(self._dest_file(source, dest, ""), REL, prefix) + + result, _ = run_client(source, dest, flags=["--append-verify"], port=shared_server.port) + assert result.returncode == 0, \ + f"--append-verify failed: {(result.stderr or result.stdout)[:400]}" + assert self._read(self._dest_file(source, dest, ""), REL) == prefix + added + + def test_append_sends_only_tail(self, shared_server): + """Sorted transfer moves only the tail: wire bytes stay well below the + full source size (incompressible payload, no -c).""" + source, dest = self._make("tail") + prefix, added = _grow_payload(4 * 1024 * 1024, 8 * 1024, seed=7) + full = prefix + added + self._place(source, REL, full) + self._place(self._dest_file(source, dest, ""), REL, prefix) + + proxy = CountingProxy(shared_server.port) + cmd = (CLIENT_CMD + ["--source-dir", source, "--dest-dir", dest, + "--save-to-disk", "--server-port", str(proxy.port), "--append"]) + result = proxy.run(cmd) + assert result.returncode == 0, \ + f"--append failed: {(result.stderr or result.stdout)[:400]}" + assert self._read(self._dest_file(source, dest, ""), REL) == full + assert proxy.client_to_server < full.__len__() // 2, \ + f"expected a tail-only transfer, sent {proxy.client_to_server}B for {full.__len__()}B" + + def test_plain_append_wrong_prefix_is_rsync_parity(self, shared_server): + """--append does NOT verify the retained prefix: a wrong prefix is kept, + so the result is prefix+tail (differs from the source). This is the + documented rsync-parity risk of plain --append.""" + source, dest = self._make("plain_wrong") + correct_prefix, added = _grow_payload(256 * 1024, 32 * 1024, seed=1) + wrong_prefix = bytes(b ^ 0xFF for b in correct_prefix) + self._place(source, REL, correct_prefix + added) + self._place(self._dest_file(source, dest, ""), REL, wrong_prefix) + + result, _ = run_client(source, dest, flags=["--append"], port=shared_server.port) + assert result.returncode == 0 + assert self._read(self._dest_file(source, dest, ""), REL) == wrong_prefix + added + + def test_append_verify_wrong_prefix_never_corrupts(self, shared_server): + """--append-verify detects the retained prefix mismatch and falls back to + a full transfer, so the result is a byte-identical source copy.""" + source, dest = self._make("verify_wrong") + correct_prefix, added = _grow_payload(256 * 1024, 32 * 1024, seed=2) + wrong_prefix = bytes(b ^ 0xFF for b in correct_prefix) + self._place(source, REL, correct_prefix + added) + self._place(self._dest_file(source, dest, ""), REL, wrong_prefix) + + result, _ = run_client(source, dest, flags=["--append-verify"], port=shared_server.port) + assert result.returncode == 0, \ + f"--append-verify mismatch fallback failed: {(result.stderr or result.stdout)[:400]}" + assert self._read(self._dest_file(source, dest, ""), REL) == correct_prefix + added + + def test_append_with_inplace(self, shared_server): + source, dest = self._make("inplace") + prefix, added = _grow_payload(128 * 1024, 16 * 1024, seed=3) + self._place(source, REL, prefix + added) + self._place(self._dest_file(source, dest, ""), REL, prefix) + + result, _ = run_client(source, dest, flags=["--append", "--inplace"], + port=shared_server.port) + assert result.returncode == 0, \ + f"--append --inplace failed: {(result.stderr or result.stdout)[:400]}" + assert self._read(self._dest_file(source, dest, ""), REL) == prefix + added + + def test_append_multithreaded(self, shared_server): + source, dest = self._make("mthread") + prefix, added = _grow_payload(512 * 1024, 32 * 1024, seed=4) + self._place(source, REL, prefix + added) + self._place(self._dest_file(source, dest, ""), REL, prefix) + + result, _ = run_client(source, dest, flags=["--append", "--threads"], port=shared_server.port) + assert result.returncode == 0, \ + f"--append -m failed: {(result.stderr or result.stdout)[:400]}" + assert self._read(self._dest_file(source, dest, ""), REL) == prefix + added \ No newline at end of file diff --git a/tests/integration/test_batch.py b/tests/integration/test_batch.py new file mode 100644 index 0000000..15565bf --- /dev/null +++ b/tests/integration/test_batch.py @@ -0,0 +1,120 @@ +"""Residual-batch (client-only) driver tests. + +--write-batch / --only-write-batch emit a self-contained batch file of a whole +source tree; --read-batch applies one locally. None of these cross the wire (no +PROTOCOL_VERSION bump, no config-frame field, no server flag): only --write-batch +also performs a live transfer and so needs a server. +""" +import os +import shutil +import subprocess +import sys +import pytest + +sys.path.insert(0, os.path.dirname(__file__)) +from common import ( + TEST_DATA_DIR, + run_client, + generate_test_files, + verify_transfer, + clean_dir, + get_dest_received_dir, + CLIENT_CMD, +) + +SOURCE_DIR = os.path.join(TEST_DATA_DIR, "batch_source") +DEST1 = os.path.join(TEST_DATA_DIR, "batch_dest1") +DEST2 = os.path.join(TEST_DATA_DIR, "batch_dest2") +BATCH_FILE = os.path.join(TEST_DATA_DIR, "batch.bin") + +BATCH_MAGIC = b"FSTRESBATCH" + + +@pytest.fixture(scope="module", autouse=True) +def setup_test_data(): + generate_test_files(SOURCE_DIR, full=False) + clean_dir(DEST1) + clean_dir(DEST2) + yield + shutil.rmtree(SOURCE_DIR, ignore_errors=True) + shutil.rmtree(DEST1, ignore_errors=True) + shutil.rmtree(DEST2, ignore_errors=True) + for p in (BATCH_FILE,): + if os.path.exists(p): + os.unlink(p) + + +def _run(args): + return CLIENT_CMD + args + + +def test_write_batch_no_server(): + """--only-write-batch emits a batch from the source with no destination and + no server connection.""" + if os.path.exists(BATCH_FILE): + os.unlink(BATCH_FILE) + cmd = _run(["--only-write-batch", BATCH_FILE, SOURCE_DIR]) + result = subprocess.run(cmd, capture_output=True, text=True, timeout=180) + assert result.returncode == 0, (result.stdout, result.stderr) + with open(BATCH_FILE, "rb") as f: + assert f.read(len(BATCH_MAGIC)) == BATCH_MAGIC + # No destination was touched (nothing was created next to the batch). + assert not os.path.exists(os.path.join(DEST1, "small.txt")) + + +def test_read_batch_roundtrip_no_source(): + """--read-batch applies an emitted batch to a fresh destination with no + source and no server; the tree is byte-identical to the source.""" + received = get_dest_received_dir(DEST2, SOURCE_DIR) + clean_dir(DEST2) + cmd = _run(["--read-batch", BATCH_FILE, DEST2]) + result = subprocess.run(cmd, capture_output=True, text=True, timeout=180) + assert result.returncode == 0, (result.stdout, result.stderr) + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"Missing: {missing[:5]}" + assert not mismatches, f"Mismatch: {mismatches[:5]}" + + +def test_write_batch_with_transfer(shared_server): + """--write-batch runs a live transfer to a server AND emits the batch file.""" + if os.path.exists(BATCH_FILE): + os.unlink(BATCH_FILE) + clean_dir(DEST1) + result, _ = run_client( + SOURCE_DIR, DEST1, + flags=["--write-batch", BATCH_FILE], port=shared_server.port) + assert result.returncode == 0, (result.stdout, result.stderr) + with open(BATCH_FILE, "rb") as f: + assert f.read(len(BATCH_MAGIC)) == BATCH_MAGIC + received = get_dest_received_dir(DEST1, SOURCE_DIR) + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"Missing: {missing[:5]}" + assert not mismatches, f"Mismatch: {mismatches[:5]}" + + +def test_read_batch_requires_destination(): + """--read-batch with no positional destination fails cleanly.""" + cmd = _run(["--read-batch", BATCH_FILE]) + result = subprocess.run(cmd, capture_output=True, text=True, timeout=180) + assert result.returncode != 0 + + +def test_only_write_batch_requires_source(): + """--only-write-batch with no source fails cleanly.""" + cmd = _run(["--only-write-batch", BATCH_FILE]) + result = subprocess.run(cmd, capture_output=True, text=True, timeout=180) + assert result.returncode != 0 + + +def test_batch_modes_conflict(): + """The three batch flags are mutually exclusive.""" + combos = [ + ["--write-batch", BATCH_FILE, "--only-write-batch", BATCH_FILE], + ["--write-batch", BATCH_FILE, "--read-batch", BATCH_FILE], + ["--only-write-batch", BATCH_FILE, "--read-batch", BATCH_FILE], + ] + for flags in combos: + cmd = _run(["--source-dir", SOURCE_DIR, "--dest-dir", DEST1] + flags) + result = subprocess.run(cmd, capture_output=True, text=True, timeout=180) + assert result.returncode != 0, \ + f"expected conflict failure for {flags}: {result.stderr}" \ No newline at end of file diff --git a/tests/integration/test_daemon.py b/tests/integration/test_daemon.py new file mode 100644 index 0000000..30410e8 --- /dev/null +++ b/tests/integration/test_daemon.py @@ -0,0 +1,1195 @@ +"""Daemon mode (--daemon + module config + host::module/path destinations) tests. + +These exercise the Wave A daemon foundation end to end: a fastsync-server +started with --daemon reads a FastSync-native module config file, the client +asks for a module with a host::module/path destination, and the transfer lands +in the configured module root only. Read-only modules, unknown modules, and +auth-required modules without valid credentials are all refused cleanly before +any data moves. The A7 auth wave adds the real credential round-trips exercised +in TestDaemonAuthentication: modules that declare `auth users` accept only a +client whose --password-file presents a username on the module's list, proven +through a SCRAM-SHA-256-style challenge/response against a salted PBKDF2 +verifier. The daemon refuses to start when such a module has no credential +store, a legacy SHA-256 store line is hard-rejected, and a replayed response +from another connection is refused. +""" +import base64 +import glob +import hashlib +import hmac +import os +import select +import shutil +import signal +import socket +import stat +import struct +import subprocess +import sys +import tempfile +import time + +import pytest + +sys.path.insert(0, os.path.dirname(__file__)) +from common import ( + TEST_DATA_DIR, + CLIENT_CMD, + SERVER_CMD, + generate_test_files, + run_client, + get_dest_received_dir, + verify_transfer, + _find_free_port, + _wait_for_port, +) + +SOURCE_DIR = os.path.join(TEST_DATA_DIR, "daemon_source") +MODULE_ROOT = os.path.join(TEST_DATA_DIR, "daemon_modules") +FILES_MODULE = os.path.join(MODULE_ROOT, "files") +READONLY_MODULE = os.path.join(MODULE_ROOT, "readonly") +AUTH_MODULE = os.path.join(MODULE_ROOT, "auth") +TEAM_MODULE = os.path.join(MODULE_ROOT, "team") +OWNER_MODULE = os.path.join(MODULE_ROOT, "owner") +CONF_FILE = os.path.join(TEST_DATA_DIR, "fastsyncd.conf") +CRED_FILE = os.path.join(TEST_DATA_DIR, "fastsyncd.passwd") +STARTFAIL_CONF = os.path.join(TEST_DATA_DIR, "fastsyncd_startfail.conf") +STARTFAIL_PORT = None +DETACH_MODULE = os.path.join(MODULE_ROOT, "detach") +DETACH_CONF = os.path.join(TEST_DATA_DIR, "fastsyncd_detach.conf") +DETACH_PORT = None + +# Passwords are never sent as plaintext and never logged; these literals are +# only hashed into the server credential file / client password file. +ALICE_PASS = "alice-s3cret" +BOB_PASS = "bob-s3cret" +WRONG_PASS = "wrong-password" + +# The store holds a salted PBKDF2 verifier (A7 SCRAM); this is the exact +# derivation the C implementation performs, recomputed here so the tests are an +# independent reference. 100000 keeps the module import fast while staying at +# the validation minimum. +CRED_ITERS = 100000 + + +def _verifier(password, salt, iters=CRED_ITERS): + key = hashlib.pbkdf2_hmac("sha256", password.encode(), salt, iters, 32) + client_key = hmac.new(key, b"Client Key", hashlib.sha256).digest() + stored_key = hashlib.sha256(client_key).digest() + server_key = hmac.new(key, b"Server Key", hashlib.sha256).digest() + return stored_key, server_key + + +def _store_line(user, password, iters=CRED_ITERS, salt=None): + if salt is None: + salt = os.urandom(16) + stored_key, server_key = _verifier(password, salt, iters) + return "%s:$fastsync$1$pbkdf2-sha256$%d$%s$%s$%s" % ( + user, iters, base64.b64encode(salt).decode(), + base64.b64encode(stored_key).decode(), base64.b64encode(server_key).decode()) + + +def _store_secrets(line): + """The base64 stored_key/server_key fields of a store line (the values that + must never appear in a log).""" + parts = line.split("$") + return parts[-2], parts[-1] + + +ALICE_LINE = _store_line("alice", ALICE_PASS) +BOB_LINE = _store_line("bob", BOB_PASS) + + +def _write_client_password_file(path, user, password): + with open(path, "w") as f: + f.write("%s:%s\n" % (user, password)) + os.chmod(path, 0o600) + return path + + +def _kill_by_cmdline_marker(marker): + """Send SIGTERM to every running process whose cmdline contains `marker` + (used to clean up the double-forked --daemon, which is orphaned to init and + no longer a child of the test's own process). Portable over /proc so the + tests do not depend on pgrep being present.""" + for proc_path in glob.glob("/proc/[0-9]*/cmdline"): + try: + with open(proc_path, "rb") as f: + data = f.read() + except OSError: + continue + if marker.encode() in data: + try: + os.kill(int(proc_path.split("/")[2]), signal.SIGTERM) + except (ProcessLookupError, ValueError): + pass + time.sleep(0.5) + + +class DaemonManager: + """Boots one fastsync-server --daemon from a config file and tears it down + (including its accept-loop children) on exit.""" + + def __init__(self): + self._proc = None + self._port = None + + def start(self, config_path, port_override=None, extra_args=None): + self.stop() + # When no override is given the daemon binds the config file's `port` + # (the plain config-port path); with an override the --dparam path. + self._port = port_override if port_override is not None else _config_port(config_path) + cmd = (SERVER_CMD + ["--daemon", "--config", config_path, "--allow-unauthenticated", + "--no-detach"]) + if port_override is not None: + cmd += ["--dparam", f"port={port_override}"] + if extra_args: + cmd += extra_args + log_path = os.path.join(TEST_DATA_DIR, "fastsyncd.log") + log = open(log_path, "w") + self._proc = subprocess.Popen( + cmd, stdout=log, stderr=log, stdin=subprocess.DEVNULL, start_new_session=True) + _wait_for_port(self._port, timeout=10) + + def stop(self): + if self._proc: + try: + os.killpg(self._proc.pid, signal.SIGTERM) + except ProcessLookupError: + pass + try: + self._proc.wait(timeout=5) + except subprocess.TimeoutExpired: + os.killpg(self._proc.pid, signal.SIGKILL) + self._proc.wait() + self._proc = None + + @property + def port(self): + return self._port + + def __enter__(self): + return self + + def __exit__(self, *args): + self.stop() + + def __del__(self): + self.stop() + + +def _config_port(config_path): + """Read the explicit `port = N` line out of the daemon config file.""" + with open(config_path) as f: + for line in f: + stripped = line.strip() + if stripped.startswith("port") and "=" in stripped: + return int(stripped.split("=", 1)[1].strip()) + raise RuntimeError(f"no port= in {config_path}") + + +@pytest.fixture(scope="module", autouse=True) +def daemon_env(): + for d in (MODULE_ROOT, FILES_MODULE, READONLY_MODULE, AUTH_MODULE, TEAM_MODULE, OWNER_MODULE, + DETACH_MODULE): + shutil.rmtree(d, ignore_errors=True) + os.makedirs(d, exist_ok=True) + generate_test_files(SOURCE_DIR, full=False) + + # Server-side credential store: alice and bob (salted PBKDF2 verifiers only; + # the plaintext passwords never appear on the daemon host or in any log). + with open(CRED_FILE, "w") as f: + f.write("# daemon credential store (A7 SCRAM)\n") + f.write(ALICE_LINE + "\n") + f.write(BOB_LINE + "\n") + os.chmod(CRED_FILE, 0o600) + + # The config's port is a free port chosen per worker; the `daemon` fixture + # boots on it (the config-port path) and the --dparam override test boots a + # second daemon on a different port. + config_port = _find_free_port() + with open(CONF_FILE, "w") as f: + f.write( + "# FastSync-native daemon config (Wave A grammar)\n" + "port = %d\n" + "\n" + "[files]\n" + "path = %s\n" + "\n" + "[readonly]\n" + "path = %s\n" + "read only = yes\n" + "\n" + "[locked]\n" + "path = %s\n" + "auth users = alice\n" + "\n" + "[team]\n" + "path = %s\n" + "auth users = alice,bob\n" + "\n" + "[owner]\n" + "path = %s\n" + "client owner = yes\n" + % (config_port, FILES_MODULE, READONLY_MODULE, AUTH_MODULE, TEAM_MODULE, OWNER_MODULE)) + + # A dedicated config for the fail-closed startup check: an auth-required + # module with no credential store must refuse to start. Its own free port + # keeps it independent of the running daemon. + global STARTFAIL_PORT + STARTFAIL_PORT = _find_free_port() + with open(STARTFAIL_CONF, "w") as f: + f.write("port = %d\n\n[locked]\npath = %s\nauth users = alice\n" + % (STARTFAIL_PORT, AUTH_MODULE)) + + # A dedicated config for the real (double-fork) detach test: an unique path + # lets cleanup identify and kill the orphaned background daemon by cmdline. + global DETACH_PORT + DETACH_PORT = _find_free_port() + with open(DETACH_CONF, "w") as f: + f.write("port = %d\n\n[detach]\npath = %s\n" % (DETACH_PORT, DETACH_MODULE)) + + yield + _kill_by_cmdline_marker(DETACH_CONF) + shutil.rmtree(MODULE_ROOT, ignore_errors=True) + shutil.rmtree(SOURCE_DIR, ignore_errors=True) + + +@pytest.fixture(scope="module") +def daemon(): + d = DaemonManager() + d.start(CONF_FILE, extra_args=["--password-file", CRED_FILE]) + yield d + d.stop() + + +def _push(dest, port): + result, _ = run_client(SOURCE_DIR, dest, port=port) + return result + + +def _push_with_creds(dest, port, user, password): + """Push using a --password-file carrying user:password (a fresh temp file + each call so tests never share mutable state).""" + cred_path = os.path.join(TEST_DATA_DIR, f"client_{user}_{os.getpid()}_{time.time_ns()}.pw") + _write_client_password_file(cred_path, user, password) + try: + result, _ = run_client(SOURCE_DIR, dest, port=port, + extra_args=["--password-file", cred_path]) + return result + finally: + os.unlink(cred_path) + + +def _tree_file_count(root): + return sum(len(files) for _, _, files in os.walk(root)) if os.path.exists(root) else 0 + + +def _can_mknod(): + """True when this process may create a char device (needs root/CAP_MKNOD).""" + probe = os.path.join(tempfile.gettempdir(), "._fastsync_mknod_probe_%d" % os.getpid()) + try: + os.mknod(probe, stat.S_IFCHR | 0o600, os.makedev(1, 3)) + os.unlink(probe) + return True + except (OSError, AttributeError): + try: + os.unlink(probe) + except OSError: + pass + return False + + +class TestDaemonModuleSelection: + @pytest.mark.ci + def test_module_transfer(self, daemon): + """A host::module/path destination lands inside the module root only.""" + result = _push("127.0.0.1::files", daemon.port) + assert result.returncode == 0, result.stderr or result.stdout + received = get_dest_received_dir(FILES_MODULE, SOURCE_DIR) + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"missing: {missing[:5]}" + assert not mismatches, f"mismatch: {mismatches[:5]}" + + def test_module_subtree(self, daemon): + """The /path part of host::module/path is relative inside the module.""" + sub = os.path.join(FILES_MODULE, "subtree") + os.makedirs(sub, exist_ok=True) + result = _push("127.0.0.1::files/subtree", daemon.port) + assert result.returncode == 0, result.stderr or result.stdout + received = get_dest_received_dir(sub, SOURCE_DIR) + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"missing: {missing[:5]}" + assert not mismatches, f"mismatch: {mismatches[:5]}" + + +class TestDaemonRejection: + def _tree_files(self): + """Snapshot every file path (module-relative) currently under the module + root tree, so confinement can be asserted by diff rather than by an + absolute 'empty' check (other tests legitimately populate modules).""" + files = set() + for root, _, names in os.walk(MODULE_ROOT): + for name in names: + full = os.path.join(root, name) + files.add(os.path.relpath(full, MODULE_ROOT)) + return files + + def test_read_only_module_blocked(self, daemon): + result = _push("127.0.0.1::readonly", daemon.port) + assert result.returncode != 0 + assert _tree_file_count(READONLY_MODULE) == 0, "read-only module must not receive a file" + + def test_read_only_no_write_anywhere(self, daemon): + """A refused read-only transfer must not add a single file anywhere under + the module root tree (negative confinement, not just the target).""" + before = self._tree_files() + result = _push("127.0.0.1::readonly", daemon.port) + assert result.returncode != 0 + assert self._tree_files() == before, "read-only rejection wrote under the module root" + + def test_unknown_module_rejected(self, daemon): + result = _push("127.0.0.1::no-such-module", daemon.port) + assert result.returncode != 0 + + def test_unknown_module_no_write_anywhere(self, daemon): + """An unknown module must be refused cleanly before any file lands + anywhere beneath the module root tree.""" + before = self._tree_files() + result = _push("127.0.0.1::no-such-module", daemon.port) + assert result.returncode != 0 + assert self._tree_files() == before, "unknown-module rejection wrote under the module root" + + def test_module_less_destination_rejected(self, daemon): + """A daemon destination with no module name (host::/path) is refused at + parse time, before any connection payload is sent.""" + result = _push("127.0.0.1::", daemon.port) + assert result.returncode != 0 + result = _push("127.0.0.1::/sub", daemon.port) + assert result.returncode != 0 + + def test_dotdot_destination_rejected(self, daemon): + """A '..' path expansion in the module-relative path is refused at parse + time so a client cannot escape the module root while it is still on the + client side of the wire.""" + result = _push("127.0.0.1::files/../..", daemon.port) + assert result.returncode != 0 + + def test_auth_module_without_credentials_rejected(self, daemon): + """Wave B: an auth-required module refuses a client that presents no + credentials (the daemon does not fall open).""" + result = _push("127.0.0.1::locked", daemon.port) + assert result.returncode != 0 + assert _tree_file_count(AUTH_MODULE) == 0 + + def _assert_ownership_refused(self, daemon, module, flags, + accept=("client-chosen ownership",)): + """A daemon module without `client owner = yes` refuses every + client-chosen ownership / super-user request at the config handshake, + before any data lands. `accept` lists the log phrases that count as the + refusal (a non-root daemon refuses --copy-as earlier, at the privilege + check, so the caller accepts that phrase too).""" + log_path = os.path.join(TEST_DATA_DIR, "fastsyncd.log") + before = os.path.getsize(log_path) if os.path.exists(log_path) else 0 + before_files = self._tree_files() + result, _ = run_client(SOURCE_DIR, f"127.0.0.1::{module}", port=daemon.port, flags=flags) + assert result.returncode != 0, f"the daemon must refuse {flags}" + assert self._tree_files() == before_files, \ + f"{flags} refusal wrote under the module root" + time.sleep(0.3) + with open(log_path, "rb") as f: + f.seek(before) + tail = f.read().decode("utf-8", "replace") + assert any(phrase in tail for phrase in accept), ( + f"daemon did not log the ownership refusal: {tail[-400:]!r}" + ) + + def test_copy_as_refused_by_daemon(self, daemon): + """P7 Wave E hardening: a daemon refuses client-chosen ownership + (--copy-as) outright unless the module opts in with `client owner = yes`, + so even a root daemon must not honor an arbitrary client-selected owner + by default. The refusal happens at the config handshake, before any data + lands.""" + self._assert_ownership_refused( + daemon, "files", ["--copy-as=@65534:@65534"], + accept=("client-chosen ownership", "requires a privileged receiver")) + + def test_super_refused_by_daemon(self, daemon): + """An explicit --super is a super-user activity request, so a daemon + module refuses it unless it opts in with `client owner = yes`. The + refusal happens at the config handshake, before any data lands.""" + self._assert_ownership_refused(daemon, "files", ["--super", "--preserve"]) + + def test_super_refused_by_no_super_daemon(self): + """A daemon started with the operator --no-super veto must still REFUSE + an explicit client --super on a non-opted module: the veto must not turn + the refusal into a silent accept.""" + port = _find_free_port() + d = DaemonManager() + log_path = os.path.join(TEST_DATA_DIR, "fastsyncd.log") + try: + d.start(CONF_FILE, port_override=port, + extra_args=["--password-file", CRED_FILE, "--no-super"]) + result, _ = run_client(SOURCE_DIR, "127.0.0.1::files", port=d.port, + flags=["--super", "--preserve"]) + assert result.returncode != 0, "the --no-super daemon must refuse --super" + with open(log_path, "rb") as f: + tail = f.read().decode("utf-8", "replace") + assert "client-chosen ownership" in tail, ( + f"daemon did not log the --super refusal: {tail[-400:]!r}" + ) + finally: + d.stop() + + def test_numeric_ids_refused_by_daemon(self, daemon): + """P7 Wave E hardening (A1): the daemon ownership gate must cover the + pre-existing identity flags too, not only --copy-as/--super. A module + without `client owner = yes` refuses --numeric-ids at the handshake.""" + self._assert_ownership_refused(daemon, "files", ["--numeric-ids", "--preserve"]) + + def test_chown_refused_by_daemon(self, daemon): + """--chown is client-chosen ownership too and must be refused by a + non-opted-in module.""" + self._assert_ownership_refused(daemon, "files", ["--chown=@65534:@65534", "--preserve"]) + + def test_owner_opt_in_allows_numeric_ids(self, daemon): + """A module that opts in with `client owner = yes` accepts the + client-chosen ownership flags (here --numeric-ids); the transfer + succeeds and lands inside that module root.""" + result, _ = run_client(SOURCE_DIR, "127.0.0.1::owner", port=daemon.port, + flags=["--numeric-ids", "--preserve"]) + assert result.returncode == 0, result.stderr or result.stdout + received = get_dest_received_dir(OWNER_MODULE, SOURCE_DIR) + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"missing: {missing[:5]}" + assert not mismatches, f"mismatch: {mismatches[:5]}" + + def _device_source(self, name): + src = os.path.join(TEST_DATA_DIR, name) + shutil.rmtree(src, ignore_errors=True) + os.makedirs(src) + with open(os.path.join(src, "f.txt"), "wb") as fh: + fh.write(b"device gate\n") + os.mknod(os.path.join(src, "null"), stat.S_IFCHR | 0o666, os.makedev(1, 3)) + return src + + @pytest.mark.skipif(not _can_mknod(), reason="device nodes need root/CAP_MKNOD") + def test_devices_skipped_without_owner_opt_in(self, daemon): + """H3: a non-opted daemon module must not create device nodes even under + the default AUTO super mode (a root daemon would otherwise let any client + mknod arbitrary devices). An ordinary -a push still succeeds; the device + entry is skipped.""" + src = self._device_source("devsrc_noowner") + os.makedirs(os.path.join(FILES_MODULE, "devskip"), exist_ok=True) + result, _ = run_client(src, "127.0.0.1::files/devskip", port=daemon.port, flags=["-a"]) + assert result.returncode == 0, result.stderr or result.stdout + received = get_dest_received_dir(os.path.join(FILES_MODULE, "devskip"), src) + node = os.path.join(received, "null") + assert not os.path.exists(node) or not stat.S_ISCHR(os.stat(node).st_mode), \ + "non-opted daemon module created a device node" + + @pytest.mark.skipif(not _can_mknod(), reason="device nodes need root/CAP_MKNOD") + def test_devices_created_with_owner_opt_in(self, daemon): + """Control: an opted-in module (`client owner = yes`) may create device + nodes under -a, proving the clamp is specific to non-opted modules.""" + src = self._device_source("devsrc_owner") + os.makedirs(os.path.join(OWNER_MODULE, "devok"), exist_ok=True) + result, _ = run_client(src, "127.0.0.1::owner/devok", port=daemon.port, flags=["-a"]) + assert result.returncode == 0, result.stderr or result.stdout + received = get_dest_received_dir(os.path.join(OWNER_MODULE, "devok"), src) + node = os.path.join(received, "null") + assert os.path.exists(node) and stat.S_ISCHR(os.stat(node).st_mode), \ + "opted-in daemon module did not create the device node" + + @pytest.mark.daemon_detach + def test_real_detach_path(self): + """--daemon WITHOUT --no-detach double-forks a real background daemon; + a client can still transfer into the module root, and the orphaned + process is terminated cleanly (via SIGTERM after polling the port).""" + log_path = os.path.join(TEST_DATA_DIR, "fastsyncd_detach.log") + log = open(log_path, "w") + cmd = SERVER_CMD + ["--daemon", "--config", DETACH_CONF, "--allow-unauthenticated"] + proc = subprocess.Popen(cmd, stdout=log, stderr=log, stdin=subprocess.DEVNULL) + try: + _wait_for_port(DETACH_PORT, timeout=15) + result = _push("127.0.0.1::detach", DETACH_PORT) + assert result.returncode == 0, result.stderr or result.stdout + received = get_dest_received_dir(DETACH_MODULE, SOURCE_DIR) + _, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"missing: {missing[:5]}" + finally: + _kill_by_cmdline_marker(DETACH_CONF) + + def test_plaintext_requires_allow_unauthenticated(self): + """Secure default: a daemon started WITHOUT --allow-unauthenticated must + refuse a plaintext client (same posture as the standalone server).""" + d = DaemonManager() + port = _find_free_port() + log_path = os.path.join(TEST_DATA_DIR, "fastsyncd_noauth.log") + log = open(log_path, "w") + cmd = SERVER_CMD + ["--daemon", "--config", CONF_FILE, "--no-detach", + "--password-file", CRED_FILE, + "--dparam", f"port={port}"] + d._proc = subprocess.Popen(cmd, stdout=log, stderr=log, stdin=subprocess.DEVNULL, + start_new_session=True) + d._port = port + _wait_for_port(port, timeout=10) + try: + result = _push("127.0.0.1::files", port) + assert result.returncode != 0 + finally: + d.stop() + + def test_dparam_port_override(self): + """--dparam port=N overrides the config's port and the daemon serves on N.""" + override = _find_free_port() + d = DaemonManager() + d.start(CONF_FILE, port_override=override, extra_args=["--password-file", CRED_FILE]) + try: + result = _push("127.0.0.1::files", override) + assert result.returncode == 0, result.stderr or result.stdout + received = get_dest_received_dir(FILES_MODULE, SOURCE_DIR) + _, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"missing: {missing[:5]}" + finally: + d.stop() + + +# Numeric status values (must match the enum order in src/shared/protocol.h). +STATUS_AUTH_CHALLENGE = 21 +STATUS_AUTH_RESPONSE = 22 +_AUTH_FRAME_MAX = 1 << 20 + + +def _wire_string_frame_len(buf, off): + """Return the total byte length of the wire string at buf[off], or None when + more bytes are needed.""" + if len(buf) < off + 8: + return None + (length,) = struct.unpack_from(" _AUTH_FRAME_MAX: + raise ValueError("oversized auth frame string") + if len(buf) < off + 8 + length: + return None + return 8 + length + + +def _client_cmd(dest, port, cred_path): + return CLIENT_CMD + ["--source-dir", SOURCE_DIR, "--dest-dir", dest, + "--save-to-disk", "--server-port", str(port), + "--password-file", cred_path] + + +class _AuthReplayProxy: + """A one-connection-at-a-time TCP relay in front of the daemon. + + The capture connection records the client's STATUS_AUTH_RESPONSE frame (the + status, the client nonce string and the proof string); the replay connection + substitutes that recorded frame for its own response, so the daemon sees a + proof bound to the FIRST connection's challenge nonce.""" + + def __init__(self, backend_port): + self.backend = ("127.0.0.1", backend_port) + self.server = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + self.server.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + self.server.bind(("127.0.0.1", 0)) + self.server.listen(4) + self.server.settimeout(20) + self.port = self.server.getsockname()[1] + self.stolen = None + # Set when a relayed connection received a SCRAM challenge from the + # backend; lets a test assert the daemon refused before any challenge. + self.saw_challenge = False + + def close(self): + try: + self.server.close() + except OSError: + pass + + def _run_connection(self, capture): + client, _ = self.server.accept() + backend = socket.create_connection(self.backend, timeout=20) + client.settimeout(20) + backend.settimeout(20) + buf_c = b"" + buf_s = b"" + state = "config" + try: + while True: + ready, _, _ = select.select([client, backend], [], [], 20) + if not ready: + break + eof = False + for sock in ready: + data = sock.recv(65536) + if not data: + eof = True + continue + if sock is client: + buf_c += data + else: + buf_s += data + if state == "config": + if buf_c: + backend.sendall(buf_c) + buf_c = b"" + if len(buf_s) >= 4: + (status,) = struct.unpack_from("= 4: + off = 4 + for _ in range(2): + frame = _wire_string_frame_len(buf_c, off) + if frame is None: + break + off += frame + else: + response = buf_c[:off] + buf_c = buf_c[off:] + if capture: + self.stolen = response + backend.sendall(response) + else: + assert self.stolen is not None + backend.sendall(self.stolen) + state = "relay" + if buf_s: + client.sendall(buf_s) + buf_s = b"" + else: + if buf_c: + backend.sendall(buf_c) + buf_c = b"" + if buf_s: + client.sendall(buf_s) + buf_s = b"" + if eof: + break + finally: + client.close() + backend.close() + + +class TestDaemonAuthentication: + """Wave B password authentication round-trips on the shared daemon (its + config declares `locked` with `auth users = alice` and `team` with + `auth users = alice,bob`; the server runs with CRED_FILE holding alice and + bob digest entries).""" + + def test_correct_password_succeeds(self, daemon): + result = _push_with_creds("127.0.0.1::locked", daemon.port, "alice", ALICE_PASS) + assert result.returncode == 0, result.stderr or result.stdout + received = get_dest_received_dir(AUTH_MODULE, SOURCE_DIR) + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"missing: {missing[:5]}" + assert not mismatches, f"mismatch: {mismatches[:5]}" + + def test_wrong_password_rejected_no_data(self, daemon): + before = _tree_file_count(AUTH_MODULE) + result = _push_with_creds("127.0.0.1::locked", daemon.port, "alice", WRONG_PASS) + assert result.returncode != 0 + assert _tree_file_count(AUTH_MODULE) == before, "wrong password must not write a file" + + def test_unknown_user_rejected(self, daemon): + """A user with a valid-shaped password but no store entry is refused + (the daemon must not fall open for unknown users).""" + before = _tree_file_count(AUTH_MODULE) + result = _push_with_creds("127.0.0.1::locked", daemon.port, "mallory", WRONG_PASS) + assert result.returncode != 0 + assert _tree_file_count(AUTH_MODULE) == before + + def test_user_not_on_module_list_rejected(self, daemon): + """bob's credentials verify against the store, but bob is not on the + `locked` module's auth users list, so the connection is refused.""" + before = _tree_file_count(AUTH_MODULE) + result = _push_with_creds("127.0.0.1::locked", daemon.port, "bob", BOB_PASS) + assert result.returncode != 0 + assert _tree_file_count(AUTH_MODULE) == before + + def test_second_module_user_succeeds(self, daemon): + """bob IS on the `team` module's list, so his correct password works + there (module list + credential store both gate).""" + result = _push_with_creds("127.0.0.1::team", daemon.port, "bob", BOB_PASS) + assert result.returncode == 0, result.stderr or result.stdout + received = get_dest_received_dir(TEAM_MODULE, SOURCE_DIR) + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"missing: {missing[:5]}" + assert not mismatches, f"mismatch: {mismatches[:5]}" + + def test_missing_password_file_rejected(self, daemon): + """A client with no --password-file at all is refused by an auth-required + module (no credentials on the wire).""" + result = _push("127.0.0.1::locked", daemon.port) + assert result.returncode != 0 + + def test_open_module_ignores_credentials(self, daemon): + """A module WITHOUT `auth users` stays open: credentials sent + opportunistically (even wrong ones) are ignored, not required.""" + result = _push_with_creds("127.0.0.1::files", daemon.port, "alice", WRONG_PASS) + assert result.returncode == 0, result.stderr or result.stdout + + def test_read_only_still_refuses_authenticated_client(self, daemon): + """Read-only is orthogonal to auth: an authenticated push to a read-only + module is still refused with no data written (Wave A behavior).""" + before = _tree_file_count(READONLY_MODULE) + result = _push_with_creds("127.0.0.1::readonly", daemon.port, "alice", ALICE_PASS) + assert result.returncode != 0 + assert _tree_file_count(READONLY_MODULE) == before + + def test_password_file_requires_daemon_dest(self, daemon): + """Client-side: --password-file without a host::module/path destination is + a client error (fail fast), not a silently ignored flag.""" + cred_path = os.path.join(TEST_DATA_DIR, "client_local.pw") + _write_client_password_file(cred_path, "alice", ALICE_PASS) + try: + # A plain (non-::) destination with --password-file is rejected client-side. + cmd = CLIENT_CMD + ["--source-dir", SOURCE_DIR, "--dest-dir", "/tmp/local-dest-xyz", + "--save-to-disk", "--password-file", cred_path, + "--server-port", str(daemon.port)] + result = subprocess.run(cmd, capture_output=True, text=True) + assert result.returncode != 0 + assert "host::module/path" in (result.stderr or result.stdout) + finally: + os.unlink(cred_path) + + @pytest.mark.ci + def test_remote_plaintext_credentials_rejected_client_side(self): + """A7-3/S1: sending daemon credentials to a clearly non-local daemon + WITHOUT --tls is refused by the client itself, before any network I/O + (192.0.2.0/24 is TEST-NET-1 and never reachable, so a network attempt + would time out instead of failing fast).""" + cred_path = os.path.join(TEST_DATA_DIR, "client_remote.pw") + _write_client_password_file(cred_path, "alice", ALICE_PASS) + try: + cmd = CLIENT_CMD + ["--source-dir", SOURCE_DIR, + "--dest-dir", "192.0.2.1::files", + "--save-to-disk", "--password-file", cred_path, + "--server-port", "873"] + result = subprocess.run(cmd, capture_output=True, text=True, timeout=15) + assert result.returncode != 0 + combined = (result.stderr or "") + (result.stdout or "") + assert "--tls" in combined, combined + finally: + os.unlink(cred_path) + + @pytest.mark.ci + def test_loopback_plaintext_refused_before_challenge_without_flag(self): + """A7-3/S1: an auth-required module reached over loopback plaintext is + refused at the config gate -- before any SCRAM challenge is sent -- when + the operator did NOT pass --allow-unauthenticated. That flag is the + explicit opt-in that makes loopback plaintext an accepted auth + transport; it never permits remote plaintext auth. A relay records the + daemon's first status frame so a challenge is directly observable.""" + d = DaemonManager() + port = _find_free_port() + log_path = os.path.join(TEST_DATA_DIR, "fastsyncd_noauth_auth.log") + log = open(log_path, "w") + cmd = SERVER_CMD + ["--daemon", "--config", CONF_FILE, "--no-detach", + "--password-file", CRED_FILE, "--dparam", f"port={port}"] + d._proc = subprocess.Popen(cmd, stdout=log, stderr=log, stdin=subprocess.DEVNULL, + start_new_session=True) + d._port = port + _wait_for_port(port, timeout=10) + proxy = _AuthReplayProxy(port) + try: + before = _tree_file_count(AUTH_MODULE) + cred = os.path.join(TEST_DATA_DIR, "noauth_loopback.pw") + _write_client_password_file(cred, "alice", ALICE_PASS) + proc = subprocess.Popen(_client_cmd("127.0.0.1::locked", proxy.port, cred), + stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) + proxy._run_connection(capture=True) + out, err = proc.communicate(timeout=30) + assert proc.returncode != 0, "auth over unflagged loopback plaintext must be refused" + assert not proxy.saw_challenge, "daemon sent a SCRAM challenge before the refusal" + assert _tree_file_count(AUTH_MODULE) == before, "a refused connection wrote data" + os.unlink(cred) + finally: + proxy.close() + d.stop() + + def test_client_empty_password_file_rejected(self): + """Client-side: an empty --password-file is rejected (no credentials).""" + cred_path = os.path.join(TEST_DATA_DIR, "client_empty.pw") + with open(cred_path, "w") as f: + f.write("# nothing here\n") + os.chmod(cred_path, 0o600) + try: + cmd = CLIENT_CMD + ["--source-dir", SOURCE_DIR, + "--dest-dir", "127.0.0.1::files", + "--save-to-disk", "--password-file", cred_path] + result = subprocess.run(cmd, capture_output=True, text=True) + assert result.returncode != 0 + assert "no 'user:password'" in (result.stderr or result.stdout) + finally: + os.unlink(cred_path) + + def test_daemon_fails_closed_without_credential_store(self): + """Fail-closed startup: a config with an auth-required module but no + --password-file/--early-input refuses to start (never serves open).""" + proc = subprocess.run( + SERVER_CMD + ["--daemon", "--config", STARTFAIL_CONF, "--no-detach"], + capture_output=True, text=True, timeout=15) + assert proc.returncode != 0 + assert "fail closed" in (proc.stderr or proc.stdout) + + def test_daemon_early_input_feeds_credential_store(self): + """--early-input is an alternative credential store source: a daemon + started with --early-input (and no --password-file) authenticates alice.""" + d = DaemonManager() + port = _find_free_port() + try: + d.start(CONF_FILE, port_override=port, extra_args=["--early-input", CRED_FILE]) + result = _push_with_creds("127.0.0.1::locked", port, "alice", ALICE_PASS) + assert result.returncode == 0, result.stderr or result.stdout + # Wrong password over the early-input store is still rejected. + result = _push_with_creds("127.0.0.1::locked", port, "alice", WRONG_PASS) + assert result.returncode != 0 + finally: + d.stop() + + @pytest.mark.ci + def test_replayed_auth_response_rejected(self, daemon): + """A7 replay defense: an auth response captured from one connection is + refused on a second connection (the proof is bound to the challenge + nonce), and nothing is written to the module root.""" + proxy = _AuthReplayProxy(daemon.port) + try: + cred_a = os.path.join(TEST_DATA_DIR, "replay_a.pw") + _write_client_password_file(cred_a, "alice", ALICE_PASS) + proc_a = subprocess.Popen(_client_cmd("127.0.0.1::locked", proxy.port, cred_a), + stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) + proxy._run_connection(capture=True) + out_a, err_a = proc_a.communicate(timeout=30) + assert proc_a.returncode == 0, err_a or out_a + assert proxy.stolen is not None + os.unlink(cred_a) + + before = _tree_file_count(AUTH_MODULE) + cred_b = os.path.join(TEST_DATA_DIR, "replay_b.pw") + _write_client_password_file(cred_b, "alice", ALICE_PASS) + proc_b = subprocess.Popen(_client_cmd("127.0.0.1::locked", proxy.port, cred_b), + stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) + proxy._run_connection(capture=False) + out_b, err_b = proc_b.communicate(timeout=30) + assert proc_b.returncode != 0, "a replayed auth response must be refused" + assert _tree_file_count(AUTH_MODULE) == before, \ + "a replayed auth response wrote data" + os.unlink(cred_b) + finally: + proxy.close() + + def test_legacy_store_refuses_to_start(self): + """A legacy `user:SHA256HEX` store is hard-rejected: the daemon must not + start and must never accept a replayable bearer digest.""" + legacy = os.path.join(TEST_DATA_DIR, "fastsyncd_legacy.passwd") + with open(legacy, "w") as f: + f.write("alice:9b90e524e94995ee4aeae2ee3c428a53405d1e8db147f44facc46797d0caf4c3\n") + os.chmod(legacy, 0o600) + conf = os.path.join(TEST_DATA_DIR, "fastsyncd_legacy.conf") + port = _find_free_port() + with open(conf, "w") as f: + f.write("port = %d\n\n[locked]\npath = %s\nauth users = alice\n" % (port, AUTH_MODULE)) + try: + proc = subprocess.run( + SERVER_CMD + ["--daemon", "--config", conf, "--no-detach", + "--password-file", legacy], + capture_output=True, text=True, timeout=15) + assert proc.returncode != 0 + combined = (proc.stderr or "") + (proc.stdout or "") + assert "legacy" in combined + assert "alice" in combined + finally: + os.unlink(legacy) + os.unlink(conf) + + def test_auth_log_does_not_leak_password(self, daemon): + """The daemon log must never contain the password or the store verifier.""" + log_path = os.path.join(TEST_DATA_DIR, "fastsyncd.log") + before = os.path.getsize(log_path) if os.path.exists(log_path) else 0 + _push_with_creds("127.0.0.1::locked", daemon.port, "alice", WRONG_PASS) + _push_with_creds("127.0.0.1::locked", daemon.port, "alice", ALICE_PASS) + time.sleep(0.3) + with open(log_path, "rb") as f: + f.seek(before) + tail = f.read().decode("utf-8", "replace") + assert ALICE_PASS not in tail + assert WRONG_PASS not in tail + for secret in _store_secrets(ALICE_LINE): + assert secret not in tail + assert "$fastsync$" not in tail + + def test_auth_secrets_not_logged_at_debug_level(self): + """Under --verbose the daemon enables LOG_DEBUG_ALL, which normally + traces every protocol string -- the auth username/proof/signature must + NOT leak into that trace even then. The redacted marker is logged + instead, while debug protocol logging is actually proving itself active.""" + d = DaemonManager() + port = _find_free_port() + try: + d.start(CONF_FILE, port_override=port, extra_args=["--verbose", + "--password-file", CRED_FILE]) + _push_with_creds("127.0.0.1::locked", port, "alice", ALICE_PASS) + _push_with_creds("127.0.0.1::locked", port, "alice", WRONG_PASS) + time.sleep(0.3) + log_path = os.path.join(TEST_DATA_DIR, "fastsyncd.log") + with open(log_path, "rb") as f: + log = f.read().decode("utf-8", "replace") + finally: + d.stop() + # Debug protocol tracing is genuinely active on the server: the auth + # fields were received (redacted marker) so the leak path is exercised. + assert "Received String: " in log + # The secret-worthy fields must never appear, at any log level. + assert ALICE_PASS not in log + assert WRONG_PASS not in log + for secret in _store_secrets(ALICE_LINE): + assert secret not in log + assert "$fastsync$" not in log + + +class TestDaemonMotd: + """Wave C MOTD: a daemon configured with a global `motd file` sends it to a + host::module/path client right after the config/auth handshake; the client + shows it on stdout unless --no-motd suppresses the display. The MOTD is + escaped at display time so a hostile motd cannot inject terminal escapes. + + Each test boots its own motd-configured daemon (the shared `daemon` fixture + config has no `motd file`). The MOTD is only sent on the daemon listener + path; these all exercise `host::module` connections. + """ + + MOTD_MODULE = os.path.join(MODULE_ROOT, "motd_module") + MOTD_CONF = os.path.join(TEST_DATA_DIR, "fastsyncd_motd.conf") + + def _start(self, motd_path): + port = _find_free_port() + motd_line = "motd file = %s\n" % motd_path if motd_path else "" + os.makedirs(self.MOTD_MODULE, exist_ok=True) + with open(self.MOTD_CONF, "w") as f: + f.write("port = %d\n%s\n[files]\npath = %s\n" % (port, motd_line, self.MOTD_MODULE)) + d = DaemonManager() + d.start(self.MOTD_CONF, port_override=port) + return d, port + + def _push(self, port, extra_args=None): + result, _ = run_client(SOURCE_DIR, "127.0.0.1::files", port=port, + extra_args=extra_args) + return result + + @pytest.mark.ci + def test_motd_displayed(self): + motd_path = os.path.join(TEST_DATA_DIR, "fastsyncd_motd_banner.txt") + banner = "Welcome to the FastSync test daemon\nSecond line here.\n" + with open(motd_path, "w") as f: + f.write(banner) + d, port = self._start(motd_path) + try: + result = self._push(port) + assert result.returncode == 0, result.stderr or result.stdout + assert "Welcome to the FastSync test daemon" in (result.stdout or "") + assert "Second line here." in (result.stdout or "") + finally: + d.stop() + + def test_motd_no_motd_suppresses_display(self): + motd_path = os.path.join(TEST_DATA_DIR, "fastsyncd_motd_banner2.txt") + banner = "This banner must never be shown.\n" + with open(motd_path, "w") as f: + f.write(banner) + d, port = self._start(motd_path) + try: + result = self._push(port, extra_args=["--no-motd"]) + assert result.returncode == 0, result.stderr or result.stdout + assert banner.strip() not in (result.stdout or "") + finally: + d.stop() + + def test_motd_absent_motd_file_no_error(self): + d, port = self._start(os.path.join(TEST_DATA_DIR, "no-such-motd-file.txt")) + try: + result = self._push(port) + assert result.returncode == 0, result.stderr or result.stdout + assert "no-such-motd" not in (result.stdout or "") + finally: + d.stop() + + def test_motd_no_config_key_sends_no_banner(self): + d, port = self._start(None) + try: + result = self._push(port) + assert result.returncode == 0, result.stderr or result.stdout + assert "FastSync test daemon" not in (result.stdout or "") + assert "banner" not in (result.stdout or "") + finally: + d.stop() + + def test_motd_control_bytes_are_escaped(self): + """A hostile motd (ANSI escape sequences) is displayed with every + control byte escaped octal-style, so no terminal escape reaches the + controlling terminal. The transfer still succeeds (the motd is only + display text, never a wire/transfer hazard).""" + motd_path = os.path.join(TEST_DATA_DIR, "fastsyncd_motd_hostile.txt") + with open(motd_path, "w") as f: + f.write("hello\033[31mred\033[0m\n") + d, port = self._start(motd_path) + try: + result = self._push(port) + assert result.returncode == 0, result.stderr or result.stdout + assert "\x1b" not in (result.stdout or ""), "raw ESC byte leaked to stdout" + assert "\\#033[31m" in (result.stdout or ""), result.stdout + assert "\\#033[0m" in (result.stdout or ""), result.stdout + finally: + d.stop() + + +def _generate_tls_certs(cert_dir, extra_san_ips=None): + """Generate a self-signed CA, server cert (with 127.0.0.1 SAN plus any + extra_san_ips) and two client certs signed by that CA: one with the + expected CN (fastsync-client) and one with a WRONG CN, for the TLS+auth + composition and wrong-identity tests.""" + os.makedirs(cert_dir, exist_ok=True) + ca_key, ca_cert = os.path.join(cert_dir, "ca.key"), os.path.join(cert_dir, "ca.pem") + server_key = os.path.join(cert_dir, "server.key") + server_cert = os.path.join(cert_dir, "server.pem") + client_key = os.path.join(cert_dir, "client.key") + client_cert = os.path.join(cert_dir, "client.pem") + wrong_client_key = os.path.join(cert_dir, "wrong_client.key") + wrong_client_cert = os.path.join(cert_dir, "wrong_client.pem") + subprocess.run(["openssl", "req", "-x509", "-newkey", "rsa:2048", "-nodes", + "-keyout", ca_key, "-out", ca_cert, "-days", "1", + "-subj", "/CN=FastSync Test CA"], check=True, capture_output=True) + san = os.path.join(cert_dir, "san.conf") + san_ips = ["IP.1 = 127.0.0.1"] + for index, ip in enumerate(extra_san_ips or [], start=2): + san_ips.append("IP.%d = %s" % (index, ip)) + with open(san, "w") as f: + f.write("[req]\ndistinguished_name = dn\nreq_extensions = v3_req\n\n" + "[dn]\nCN = localhost\n\n[v3_req]\nsubjectAltName = @an\n\n" + "[an]\nDNS.1 = localhost\n" + "\n".join(san_ips) + "\n") + subprocess.run(["openssl", "req", "-newkey", "rsa:2048", "-nodes", + "-keyout", server_key, "-out", os.path.join(cert_dir, "server.csr"), + "-subj", "/CN=localhost", "-config", san], check=True, capture_output=True) + subprocess.run(["openssl", "x509", "-req", "-in", os.path.join(cert_dir, "server.csr"), + "-CA", ca_cert, "-CAkey", ca_key, "-CAcreateserial", + "-out", server_cert, "-days", "1", + "-extfile", san, "-extensions", "v3_req"], check=True, capture_output=True) + subprocess.run(["openssl", "req", "-newkey", "rsa:2048", "-nodes", + "-keyout", client_key, "-out", os.path.join(cert_dir, "client.csr"), + "-subj", "/CN=fastsync-client"], check=True, capture_output=True) + subprocess.run(["openssl", "x509", "-req", "-in", os.path.join(cert_dir, "client.csr"), + "-CA", ca_cert, "-CAkey", ca_key, "-CAcreateserial", + "-out", client_cert, "-days", "1"], check=True, capture_output=True) + subprocess.run(["openssl", "req", "-newkey", "rsa:2048", "-nodes", + "-keyout", wrong_client_key, "-out", os.path.join(cert_dir, "wrong_client.csr"), + "-subj", "/CN=wrong-client"], check=True, capture_output=True) + subprocess.run(["openssl", "x509", "-req", "-in", os.path.join(cert_dir, "wrong_client.csr"), + "-CA", ca_cert, "-CAkey", ca_key, "-CAcreateserial", + "-out", wrong_client_cert, "-days", "1"], check=True, capture_output=True) + return { + "ca": ca_cert, + "server_cert": server_cert, + "server_key": server_key, + "client_cert": client_cert, + "client_key": client_key, + "wrong_client_cert": wrong_client_cert, + "wrong_client_key": wrong_client_key, + } + + +@pytest.mark.skipif(shutil.which("openssl") is None, + reason="openssl CLI required to mint test certificates") +class TestDaemonTLSAuth: + """TLS + password-auth composition: --client-cn (TLS client identity) and + the module password credential check are independent; both can be required + on the same auth-required module. Env-dependent: needs the openssl CLI.""" + + def test_tls_and_password_auth_compose(self): + cert_dir = os.path.join(TEST_DATA_DIR, "daemon_tls_certs") + certs = _generate_tls_certs(cert_dir) + client_creds = os.path.join(TEST_DATA_DIR, "daemon_tls_client.pw") + _write_client_password_file(client_creds, "alice", ALICE_PASS) + d = DaemonManager() + port = _find_free_port() + try: + d.start(CONF_FILE, port_override=port, extra_args=[ + "--tls", "--cert", certs["server_cert"], "--key", certs["server_key"], + "--ca", certs["ca"], "--client-cn", "fastsync-client", + "--password-file", CRED_FILE]) + tls_flags = ["--tls", + "--cert", certs["client_cert"], "--key", certs["client_key"], + "--ca", certs["ca"]] + # Correct password over TLS, with the right client CN: succeeds. + result, _ = run_client(SOURCE_DIR, "127.0.0.1::locked", port=port, + flags=tls_flags, extra_args=["--password-file", client_creds]) + assert result.returncode == 0, (result.stderr or result.stdout)[:300] + # Wrong password over TLS is still refused by the credential check. + bad_creds = os.path.join(TEST_DATA_DIR, "daemon_tls_client_bad.pw") + _write_client_password_file(bad_creds, "alice", WRONG_PASS) + result, _ = run_client(SOURCE_DIR, "127.0.0.1::locked", port=port, + flags=tls_flags, extra_args=["--password-file", bad_creds]) + assert result.returncode != 0 + os.unlink(bad_creds) + finally: + d.stop() + os.unlink(client_creds) + shutil.rmtree(cert_dir, ignore_errors=True) + + @pytest.mark.ci + def test_wrong_client_cn_refused_before_auth_challenge(self): + """A7-3/S1: with --tls AND --allow-unauthenticated, a loopback TLS peer + whose CA-valid client certificate does not match --client-cn is still + refused at the config gate -- before any SCRAM challenge is sent and + before any file data moves. The --allow-unauthenticated flag only opts + in loopback PLAINTEXT; it must never turn a wrong-CN TLS peer into an + accepted auth transport. Runs over 127.0.0.1 so it is deterministic and + never skips; the gate log line (emitted before server_auth_handshake) + plus the unchanged module tree prove the refusal preceded any challenge.""" + cert_dir = os.path.join(TEST_DATA_DIR, "daemon_tls_certs_wrong") + certs = _generate_tls_certs(cert_dir) + client_creds = os.path.join(TEST_DATA_DIR, "daemon_tls_wrong_client.pw") + _write_client_password_file(client_creds, "alice", ALICE_PASS) + d = DaemonManager() + port = _find_free_port() + log_path = os.path.join(TEST_DATA_DIR, "fastsyncd.log") + try: + d.start(CONF_FILE, port_override=port, extra_args=[ + "--tls", "--cert", certs["server_cert"], "--key", certs["server_key"], + "--ca", certs["ca"], "--client-cn", "fastsync-client", + "--password-file", CRED_FILE]) + before_files = _tree_file_count(AUTH_MODULE) + log_before = os.path.getsize(log_path) if os.path.exists(log_path) else 0 + tls_flags = ["--tls", + "--cert", certs["wrong_client_cert"], "--key", + certs["wrong_client_key"], "--ca", certs["ca"]] + result, _ = run_client(SOURCE_DIR, "127.0.0.1::locked", port=port, + flags=tls_flags, extra_args=["--password-file", client_creds]) + assert result.returncode != 0, "a wrong client CN must be refused" + assert _tree_file_count(AUTH_MODULE) == before_files, \ + "a refused connection wrote file data" + with open(log_path, "rb") as f: + f.seek(log_before) + tail = f.read().decode("utf-8", "replace") + assert "requires authentication over an encrypted, verified TLS connection" in tail, \ + tail[-400:] + finally: + d.stop() + os.unlink(client_creds) + shutil.rmtree(cert_dir, ignore_errors=True) diff --git a/tests/integration/test_features.py b/tests/integration/test_features.py index 63a5bf6..bdbec46 100644 --- a/tests/integration/test_features.py +++ b/tests/integration/test_features.py @@ -1,6 +1,11 @@ """Feature tests: incremental sync, bandwidth limiting, dry run, metadata, filters.""" +import filecmp import os +import random import shutil +import socket +import stat +import subprocess import sys import time import pytest @@ -8,13 +13,285 @@ import pytest sys.path.insert(0, os.path.dirname(__file__)) from common import ( PROJECT_ROOT, BUILD_DIR, TEST_DATA_DIR, - run_client, + run_client, CountingProxy, generate_test_files, verify_transfer, clean_dir, make_result, - get_dest_received_dir, CLIENT_CMD, + get_dest_received_dir, CLIENT_CMD, SERVER_CMD, ServerManager, + _find_free_port, _wait_for_port, _wait_proc, ) SOURCE_DIR = os.path.join(TEST_DATA_DIR, "feature_source") DEST_DIR = os.path.join(TEST_DATA_DIR, "feature_dest") +DEVICE_SOURCE = os.path.join(TEST_DATA_DIR, "device_source") +DEVICE_DEST = os.path.join(TEST_DATA_DIR, "device_dest") + + +def _start_captured_server(prefix=None, extra_args=None): + """Start a plain-TCP server with captured stdout/stderr for one test. + + Returns (proc, port). The caller owns `proc` and must terminate it via + `_wait_proc` so a server that ignores SIGTERM is killed instead of leaving + a zombie or raising TimeoutExpired. The shared session server discards its + output, so tests that lock in a receiver-side warning need their own. The + server's SIGTERM handler exits via `_exit`, which does not flush stdio, so + `stdbuf -oL` keeps stdout line-buffered and the warning observable.""" + port = _find_free_port() + cmd = ["stdbuf", "-oL"] + (prefix or []) + SERVER_CMD + ["-p", str(port), "--allow-unauthenticated"] + if extra_args: + cmd += extra_args + proc = subprocess.Popen(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) + _wait_for_port(port) + return proc, port + + +def _stop_captured_server(proc): + """Terminate a captured server and return its (stdout, stderr) text.""" + proc.terminate() + _wait_proc(proc) + return proc.communicate() + + +class TestDeviceSpecial: + """Phase 4: --devices / --specials / -D / --copy-devices / --write-devices. + + Device node CREATION (mknod) is privileged (CAP_MKNOD); CI runs non-root, so + only the FIFO path (mkfifo, unprivileged) is asserted unconditionally. The + real-device-created assertions are guarded to run only as root. Everything + else must simply succeed / skip without aborting. + """ + + def _setup(self): + clean_dir(DEVICE_SOURCE) + clean_dir(DEVICE_DEST) + with open(os.path.join(DEVICE_SOURCE, "plain.txt"), "wb") as f: + f.write(b"regular content\n") + + def test_specials_recreates_fifo(self, shared_server): + self._setup() + os.mkfifo(os.path.join(DEVICE_SOURCE, "pipe.fifo")) + result, _ = run_client(DEVICE_SOURCE, DEVICE_DEST, + flags=["--specials"], port=shared_server.port) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:200]}" + received = get_dest_received_dir(DEVICE_DEST, DEVICE_SOURCE) + fifo = os.path.join(received, "pipe.fifo") + assert os.path.exists(fifo) and stat.S_ISFIFO(os.stat(fifo).st_mode), ( + "source FIFO was not recreated as a FIFO on the destination" + ) + # The regular file alongside it still transferred normally. + with open(os.path.join(received, "plain.txt")) as f: + assert f.read() == "regular content\n" + + def test_D_implies_devices_and_specials_fifo(self, shared_server): + """-D implies --devices --specials; a FIFO is preserved without a crash + even though no device mknod is attempted on the (non-root) receiver.""" + self._setup() + os.mkfifo(os.path.join(DEVICE_SOURCE, "pipe.fifo")) + result, _ = run_client(DEVICE_SOURCE, DEVICE_DEST, + flags=["-D"], port=shared_server.port) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:200]}" + received = get_dest_received_dir(DEVICE_DEST, DEVICE_SOURCE) + assert stat.S_ISFIFO(os.stat(os.path.join(received, "pipe.fifo")).st_mode) + + @pytest.mark.ci + def test_specials_socket_source_skipped_safely(self): + """A socket cannot be recreated by any standard filesystem call, so + --specials must skip it with a note and still complete the run (the + adjacent regular file transfers normally; no socket node appears).""" + self._setup() + sock_path = os.path.join(DEVICE_SOURCE, "source.sock") + s = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) + server, port = _start_captured_server() + try: + s.bind(sock_path) + result, _ = run_client(DEVICE_SOURCE, DEVICE_DEST, + flags=["--specials"], port=port) + finally: + s.close() + out, err = _stop_captured_server(server) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:200]}" + received = get_dest_received_dir(DEVICE_DEST, DEVICE_SOURCE) + with open(os.path.join(received, "plain.txt")) as f: + assert f.read() == "regular content\n" + assert not os.path.lexists(os.path.join(received, "source.sock")), ( + "socket source must be skipped, not materialized" + ) + assert "socket not recreated" in (out + err), ( + f"receiver did not log the documented socket skip: out={out!r} err={err!r}" + ) + + @pytest.mark.ci + @pytest.mark.parametrize("flags", [["--copy-devices"], ["--copy-devices", "--sendfile"]]) + def test_copy_devices_fifo_becomes_regular_file(self, shared_server, flags): + """--copy-devices treats a special source as an ordinary regular-file + copy: a FIFO (st_size 0) becomes a zero-length REGULAR file on the + destination (never a FIFO, never a hang), and the run succeeds. The + --sendfile variant previously blocked forever in the sendfile open(); + the non-regular source now falls back to the buffered read path, so it + must complete within the bounded-time assertion below.""" + self._setup() + os.mkfifo(os.path.join(DEVICE_SOURCE, "device_copy.fifo")) + result, dur = run_client(DEVICE_SOURCE, DEVICE_DEST, + flags=flags, port=shared_server.port) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:200]}" + received = get_dest_received_dir(DEVICE_DEST, DEVICE_SOURCE) + copied = os.path.join(received, "device_copy.fifo") + assert os.path.lexists(copied), "copy-devices source was not transferred" + st = os.lstat(copied) + assert stat.S_ISREG(st.st_mode), ( + f"copy-devices must produce a regular file, got mode {oct(st.st_mode)}" + ) + assert st.st_size == 0, f"expected a size-bounded 0-byte copy, got {st.st_size}" + assert dur < 60, f"{' '.join(flags)} hung on a FIFO source" + + def test_write_devices_non_crash(self, shared_server): + """--write-devices writes into an existing device only; when the + destination holds no device node the entry is skipped safely and the + run still succeeds (never aborts).""" + self._setup() + # Destination already holds a regular file at the source FIFO's path: + # the receiver must not clobber it and must not crash. + os.mkfifo(os.path.join(DEVICE_SOURCE, "target.fifo")) + result, _ = run_client(DEVICE_SOURCE, DEVICE_DEST, + flags=["--write-devices"], port=shared_server.port) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:200]}" + + @pytest.mark.ci + def test_write_devices_regular_file_target_skipped(self, shared_server): + """--write-devices only ever writes into an existing char/block node: a + pre-existing REGULAR file at the destination path is left byte-identical + (not clobbered) and the run still succeeds.""" + self._setup() + received = get_dest_received_dir(DEVICE_DEST, DEVICE_SOURCE) + os.makedirs(received, exist_ok=True) + target = os.path.join(received, "plain.txt") + with open(target, "wb") as f: + f.write(b"pre-existing local content\n") + result, _ = run_client(DEVICE_SOURCE, DEVICE_DEST, + flags=["--write-devices"], port=shared_server.port) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:200]}" + with open(target, "rb") as f: + assert f.read() == b"pre-existing local content\n", ( + "write-devices clobbered a non-device destination" + ) + + @pytest.mark.setpriv + def test_devices_nonroot_receiver_skips_safely(self): + """A receiver without CAP_MKNOD must skip a device entry with a warning + and never abort. A root runner drops the receiver (server) to nobody + via setpriv; on a non-root runner (or without setpriv) the test skips.""" + if os.geteuid() != 0 or shutil.which("setpriv") is None: + pytest.skip("requires root + setpriv to run the receiver unprivileged") + self._setup() + os.mknod(os.path.join(DEVICE_SOURCE, "chardev"), stat.S_IFCHR | 0o666, + os.makedev(1, 3)) + # The unprivileged receiver must be able to create the destination tree. + os.makedirs(DEVICE_DEST, exist_ok=True) + os.chmod(DEVICE_DEST, 0o777) + server, port = _start_captured_server( + prefix=["setpriv", "--reuid=65534", "--regid=65534", "--clear-groups"]) + try: + result, _ = run_client(DEVICE_SOURCE, DEVICE_DEST, + flags=["--devices"], port=port) + finally: + out, err = _stop_captured_server(server) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:300]}" + received = get_dest_received_dir(DEVICE_DEST, DEVICE_SOURCE) + with open(os.path.join(received, "plain.txt")) as f: + assert f.read() == "regular content\n" + assert not os.path.lexists(os.path.join(received, "chardev")), ( + "a receiver without CAP_MKNOD must skip the device node, not create it" + ) + assert ("cannot create device node" in (out + err) + or "device-node creation is not permitted" in (out + err)), ( + f"receiver did not log the documented device skip: out={out!r} err={err!r}" + ) + + @pytest.mark.skipif(os.geteuid() != 0, reason="requires root to create device nodes") + def test_devices_recreates_real_char_device(self, shared_server): + """Root-only: a source char device node is recreated on the destination + with the same type and rdev (privilege-gated mknod path).""" + self._setup() + src_dev = os.path.join(DEVICE_SOURCE, "realdev") + os.mknod(src_dev, stat.S_IFCHR | 0o666, os.makedev(1, 3)) + result, _ = run_client(DEVICE_SOURCE, DEVICE_DEST, + flags=["--devices"], port=shared_server.port) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:200]}" + received = get_dest_received_dir(DEVICE_DEST, DEVICE_SOURCE) + st = os.lstat(os.path.join(received, "realdev")) + assert stat.S_ISCHR(st.st_mode) + assert os.major(st.st_rdev) == 1 and os.minor(st.st_rdev) == 3 + + def test_m_remove_source_files_keeps_recreated_fifo(self, shared_server): + """--threads --remove-source-files --specials: a recreated FIFO must NOT be + acknowledged as a removable source (its outcome must not shift the + per-file status stream, which would break the run and mis-remove the + adjacent regular file). The regular file is removed; the FIFO stays.""" + self._setup() + os.mkfifo(os.path.join(DEVICE_SOURCE, "pipe.fifo")) + result, _ = run_client(DEVICE_SOURCE, DEVICE_DEST, + flags=["--threads", "--remove-source-files", "--specials"], + port=shared_server.port) + assert result.returncode == 0, ( + f"Exit {result.returncode}: {result.stderr[:300]}" + ) + assert not os.path.exists(os.path.join(DEVICE_SOURCE, "plain.txt")), ( + "regular source file should have been removed" + ) + assert os.path.exists(os.path.join(DEVICE_SOURCE, "pipe.fifo")), ( + "recreated FIFO source must never be removed" + ) + + @pytest.mark.skipif(os.geteuid() != 0, reason="requires root to create device nodes") + def test_m_remove_source_files_keeps_recreated_device(self, shared_server): + """Root-only: --threads --remove-source-files --devices must not remove a + source device node the receiver recreated (mirrors the single-threaded + behavior; the special is never acknowledged as a removable source).""" + self._setup() + src_dev = os.path.join(DEVICE_SOURCE, "realdev") + os.mknod(src_dev, stat.S_IFCHR | 0o666, os.makedev(1, 3)) + result, _ = run_client(DEVICE_SOURCE, DEVICE_DEST, + flags=["--threads", "--remove-source-files", "--devices"], + port=shared_server.port) + assert result.returncode == 0, ( + f"Exit {result.returncode}: {result.stderr[:300]}" + ) + assert not os.path.exists(os.path.join(DEVICE_SOURCE, "plain.txt")), ( + "regular source file should have been removed" + ) + assert os.path.exists(src_dev) and stat.S_ISCHR(os.lstat(src_dev).st_mode), ( + "recreated device source must never be removed" + ) + + def test_write_devices_fifo_target_skips_not_hangs(self, shared_server): + """--write-devices must never block on a pre-existing FIFO at the + destination mirror: opening with O_NONBLOCK fails with ENXIO and the + entry is skipped (the FIFO is left untouched and the run succeeds).""" + self._setup() + # Pre-plant a FIFO at the destination mirror of the source file's path. + received = get_dest_received_dir(DEVICE_DEST, DEVICE_SOURCE) + os.makedirs(received, exist_ok=True) + target = os.path.join(received, "plain.txt") + os.mkfifo(target) + result, dur = run_client(DEVICE_SOURCE, DEVICE_DEST, + flags=["--write-devices"], port=shared_server.port) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:200]}" + assert stat.S_ISFIFO(os.lstat(target).st_mode), "FIFO target was clobbered" + assert dur < 60, "write-devices hung on a FIFO target" + + def test_special_confined_to_receive_root(self, shared_server): + """A special node is created only under the receive root; nothing is + ever materialized outside it (the receiver is confined to its + authorized root).""" + self._setup() + os.mkfifo(os.path.join(DEVICE_SOURCE, "confined.fifo")) + result, _ = run_client(DEVICE_SOURCE, DEVICE_DEST, + flags=["--specials"], port=shared_server.port) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:200]}" + # The only new FIFO is under the receive tree; its sibling watchers + # confirm the confined dest layout (no stray node at the source root). + source_fifo_escaped = os.path.join(DEVICE_DEST, "confined.fifo") + assert not os.path.lexists(source_fifo_escaped), "special escaped the receive root" + received = get_dest_received_dir(DEVICE_DEST, DEVICE_SOURCE) + assert stat.S_ISFIFO(os.stat(os.path.join(received, "confined.fifo")).st_mode) @pytest.fixture(scope="module", autouse=True) @@ -22,10 +299,34 @@ def setup_test_data(): generate_test_files(SOURCE_DIR, full=False) clean_dir(DEST_DIR) yield - shutil.rmtree(TEST_DATA_DIR, ignore_errors=True) + # Remove only this module's own dirs. Under pytest-xdist the whole + # (worker-keyed) TEST_DATA_DIR is shared with concurrently-interleaved + # modules, so never rmtree it here. + shutil.rmtree(SOURCE_DIR, ignore_errors=True) + shutil.rmtree(DEST_DIR, ignore_errors=True) class TestDryRun: + def test_trust_sender_transfer_completes(self, shared_server): + """--trust-sender is a receiver-local policy (never sent to the peer). + A transfer run with it must still complete and produce byte-identical + results: the receiver keeps its low-level root confinement, so a normal + trusted transfer is unchanged.""" + clean_dir(DEST_DIR) + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, + flags=["--trust-sender"], port=shared_server.port) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:200]}" + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"Missing files: {missing[:5]}" + assert not mismatches, f"Mismatched files: {mismatches[:5]}" + + def test_human_readable_dry_run(self): + result, dur = run_client(SOURCE_DIR, DEST_DIR, flags=["-h", "--dry-run"]) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:100]}" + assert "Total:" in result.stdout + assert "KB" in result.stdout + def test_dry_run(self): clean_dir(DEST_DIR) result, dur = run_client( @@ -35,8 +336,124 @@ class TestDryRun: assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:100]}" assert "Dry run:" in result.stdout, f"No dry run output: {result.stdout[:200]}" + def test_quiet_suppresses_dry_run_output(self): + result, dur = run_client( + SOURCE_DIR, DEST_DIR, + flags=["-q", "-n", "--progress", "--stats"], + ) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:100]}" + assert result.stdout == "" + assert result.stderr == "" + + def test_quiet_preserves_errors(self): + result, dur = run_client( + SOURCE_DIR, DEST_DIR, + flags=["--quiet", "--server-port", "1"], + ) + assert result.returncode != 0 + assert result.stderr != "" + + @pytest.mark.parametrize("flags", [["-q", "-v"], ["-v", "-q"]]) + def test_quiet_successful_transfer_and_verbose_order(self, shared_server, flags): + clean_dir(DEST_DIR) + result, dur = run_client( + SOURCE_DIR, DEST_DIR, + flags=flags, + port=shared_server.port, + ) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:100]}" + assert result.stdout == "" + assert result.stderr == "" + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"Missing: {missing}" + assert not mismatches, f"Mismatch: {mismatches}" + + +class TestRemoveSourceFiles: + def test_removes_only_transferred_regular_files(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "remove_source") + dest = os.path.join(TEST_DATA_DIR, "remove_dest") + clean_dir(source) + clean_dir(dest) + with open(os.path.join(source, "one.txt"), "wb") as f: + f.write(b"one") + with open(os.path.join(source, "two.txt"), "wb") as f: + f.write(b"two") + os.makedirs(os.path.join(source, "directory")) + os.symlink("one.txt", os.path.join(source, "link.txt")) + + result, _ = run_client(source, dest, flags=["--remove-source-files", "--threads"], + port=shared_server.port) + assert result.returncode == 0, f"Remove-source sync failed: {result.stderr[:200]}" + assert not os.path.exists(os.path.join(source, "one.txt")) + assert not os.path.exists(os.path.join(source, "two.txt")) + assert os.path.isdir(os.path.join(source, "directory")) + assert os.path.islink(os.path.join(source, "link.txt")) + + def test_single_threaded_removes_transferred_file(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "remove_single_source") + dest = os.path.join(TEST_DATA_DIR, "remove_single_dest") + clean_dir(source) + clean_dir(dest) + source_file = os.path.join(source, "file.txt") + with open(source_file, "wb") as f: + f.write(b"single threaded") + + result, _ = run_client(source, dest, flags=["--remove-source-files"], + port=shared_server.port) + assert result.returncode == 0 + assert not os.path.exists(source_file) + + def test_dry_run_preserves_source_files(self): + source = os.path.join(TEST_DATA_DIR, "remove_dry_source") + dest = os.path.join(TEST_DATA_DIR, "remove_dry_dest") + clean_dir(source) + clean_dir(dest) + source_file = os.path.join(source, "file.txt") + with open(source_file, "wb") as f: + f.write(b"keep") + + result, _ = run_client(source, dest, flags=["--remove-source-files", "--dry-run"]) + assert result.returncode == 0 + assert os.path.isfile(source_file) + + def test_failed_connection_preserves_source_files(self): + source = os.path.join(TEST_DATA_DIR, "remove_failed_source") + dest = os.path.join(TEST_DATA_DIR, "remove_failed_dest") + clean_dir(source) + clean_dir(dest) + source_file = os.path.join(source, "file.txt") + with open(source_file, "wb") as f: + f.write(b"keep after failure") + + result, _ = run_client(source, dest, flags=["--remove-source-files"], port=1) + assert result.returncode != 0 + assert os.path.isfile(source_file) + + def test_incremental_skip_preserves_source_file(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "remove_skipped_source") + dest = os.path.join(TEST_DATA_DIR, "remove_skipped_dest") + clean_dir(source) + clean_dir(dest) + source_file = os.path.join(source, "file.txt") + with open(source_file, "wb") as f: + f.write(b"keep after skip") + + # The seed run preserves timestamps (--preserve) so the destination copy has the + # source's exact mtime; otherwise the incremental skip would depend on + # both writes landing in the same whole second (a race). + result, _ = run_client(source, dest, flags=["--preserve"], port=shared_server.port) + assert result.returncode == 0 + result, _ = run_client(source, dest, + flags=["--remove-source-files", "--incremental"], + port=shared_server.port) + assert result.returncode == 0 + assert os.path.isfile(source_file) + class TestArchiveMode: + @pytest.mark.ci def test_archive_mode(self, shared_server): clean_dir(DEST_DIR) result, dur = run_client( @@ -51,6 +468,189 @@ class TestArchiveMode: assert not missing, f"Missing: {missing}" assert not mismatches, f"Mismatch: {mismatches}" + def test_archive_with_negated_links(self, shared_server): + # --archive implies links + metadata + devices + specials. Devices/specials + # force metadata transmission (recreating a node needs the metadata mode), so + # the post-parse layer keeps use_metadata on even under --no-preserve; only the + # independently-negatable --no-links actually takes effect here. + clean_dir(DEST_DIR) + result, dur = run_client( + SOURCE_DIR, DEST_DIR, + flags=["--archive", "--no-links"], + port=shared_server.port, + ) + if result.returncode != 0: + pytest.fail(f"Exit {result.returncode}: {(result.stderr or result.stdout)[:200]}") + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"Missing: {missing}" + assert not mismatches, f"Mismatch: {mismatches}" + + +class TestExecutability: + @pytest.mark.ci + def test_preserves_only_executable_bits(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "executability_source") + dest = os.path.join(TEST_DATA_DIR, "executability_dest") + clean_dir(source) + clean_dir(dest) + source_file = os.path.join(source, "tool.sh") + with open(source_file, "w") as f: + f.write("#!/bin/sh\necho test\n") + os.chmod(source_file, 0o751) + + result, _ = run_client(source, dest, flags=["-E"], port=shared_server.port) + assert result.returncode == 0, f"Executability sync failed: {result.stderr[:200]}" + received_file = os.path.join(get_dest_received_dir(dest, source), "tool.sh") + received_mode = os.stat(received_file).st_mode + assert received_mode & 0o111 == 0o111 + assert received_mode & 0o600 == 0o600 + assert received_mode & 0o077 == 0o011 + + +class TestChmod: + @pytest.mark.ci + def test_chmod_applies_to_transferred_files(self, shared_server): + clean_dir(DEST_DIR) + source_file = os.path.join(SOURCE_DIR, "small.txt") + os.chmod(source_file, 0o777) + result, dur = run_client( + SOURCE_DIR, DEST_DIR, + flags=["--chmod=u=rw,go=r"], + port=shared_server.port, + ) + if result.returncode != 0: + pytest.fail(f"Exit {result.returncode}: {(result.stderr or result.stdout)[:200]}") + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + assert (os.stat(os.path.join(received, "small.txt")).st_mode & 0o777) == 0o644 + + +class TestPreallocate: + """--preallocate allocates the destination file space up front; the final + destination content must be byte-identical to a normal run.""" + + def test_preallocate_transfer_succeeds(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "prealloc_source") + dest = os.path.join(TEST_DATA_DIR, "prealloc_dest") + clean_dir(source) + clean_dir(dest) + payload = os.urandom(2 * 1024 * 1024 + 137) + with open(os.path.join(source, "data.bin"), "wb") as f: + f.write(payload) + with open(os.path.join(source, "small.txt"), "wb") as f: + f.write(b"hello\n") + + result, _ = run_client(source, dest, flags=["--preallocate"], + port=shared_server.port) + assert result.returncode == 0, \ + f"--preallocate failed: {(result.stderr or result.stdout)[:400]}" + + received_dir = os.path.join(dest, os.path.abspath(source).lstrip(os.sep)) + data_path = os.path.join(received_dir, "data.bin") + assert os.path.isfile(data_path), f"destination file not created: {data_path}" + with open(data_path, "rb") as f: + assert f.read() == payload, "destination content mismatch" + small_path = os.path.join(received_dir, "small.txt") + with open(small_path, "rb") as f: + assert f.read() == b"hello\n", "small file content mismatch" + + def test_preallocate_combines_with_partial(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "prealloc_partial_src") + dest = os.path.join(TEST_DATA_DIR, "prealloc_partial_dst") + clean_dir(source) + clean_dir(dest) + with open(os.path.join(source, "f.txt"), "wb") as f: + f.write(b"partial + preallocate\n") + result, _ = run_client(source, dest, flags=["--preallocate", "--partial"], + port=shared_server.port) + assert result.returncode == 0, \ + f"--preallocate --partial failed: {(result.stderr or result.stdout)[:400]}" + received_dir = os.path.join(dest, os.path.abspath(source).lstrip(os.sep)) + with open(os.path.join(received_dir, "f.txt"), "rb") as f: + assert f.read() == b"partial + preallocate\n" + + +class TestCompressionChoice: + @pytest.mark.ci + def test_zstd_choice_compresses(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["--zc", "zstd"], + port=shared_server.port) + assert result.returncode == 0, f"zstd sync failed: {(result.stderr or result.stdout)[:200]}" + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"Missing: {missing}" + assert not mismatches, f"Mismatch: {mismatches}" + + def test_none_choice_disables_compression(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["-z", "--compress-choice", "none"], + port=shared_server.port) + assert result.returncode == 0, f"none sync failed: {(result.stderr or result.stdout)[:200]}" + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"Missing: {missing}" + assert not mismatches, f"Mismatch: {mismatches}" + + +class TestSkipCompress: + @pytest.mark.ci + def test_skip_compress_case_insensitive(self, shared_server): + clean_dir(DEST_DIR) + with open(os.path.join(SOURCE_DIR, "skip-case.TXT"), "wb") as f: + f.write((b"skip compression case test\n" * 100)) + result, _ = run_client( + SOURCE_DIR, DEST_DIR, + flags=["-z", "--skip-compress=.txt"], + port=shared_server.port, + ) + assert result.returncode == 0, f"Skip-compress sync failed: {(result.stderr or result.stdout)[:200]}" + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + with open(os.path.join(received, "skip-case.TXT"), "rb") as f: + assert f.read() == b"skip compression case test\n" * 100 + + def test_skip_compress_empty_list(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client( + SOURCE_DIR, DEST_DIR, + flags=["-z", "--skip-compress="], + port=shared_server.port, + ) + assert result.returncode == 0, f"Empty skip-compress sync failed: {(result.stderr or result.stdout)[:200]}" + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"Missing: {missing}" + assert not mismatches, f"Mismatch: {mismatches}" + + def test_skip_compress_incremental_full_fallback(self, shared_server): + clean_dir(DEST_DIR) + path = os.path.join(SOURCE_DIR, "incremental-skip.TXT") + with open(path, "wb") as f: + f.write(b"original skipped content\n") + flags = ["-z", "--preserve", "--skip-compress=.txt"] + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"Initial sync failed: {(result.stderr or result.stdout)[:200]}" + with open(path, "wb") as f: + f.write(b"updated skipped content\n") + result, _ = run_client( + SOURCE_DIR, DEST_DIR, + flags=flags + ["--incremental"], + port=shared_server.port, + ) + assert result.returncode == 0, f"Incremental sync failed: {(result.stderr or result.stdout)[:200]}" + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + with open(os.path.join(received, "incremental-skip.TXT"), "rb") as f: + assert f.read() == b"updated skipped content\n" + + def test_skip_compress_rejects_chunk_serialization(self, shared_server): + result, _ = run_client( + SOURCE_DIR, DEST_DIR, + flags=["-z", "--chunk-serialization", "--skip-compress=.txt"], + port=shared_server.port, + ) + assert result.returncode != 0 + assert "cannot be combined" in (result.stderr or result.stdout) + class TestExclude: def test_exclude_single(self, shared_server): @@ -139,11 +739,12 @@ class TestSizeFilters: class TestIncremental: + @pytest.mark.ci def test_incremental_skips_unchanged(self, shared_server): clean_dir(DEST_DIR) result, dur = run_client( SOURCE_DIR, DEST_DIR, - flags=["-M"], + flags=["--preserve"], port=shared_server.port, ) assert result.returncode == 0, f"First sync failed: {result.stderr[:100]}" @@ -151,7 +752,7 @@ class TestIncremental: start = time.monotonic() result, dur = run_client( SOURCE_DIR, DEST_DIR, - flags=["-M", "--incremental"], + flags=["--preserve", "--incremental"], port=shared_server.port, ) incremental_time = time.monotonic() - start @@ -163,11 +764,12 @@ class TestIncremental: assert not missing, f"Missing: {missing}" assert not mismatches, f"Mismatch: {mismatches}" + @pytest.mark.ci def test_incremental_detects_changes(self, shared_server): clean_dir(DEST_DIR) result, _ = run_client( SOURCE_DIR, DEST_DIR, - flags=["-M"], + flags=["--preserve"], port=shared_server.port, ) assert result.returncode == 0 @@ -178,7 +780,7 @@ class TestIncremental: result, dur = run_client( SOURCE_DIR, DEST_DIR, - flags=["-M", "--incremental"], + flags=["--preserve", "--incremental"], port=shared_server.port, ) assert result.returncode == 0 @@ -193,13 +795,396 @@ class TestIncremental: content = f.read() assert b"modified content" in content, f"Modified content not transferred: {content[:50]}" + def test_checksum_detects_same_size_and_mtime_change(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["--preserve"], port=shared_server.port) + assert result.returncode == 0 + + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + source_file = os.path.join(SOURCE_DIR, "small.txt") + received_file = os.path.join(received, "small.txt") + source_stat = os.stat(source_file) + with open(received_file, "wb") as f: + f.write(b"different!\n") + os.utime(received_file, (source_stat.st_atime, source_stat.st_mtime)) + + result, _ = run_client(SOURCE_DIR, DEST_DIR, + flags=["--preserve", "--incremental", "--checksum"], + port=shared_server.port) + assert result.returncode == 0, f"Checksum sync failed: {result.stderr[:200]}" + with open(received_file, "rb") as f: + assert f.read() == b"hello world\n" + + def test_size_only_skips_same_size_with_different_mtime(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["--preserve"], port=shared_server.port) + assert result.returncode == 0 + + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + received_file = os.path.join(received, "small.txt") + with open(received_file, "wb") as f: + f.write(b"different!!\n") + os.utime(received_file, (time.time() - 3600, time.time() - 3600)) + + result, _ = run_client( + SOURCE_DIR, DEST_DIR, + flags=["--preserve", "--incremental", "--size-only"], + port=shared_server.port, + ) + assert result.returncode == 0, f"Size-only sync failed: {result.stderr[:200]}" + with open(received_file, "rb") as f: + assert f.read() == b"different!!\n" + + def test_ignore_times_transfers_same_size_and_mtime(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["--preserve"], port=shared_server.port) + assert result.returncode == 0 + + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + source_file = os.path.join(SOURCE_DIR, "small.txt") + received_file = os.path.join(received, "small.txt") + source_stat = os.stat(source_file) + with open(received_file, "wb") as f: + f.write(b"stale data!\n") + os.utime(received_file, (source_stat.st_atime, source_stat.st_mtime)) + + result, _ = run_client(SOURCE_DIR, DEST_DIR, + flags=["--preserve", "--incremental", "--ignore-times"], + port=shared_server.port) + assert result.returncode == 0, f"Ignore-times sync failed: {result.stderr[:200]}" + with open(received_file, "rb") as f: + assert f.read() == b"hello world\n" + + def test_modify_window_allows_subsecond_mtime_difference(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["--preserve"], port=shared_server.port) + assert result.returncode == 0 + + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + source_file = os.path.join(SOURCE_DIR, "small.txt") + received_file = os.path.join(received, "small.txt") + source_stat = os.stat(source_file) + with open(received_file, "wb") as f: + f.write(b"modified!!!\n") + os.utime(received_file, ns=(source_stat.st_atime_ns, + source_stat.st_mtime_ns - 1500000000)) + + result, _ = run_client(SOURCE_DIR, DEST_DIR, + flags=["--preserve", "--incremental", "--modify-window=2"], + port=shared_server.port) + assert result.returncode == 0, f"Modify-window sync failed: {result.stderr[:200]}" + with open(received_file, "rb") as f: + assert f.read() == b"modified!!!\n" + + def test_whole_file_disables_delta_and_keeps_compression(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["--preserve"], port=shared_server.port) + assert result.returncode == 0 + + source_file = os.path.join(SOURCE_DIR, "medium.txt") + with open(source_file, "wb") as f: + f.write(b"whole-file replacement\n" * 5000) + + result, _ = run_client( + SOURCE_DIR, + DEST_DIR, + flags=["--preserve", "--incremental", "--delta", "-W", "-z"], + port=shared_server.port, + ) + assert result.returncode == 0, f"Whole-file sync failed: {(result.stderr or result.stdout)[:200]}" + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"Missing: {missing}" + assert not mismatches, f"Mismatch: {mismatches}" + + +class TestChecksumChoice: + """--checksum-choice/--cc and --checksum-seed: the whole-file digest used by + the --incremental/--checksum handshake is selectable and seedable. The + receiver hashes the on-disk old file with the SAME algorithm+seed, so an + unchanged file is skipped and a changed file (even with identical size and + mtime) is transferred -- and the transfer always lands byte-exact. + FastSync accepts xxh64 (default, seed-aware) and md5; names it does not + implement are rejected, never silently ignored.""" + + def test_unsupported_algorithm_is_rejected(self, shared_server): + result, _ = run_client( + SOURCE_DIR, DEST_DIR, + flags=["--checksum", "--checksum-choice=sha256"], + port=shared_server.port, + ) + assert result.returncode != 0, "sha256 must be rejected, not silently ignored" + + @pytest.mark.parametrize("algo", ["xxh64", "md5"]) + @pytest.mark.parametrize("mt", [False, True]) + def test_unchanged_skipped_and_bytes_preserved(self, shared_server, algo, mt): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["--preserve"], port=shared_server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + + flags = (["--preserve", "--incremental", "--checksum", f"--checksum-choice={algo}"] + + (["--threads"] if mt else [])) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"checksum {algo} run failed: {result.stderr[:200]}" + + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"Missing: {missing}" + assert not mismatches, f"Mismatch: {mismatches}" + + # A changed source file with the SAME size and mtime must still be + # detected (and re-transferred byte-exactly) because the whole-file digest + # differs -- the explicit reason --checksum exists. This exercises the + # sender/receiver digest agreement for a non-default algorithm. + @pytest.mark.parametrize("algo", ["xxh64", "md5"]) + @pytest.mark.parametrize("mt", [False, True]) + def test_changed_same_size_mtime_redetected(self, shared_server, algo, mt): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["--preserve"], port=shared_server.port) + assert result.returncode == 0 + + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + source_file = os.path.join(SOURCE_DIR, "small.txt") # "hello world\n" (12 bytes) + received_file = os.path.join(received, "small.txt") + source_stat = os.stat(source_file) + with open(received_file, "wb") as f: + f.write(b"DDDDDDDDDDDD") # same size, different content + os.utime(received_file, (source_stat.st_atime, source_stat.st_mtime)) + + flags = (["--preserve", "--incremental", "--checksum", f"--checksum-choice={algo}"] + + (["--threads"] if mt else [])) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"checksum {algo} redetect failed: {result.stderr[:200]}" + with open(received_file, "rb") as f: + assert f.read() == b"hello world\n" + + @pytest.mark.parametrize("algo", ["xxh64", "md5"]) + def test_unchanged_run_transfers_almost_no_data(self, shared_server, algo): + # A fully-unchanged --checksum run skips every file: only the config + a + # small handshake travels, not the payloads. Proxy byte counts are not + # available for -m (multithreaded connections), so single-thread only. + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["--preserve"], port=shared_server.port) + assert result.returncode == 0 + + flags = ["--preserve", "--incremental", "--checksum", f"--checksum-choice={algo}"] + proxy = CountingProxy(shared_server.port) + cmd = (CLIENT_CMD + ["--source-dir", SOURCE_DIR, "--dest-dir", DEST_DIR, + "--save-to-disk", "--server-port", str(proxy.port)] + flags) + result = proxy.run(cmd) + assert result.returncode == 0, f"checksum {algo} skip run failed: {result.stderr[:200]}" + assert proxy.client_to_server < 100000, \ + f"unchanged --checksum run sent {proxy.client_to_server} bytes; expected a skip" + + @pytest.mark.parametrize("mt", [False, True]) + def test_seed_is_deterministic_and_preserves_content(self, shared_server, mt): + clean_dir(DEST_DIR) + flags = ["--preserve", "--incremental", "--checksum", + "--checksum-choice=xxh64", "--checksum-seed=987654"] + (["--threads"] if mt else []) + first, _ = run_client(SOURCE_DIR, DEST_DIR, flags=flags, port=shared_server.port) + assert first.returncode == 0, f"seeded run failed: {first.stderr[:200]}" + + # A second run with the SAME seed and unchanged content skips everything + # deterministically (same digests both sides). + second, _ = run_client(SOURCE_DIR, DEST_DIR, flags=flags, port=shared_server.port) + assert second.returncode == 0, f"deterministic rerun failed: {second.stderr[:200]}" + + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing and not mismatches, f"missing={missing} mismatches={mismatches}" + + # A changed file with the same size and mtime is still caught and fixed + # (a non-zero seed does not weaken the comparison). + source_file = os.path.join(SOURCE_DIR, "medium.txt") + received_file = os.path.join(received, "medium.txt") + source_stat = os.stat(source_file) + with open(received_file, "wb") as f: + f.write(b"z" * os.path.getsize(source_file)) + os.utime(received_file, (source_stat.st_atime, source_stat.st_mtime)) + third, _ = run_client(SOURCE_DIR, DEST_DIR, flags=flags, port=shared_server.port) + assert third.returncode == 0, f"seeded redetect failed: {third.stderr[:200]}" + with open(received_file, "rb") as f: + assert f.read() == open(source_file, "rb").read() + + # --checksum-seed also feeds the delta path's per-block strong checksum on + # both ends (receiver signature and sender window hash use the same seed), + # so a seeded delta transfer still lands byte-exact. + @pytest.mark.parametrize("mt", [False, True]) + def test_seed_delta_block_hash_transfers_byte_exact(self, shared_server, mt): + source = os.path.join(TEST_DATA_DIR, f"ccseed_{'m' if mt else 's'}_src") + dest = os.path.join(TEST_DATA_DIR, f"ccseed_{'m' if mt else 's'}_dst") + clean_dir(source) + clean_dir(dest) + big = os.path.join(source, "big.bin") + with open(big, "wb") as f: + f.write(bytes(range(256)) * 200) # 51200 bytes > delta 16K floor + result, _ = run_client(source, dest, flags=["--preserve"], port=shared_server.port) + assert result.returncode == 0, f"seed delta seed failed: {result.stderr[:200]}" + + # Edit a region so the receiver must match a changed block with the seed. + with open(big, "r+b") as f: + f.seek(1000) + f.write(b"\x00" * 64) + # Force an mtime mismatch: the incremental quick-check skips files whose + # stored mtime second equals the source's, which can collide when the + # edit and the prior sync share a second. Setting an old dest mtime + # guarantees the delta path is exercised deterministically. + os.utime(os.path.join(get_dest_received_dir(dest, source), "big.bin"), (0, 0)) + flags = (["--preserve", "--incremental", "--delta", "--checksum-seed=314159"] + + (["--threads"] if mt else [])) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"seed delta run failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, "big.bin")) == _read_file(big), \ + "seeded delta transfer is not byte-exact" + + +class TestUpdate: + @pytest.mark.ci + def test_update_skips_older_destination_and_allows_equal_or_newer_source(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["-u"], port=shared_server.port) + assert result.returncode == 0 + + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + source_file = os.path.join(SOURCE_DIR, "small.txt") + received_file = os.path.join(received, "small.txt") + source_stat = os.stat(source_file) + + with open(received_file, "wb") as f: + f.write(b"newer destination\n") + os.utime(received_file, ns=(source_stat.st_atime_ns, source_stat.st_mtime_ns + 10_000_000_000)) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["-u"], port=shared_server.port) + assert result.returncode == 0 + with open(received_file, "rb") as f: + assert f.read() == b"newer destination\n" + + os.utime(received_file, ns=(source_stat.st_atime_ns, source_stat.st_mtime_ns)) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["-u"], port=shared_server.port) + assert result.returncode == 0 + with open(received_file, "rb") as f: + assert f.read() == b"hello world\n" + + def test_update_skips_unreadable_newer_destination(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["-u"], port=shared_server.port) + assert result.returncode == 0 + + received_file = os.path.join(get_dest_received_dir(DEST_DIR, SOURCE_DIR), "small.txt") + source_stat = os.stat(os.path.join(SOURCE_DIR, "small.txt")) + with open(received_file, "wb") as f: + f.write(b"protected destination\n") + os.utime(received_file, ns=(source_stat.st_atime_ns, source_stat.st_mtime_ns + 10_000_000_000)) + original_mode = os.stat(received_file).st_mode + try: + os.chmod(received_file, 0) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["-u"], port=shared_server.port) + assert result.returncode == 0 + os.chmod(received_file, original_mode) + with open(received_file, "rb") as f: + assert f.read() == b"protected destination\n" + finally: + os.chmod(received_file, original_mode) + + with open(received_file, "wb") as f: + f.write(b"older destination\n") + os.utime(received_file, ns=(source_stat.st_atime_ns, source_stat.st_mtime_ns - 10_000_000_000)) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["-u"], port=shared_server.port) + assert result.returncode == 0 + with open(received_file, "rb") as f: + assert f.read() == b"hello world\n" + + +class TestExisting: + @pytest.mark.ci + def test_existing_updates_existing_and_skips_new(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["--preserve"], port=shared_server.port) + assert result.returncode == 0, f"Initial sync failed: {(result.stderr or result.stdout)[:200]}" + + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + source_file = os.path.join(SOURCE_DIR, "small.txt") + new_source_file = os.path.join(SOURCE_DIR, "new-existing-test.txt") + with open(source_file, "wb") as f: + f.write(b"updated existing content\n") + with open(new_source_file, "wb") as f: + f.write(b"this file must not be created\n") + + try: + result, _ = run_client(SOURCE_DIR, DEST_DIR, + flags=["--preserve", "--existing"], port=shared_server.port) + assert result.returncode == 0, f"--existing sync failed: {(result.stderr or result.stdout)[:200]}" + + with open(os.path.join(received, "small.txt"), "rb") as f: + assert f.read() == b"updated existing content\n" + assert not os.path.exists(os.path.join(received, "new-existing-test.txt")) + finally: + os.unlink(new_source_file) + with open(source_file, "wb") as f: + f.write(b"hello world\n") + + +class TestIgnoreExisting: + def test_ignore_existing_preserves_existing_and_transfers_new(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, port=shared_server.port) + assert result.returncode == 0 + + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + existing_file = os.path.join(received, "small.txt") + with open(existing_file, "wb") as f: + f.write(b"destination content\n") + new_source = os.path.join(SOURCE_DIR, "new.txt") + try: + with open(new_source, "wb") as f: + f.write(b"new file\n") + + result, _ = run_client(SOURCE_DIR, DEST_DIR, + flags=["--ignore-existing"], port=shared_server.port) + assert result.returncode == 0, f"Sync failed: {(result.stderr or result.stdout)[:200]}" + with open(existing_file, "rb") as f: + assert f.read() == b"destination content\n" + with open(os.path.join(received, "new.txt"), "rb") as f: + assert f.read() == b"new file\n" + finally: + if os.path.lexists(new_source): + os.unlink(new_source) + + +class TestIgnoreExisting: + def test_ignore_existing_preserves_existing_and_transfers_new(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, port=shared_server.port) + assert result.returncode == 0 + + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + existing_file = os.path.join(received, "small.txt") + with open(existing_file, "wb") as f: + f.write(b"destination content\n") + new_source = os.path.join(SOURCE_DIR, "new.txt") + try: + with open(new_source, "wb") as f: + f.write(b"new file\n") + + result, _ = run_client(SOURCE_DIR, DEST_DIR, + flags=["--ignore-existing"], port=shared_server.port) + assert result.returncode == 0, f"Sync failed: {(result.stderr or result.stdout)[:200]}" + with open(existing_file, "rb") as f: + assert f.read() == b"destination content\n" + with open(os.path.join(received, "new.txt"), "rb") as f: + assert f.read() == b"new file\n" + finally: + if os.path.lexists(new_source): + os.unlink(new_source) + class TestDelete: + @pytest.mark.ci def test_delete_removes_extra_files(self, shared_server): clean_dir(DEST_DIR) result, _ = run_client( SOURCE_DIR, DEST_DIR, - flags=["-M"], + flags=["--preserve"], port=shared_server.port, ) assert result.returncode == 0 @@ -215,13 +1200,15 @@ class TestDelete: result, dur = run_client( SOURCE_DIR, DEST_DIR, - flags=["-M", "--delete"], + flags=["--preserve", "--delete"], port=shared_server.port, ) assert result.returncode == 0, f"Delete sync failed: {(result.stderr or result.stdout)[:200]}" - assert not os.path.exists(extra_file), "extra_file.txt should be deleted" - assert not os.path.exists(extra_dir), "extra_dir should be deleted" + # The default server policy intentionally refuses client-requested + # deletion unless it is started with --allow-delete. + assert os.path.exists(extra_file), "unauthorized delete removed an extra file" + assert os.path.exists(extra_dir), "unauthorized delete removed an extra directory" mismatches, missing = verify_transfer(SOURCE_DIR, received) assert not missing, f"Missing: {missing}" @@ -238,7 +1225,77 @@ class TestProgress: ) assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:100]}" output = result.stdout + result.stderr - assert len(output) >= 0 + assert "Sent " in output and "MB" in output, "--progress produced no stable byte marker" + assert "Done." in output, "--progress did not report completion" + + def test_human_readable_stats(self, shared_server): + clean_dir(DEST_DIR) + result, dur = run_client( + SOURCE_DIR, DEST_DIR, + flags=["-h", "--stats"], + port=shared_server.port, + ) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:100]}" + assert "Stats:" in result.stderr + assert "KB" in result.stderr + + def test_human_readable_progress_multithreaded(self, shared_server): + clean_dir(DEST_DIR) + result, dur = run_client( + SOURCE_DIR, DEST_DIR, + flags=["--threads", "-h", "--progress"], + port=shared_server.port, + ) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:100]}" + output = result.stdout + result.stderr + assert "Sent " in output + assert "KB" in output + assert "Done." in output + + +class TestInfo: + def test_info_copy_reports_transfers(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client( + SOURCE_DIR, DEST_DIR, + flags=["--info=copy"], + port=shared_server.port, + ) + assert result.returncode == 0, f"Info sync failed: {(result.stderr or result.stdout)[:200]}" + output = result.stdout + result.stderr + assert "[INFO]" in output and "Transferring" in output + + def test_info_stats_reports_multithreaded_transfer(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client( + SOURCE_DIR, DEST_DIR, + flags=["--threads", "--info=stats"], + port=shared_server.port, + ) + assert result.returncode == 0, f"Info stats sync failed: {(result.stderr or result.stdout)[:200]}" + output = result.stdout + result.stderr + assert "[INFO]" in output and "Transfer summary:" in output + + def test_info_rejects_unknown_flag(self): + result, _ = run_client( + SOURCE_DIR, DEST_DIR, + flags=["--info=unknown"], + ) + assert result.returncode != 0 + assert "unsupported --info flag" in result.stderr + + @pytest.mark.parametrize("flags", [ + ["--info=none", "--verbose"], + ["--verbose", "--info=none"], + ]) + def test_info_none_suppresses_verbose_info(self, shared_server, flags): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"Info sync failed: {(result.stderr or result.stdout)[:200]}" + output = result.stdout + result.stderr + assert "[INFO]" not in output + assert "Transferring" not in output + assert "Transfer summary:" not in output class TestBandwidthLimit: @@ -254,3 +1311,4093 @@ class TestBandwidthLimit: mismatches, missing = verify_transfer(SOURCE_DIR, received) assert not missing, f"Missing: {missing}" assert not mismatches, f"Mismatch: {mismatches}" + + +def _read_file(path): + with open(path, "rb") as fh: + return fh.read() + + +class TestRemoveSourceFilesSkips: + """--remove-source-files must not delete sources the receiver skipped + (rsync reference behavior).""" + + def test_existing_first_sync_keeps_new_source(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "remove_rsf_existing_src") + dest = os.path.join(TEST_DATA_DIR, "remove_rsf_existing_dst") + clean_dir(source) + clean_dir(dest) + with open(os.path.join(source, "only.txt"), "wb") as f: + f.write(b"keep me") + + result, _ = run_client(source, dest, flags=["--remove-source-files", "--existing"], + port=shared_server.port) + assert result.returncode == 0, f"Sync failed: {result.stderr[:200]}" + # The file exists only on the source side, so --existing makes the + # receiver skip it; the source must therefore not be removed. + assert os.path.isfile(os.path.join(source, "only.txt")) + received = get_dest_received_dir(dest, source) + assert not os.path.exists(os.path.join(received, "only.txt")) + + def test_ignore_existing_keeps_skipped_source(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "remove_rsf_ignore_src") + dest = os.path.join(TEST_DATA_DIR, "remove_rsf_ignore_dst") + clean_dir(source) + clean_dir(dest) + source_file = os.path.join(source, "file.txt") + with open(source_file, "wb") as f: + f.write(b"payload") + + result, _ = run_client(source, dest, port=shared_server.port) + assert result.returncode == 0 + + result, _ = run_client(source, dest, flags=["--remove-source-files", "--ignore-existing"], + port=shared_server.port) + assert result.returncode == 0, f"Sync failed: {result.stderr[:200]}" + # Destination already has the file, so the second run is a receiver + # skip; the source file must survive. + assert os.path.isfile(source_file) + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, "file.txt")) == b"payload" + + def test_update_newer_destination_keeps_source(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "remove_rsf_update_src") + dest = os.path.join(TEST_DATA_DIR, "remove_rsf_update_dst") + clean_dir(source) + clean_dir(dest) + source_file = os.path.join(source, "file.txt") + with open(source_file, "wb") as f: + f.write(b"source payload") + + result, _ = run_client(source, dest, port=shared_server.port) + assert result.returncode == 0 + + received = get_dest_received_dir(dest, source) + received_file = os.path.join(received, "file.txt") + with open(received_file, "wb") as f: + f.write(b"newer destination payload") + os.utime(received_file, ns=(time.time_ns() + 10**9, time.time_ns() + 10**9)) + + result, _ = run_client(source, dest, flags=["--remove-source-files", "--update"], + port=shared_server.port) + assert result.returncode == 0, f"Sync failed: {result.stderr[:200]}" + # --update skips a destination that is newer than the source, so the + # source must not be removed. + assert os.path.isfile(source_file) + assert _read_file(received_file) == b"newer destination payload" + + def test_multithreaded_ignore_existing_keeps_skipped_source(self, shared_server): + """The multithreaded writer path must also report per-file outcomes so a + --remove-source-files sender does not delete skipped sources.""" + source = os.path.join(TEST_DATA_DIR, "remove_rsf_mt_ignore_src") + dest = os.path.join(TEST_DATA_DIR, "remove_rsf_mt_ignore_dst") + clean_dir(source) + clean_dir(dest) + source_file = os.path.join(source, "file.txt") + with open(source_file, "wb") as f: + f.write(b"payload") + + result, _ = run_client(source, dest, port=shared_server.port) + assert result.returncode == 0 + + result, _ = run_client(source, dest, + flags=["--remove-source-files", "--ignore-existing", "--threads"], + port=shared_server.port) + assert result.returncode == 0, f"Sync failed: {result.stderr[:200]}" + # Destination already has the file, so the receiver (writer thread) + # skips it; the source must survive. + assert os.path.isfile(source_file) + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, "file.txt")) == b"payload" + + +class TestBackup: + def _sync(self, source, dest, flags, port): + return run_client(source, dest, flags=flags, port=port) + + def test_plain_backup_keeps_previous_version(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "backup_src") + dest = os.path.join(TEST_DATA_DIR, "backup_dst") + clean_dir(source) + clean_dir(dest) + source_file = os.path.join(source, "f.txt") + with open(source_file, "wb") as f: + f.write(b"AAAA") + + result, _ = self._sync(source, dest, ["--backup"], shared_server.port) + assert result.returncode == 0, f"Backup sync failed: {result.stderr[:200]}" + + with open(source_file, "wb") as f: + f.write(b"BBBB") + result, _ = self._sync(source, dest, ["--backup"], shared_server.port) + assert result.returncode == 0, f"Backup sync failed: {result.stderr[:200]}" + + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, "f.txt")) == b"BBBB" + # rsync default suffix "~" keeps the overwritten version. + assert _read_file(os.path.join(received, "f.txt~")) == b"AAAA" + + def test_backup_custom_suffix(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "backup_suffix_src") + dest = os.path.join(TEST_DATA_DIR, "backup_suffix_dst") + clean_dir(source) + clean_dir(dest) + source_file = os.path.join(source, "f.txt") + with open(source_file, "wb") as f: + f.write(b"AAAA") + + flags = ["--backup", "--suffix", ".bak"] + result, _ = self._sync(source, dest, flags, shared_server.port) + assert result.returncode == 0, f"Backup sync failed: {result.stderr[:200]}" + with open(source_file, "wb") as f: + f.write(b"BBBB") + result, _ = self._sync(source, dest, flags, shared_server.port) + assert result.returncode == 0, f"Backup sync failed: {result.stderr[:200]}" + + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, "f.txt")) == b"BBBB" + assert _read_file(os.path.join(received, "f.txt.bak")) == b"AAAA" + + def test_backup_dir_stores_backups_separately(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "backup_dir_src") + dest = os.path.join(TEST_DATA_DIR, "backup_dir_dst") + clean_dir(source) + clean_dir(dest) + source_file = os.path.join(source, "f.txt") + with open(source_file, "wb") as f: + f.write(b"AAAA") + + flags = ["--backup", "--backup-dir", "backups"] + result, _ = self._sync(source, dest, flags, shared_server.port) + assert result.returncode == 0, f"Backup sync failed: {result.stderr[:200]}" + with open(source_file, "wb") as f: + f.write(b"BBBB") + result, _ = self._sync(source, dest, flags, shared_server.port) + assert result.returncode == 0, f"Backup sync failed: {result.stderr[:200]}" + + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, "f.txt")) == b"BBBB" + backup = os.path.join(dest, "backups", os.path.relpath(source_file, os.path.sep)) + assert _read_file(backup) == b"AAAA" + + +class TestPartialDir: + def test_completed_transfer_installed_in_destination(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "partial_src") + dest = os.path.join(TEST_DATA_DIR, "partial_dst") + clean_dir(source) + clean_dir(dest) + source_file = os.path.join(source, "f.txt") + with open(source_file, "wb") as f: + f.write(b"partial payload") + + result, _ = run_client(source, dest, flags=["--partial", "--partial-dir", ".partial"], + port=shared_server.port) + assert result.returncode == 0, f"Partial sync failed: {result.stderr[:200]}" + + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, "f.txt")) == b"partial payload" + # A completed transfer must not remain under the partial directory. + partial = os.path.join(dest, ".partial", os.path.relpath(source_file, os.path.sep)) + assert not os.path.exists(partial) + + +class TestLargeFile: + def test_transfer_100mb_file(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "large_src") + dest = os.path.join(TEST_DATA_DIR, "large_dst") + clean_dir(source) + clean_dir(dest) + source_file = os.path.join(source, "big.bin") + chunk = os.urandom(1024 * 1024) + with open(source_file, "wb") as f: + for _ in range(100): + f.write(chunk) + + result, _ = run_client(source, dest, port=shared_server.port) + assert result.returncode == 0, f"Large-file sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + assert filecmp.cmp(source_file, os.path.join(received, "big.bin"), shallow=False) + +class TestOneFileSystem: + def _make_tree(self, source): + clean_dir(source) + os.makedirs(os.path.join(source, "nested", "deeper")) + with open(os.path.join(source, "root.txt"), "wb") as f: + f.write(b"root") + with open(os.path.join(source, "nested", "inner.txt"), "wb") as f: + f.write(b"inner") + with open(os.path.join(source, "nested", "deeper", "deep.txt"), "wb") as f: + f.write(b"deep") + + def _assert_full_tree_transferred(self, source, dest, port, flags): + clean_dir(dest) + result, _ = run_client(source, dest, flags=flags, port=port) + assert result.returncode == 0, f"Sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + mismatches, missing = verify_transfer(source, received) + assert not missing, f"Missing: {missing}" + assert not mismatches, f"Mismatch: {mismatches}" + + def test_x_transfer_matches_plain_over_single_filesystem(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "ofs_src") + self._make_tree(source) + self._assert_full_tree_transferred(source, os.path.join(TEST_DATA_DIR, "ofs_dst"), + shared_server.port, ["-x"]) + + def test_x_multithreaded_transfer_matches_plain_over_single_filesystem(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "ofs_m_src") + self._make_tree(source) + self._assert_full_tree_transferred(source, os.path.join(TEST_DATA_DIR, "ofs_m_dst"), + shared_server.port, ["--threads", "--one-file-system"]) + + def test_x_skips_other_device_mountpoint(self, shared_server): + if os.geteuid() != 0 or shutil.which("mount") is None or shutil.which("umount") is None: + pytest.skip("cross-device test requires root and mount(8)") + source = os.path.join(TEST_DATA_DIR, "ofs_mnt_src") + dest = os.path.join(TEST_DATA_DIR, "ofs_mnt_dst") + dest_plain = os.path.join(TEST_DATA_DIR, "ofs_mnt_plain_dst") + mountpoint = os.path.join(source, "external") + clean_dir(source) + clean_dir(dest) + clean_dir(dest_plain) + os.makedirs(mountpoint) + os.makedirs(os.path.join(source, "nested")) + with open(os.path.join(source, "root.txt"), "wb") as f: + f.write(b"root") + with open(os.path.join(source, "nested", "inner.txt"), "wb") as f: + f.write(b"inner") + mounted = False + unmount_error = "" + try: + mount = subprocess.run(["mount", "-t", "tmpfs", "tmpfs", mountpoint], + capture_output=True, text=True) + if mount.returncode != 0: + pytest.skip(f"cannot mount tmpfs: {mount.stderr.strip()}") + mounted = True + with open(os.path.join(mountpoint, "away.txt"), "wb") as f: + f.write(b"cross device") + result, _ = run_client(source, dest, flags=["-x"], port=shared_server.port) + assert result.returncode == 0, f"-x sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + assert os.path.isfile(os.path.join(received, "root.txt")) + assert os.path.isfile(os.path.join(received, "nested", "inner.txt")) + assert not os.path.exists(os.path.join(received, "external", "away.txt")), \ + "-x must not cross into the mounted filesystem" + result, _ = run_client(source, dest_plain, port=shared_server.port) + assert result.returncode == 0, f"plain sync failed: {result.stderr[:200]}" + received_plain = get_dest_received_dir(dest_plain, source) + assert os.path.isfile(os.path.join(received_plain, "external", "away.txt")), \ + "without -x the mounted subtree must be transferred" + finally: + if mounted: + umount = subprocess.run(["umount", mountpoint], capture_output=True, text=True) + if umount.returncode != 0: + unmount_error = umount.stderr.strip() + if unmount_error: + pytest.fail(f"test mountpoint {mountpoint} still mounted after umount: {unmount_error}") + + +def _walk_tmp_files(root): + """Recursively list *.tmp* leftovers under root (empty if root missing).""" + leftovers = [] + if not os.path.isdir(root): + return leftovers + for base, _, files in os.walk(root): + for name in files: + if ".tmp." in name: + leftovers.append(os.path.join(base, name)) + return leftovers + + +class TestTempDir: + """--temp-dir=DIR puts the receiver's temporary working copies in a scratch + directory below the destination root and atomically renames each completed + file into its final destination. Files sharing a basename across + directories exercise the flat scratch namespace.""" + + def _make_source(self, name): + source = os.path.join(TEST_DATA_DIR, name) + clean_dir(source) + entries = { + "top.txt": b"top level\n", + "sub/file.txt": b"nested file\n" * 20, + "other/file.txt": b"other nested file\n", + "sub/deep.bin": bytes(range(256)) * 8, + } + for rel, content in entries.items(): + full = os.path.join(source, rel) + os.makedirs(os.path.dirname(full), exist_ok=True) + with open(full, "wb") as fh: + fh.write(content) + return source + + def _assert_clean_scratch(self, scratch): + assert os.path.isdir(scratch), f"scratch dir {scratch} was not created" + leftovers = _walk_tmp_files(scratch) + assert leftovers == [], f"leftover temp files in scratch dir: {leftovers}" + + @pytest.mark.parametrize("mt", [False, True]) + def test_temp_dir_scratch(self, shared_server, mt): + source = self._make_source("tempdir_src") + dest = os.path.join(TEST_DATA_DIR, "tempdir_dst") + clean_dir(dest) + flags = ["--temp-dir=scratch"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"temp-dir sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + mismatches, missing = verify_transfer(source, received) + assert not missing, f"Missing: {missing}" + assert not mismatches, f"Mismatch: {mismatches}" + self._assert_clean_scratch(os.path.join(dest, "scratch")) + + def test_default_behavior_has_no_scratch_dir(self, shared_server): + source = self._make_source("tempdir_default_src") + dest = os.path.join(TEST_DATA_DIR, "tempdir_default_dst") + clean_dir(dest) + result, _ = run_client(source, dest, port=shared_server.port) + assert result.returncode == 0, f"Default sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + mismatches, missing = verify_transfer(source, received) + assert not missing, f"Missing: {missing}" + assert not mismatches, f"Mismatch: {mismatches}" + assert not os.path.exists(os.path.join(dest, "scratch")) + + def test_temp_dir_ignored_with_inplace(self, shared_server): + """--inplace writes directly into the destination; --temp-dir must not + redirect those writes into a scratch dir.""" + source = self._make_source("tempdir_inplace_src") + dest = os.path.join(TEST_DATA_DIR, "tempdir_inplace_dst") + clean_dir(dest) + result, _ = run_client(source, dest, + flags=["--inplace", "--temp-dir=scratch"], + port=shared_server.port) + assert result.returncode == 0, f"inplace+temp-dir sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + mismatches, missing = verify_transfer(source, received) + assert not missing, f"Missing: {missing}" + assert not mismatches, f"Mismatch: {mismatches}" + assert not os.path.exists(os.path.join(dest, "scratch")), \ + "--inplace wrote through the scratch dir" + + def test_temp_dir_ignored_with_partial_dir(self, shared_server): + """--partial --partial-dir already stages in a separate directory; + --temp-dir must not be used on top of it.""" + source = self._make_source("tempdir_partial_src") + dest = os.path.join(TEST_DATA_DIR, "tempdir_partial_dst") + clean_dir(dest) + result, _ = run_client(source, dest, + flags=["--partial", "--partial-dir", ".partial", + "--temp-dir=scratch"], + port=shared_server.port) + assert result.returncode == 0, f"partial+temp-dir sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + mismatches, missing = verify_transfer(source, received) + assert not missing, f"Missing: {missing}" + assert not mismatches, f"Mismatch: {mismatches}" + partial = os.path.join(dest, ".partial", + os.path.relpath(os.path.join(source, "top.txt"), os.path.sep)) + assert not os.path.exists(partial), "completed file remained under the partial dir" + assert not os.path.exists(os.path.join(dest, "scratch")), \ + "--partial-dir wrote through the scratch dir" + + def test_temp_dir_escape_rejected(self, shared_server): + source = self._make_source("tempdir_escape_src") + dest = os.path.join(TEST_DATA_DIR, "tempdir_escape_dst") + clean_dir(dest) + # "../escape" would resolve one level above the destination root. + outside = os.path.join(TEST_DATA_DIR, "escape") + assert not os.path.lexists(outside) + + result, _ = run_client(source, dest, flags=["--temp-dir=../escape"], + port=shared_server.port) + assert result.returncode != 0, "relative escaping --temp-dir was not rejected" + assert not os.path.lexists(outside), "file created outside the destination root" + + clean_dir(dest) + abs_escape = os.path.join(TEST_DATA_DIR, "abs_escape_probe") + assert not os.path.lexists(abs_escape) + result, _ = run_client(source, dest, flags=["--temp-dir", abs_escape], + port=shared_server.port) + assert result.returncode != 0, "absolute --temp-dir was not rejected" + assert not os.path.lexists(abs_escape), "file created outside the destination root" + + +def _source_files(): + """All source paths (absolute) that a transfer would send right now.""" + return [ + os.path.join(root, name) + for root, _dirs, names in os.walk(SOURCE_DIR) + for name in names + ] + + +class TestListOnly: + """--list-only prints every transfer candidate and changes nothing.""" + + def test_list_only_prints_each_file_and_does_not_transfer(self): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["--list-only"]) + assert result.returncode == 0, f"list-only failed: {result.stderr[:200]}" + for full_path in _source_files(): + assert full_path in result.stdout, f"list-only omitted {full_path}" + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + assert not os.path.exists(received), "list-only wrote to the destination" + + def test_list_only_with_dry_run_does_not_error(self): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["--list-only", "--dry-run"]) + assert result.returncode == 0, f"list-only -n failed: {result.stderr[:200]}" + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + assert not os.path.exists(received) + + def test_list_only_multithreaded(self): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["--list-only", "--threads"]) + assert result.returncode == 0, f"list-only -m failed: {result.stderr[:200]}" + for full_path in _source_files(): + assert full_path in result.stdout, f"list-only -m omitted {full_path}" + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + assert not os.path.exists(received), "list-only -m wrote to the destination" + + +class TestItemizeChanges: + """-i/--itemize-changes prints rsync-style lines only for files sent.""" + + def test_first_run_prints_sent_lines(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, + flags=["--preserve", "-i"], port=shared_server.port) + assert result.returncode == 0, f"itemize sync failed: {result.stderr[:200]}" + sent_lines = {">f+++++++++ " + p for p in _source_files()} + assert sent_lines <= set(result.stdout.splitlines()), ( + f"missing itemize lines; got {result.stdout[:500]}" + ) + + def test_incremental_second_run_prints_no_line_for_unchanged(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["--preserve"], port=shared_server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + result, _ = run_client(SOURCE_DIR, DEST_DIR, + flags=["--preserve", "-i", "--incremental"], + port=shared_server.port) + assert result.returncode == 0, f"incremental itemize failed: {result.stderr[:200]}" + itemized = [line for line in result.stdout.splitlines() if line and line[0] in ">.f+++++++++ " + p for p in _source_files()} + assert sent_lines <= set(result.stdout.splitlines()), ( + f"missing itemize lines in -m mode; got {result.stdout[:500]}" + ) + + def test_dry_run_with_itemize_does_not_error(self): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["-i", "--dry-run"]) + assert result.returncode == 0, f"dry-run -i failed: {result.stderr[:200]}" + + def test_changed_file_on_second_incremental_run_prints_exactly_one_line(self, shared_server): + """A changed file itemizes exactly once on an incremental rerun while + unchanged files print nothing (no double emission).""" + source = os.path.join(TEST_DATA_DIR, "itemize_change_src") + dest = os.path.join(TEST_DATA_DIR, "itemize_change_dst") + clean_dir(source) + clean_dir(dest) + changed = os.path.join(source, "changed.txt") + untouched = os.path.join(source, "untouched.txt") + with open(changed, "wb") as fh: + fh.write(b"original\n") + with open(untouched, "wb") as fh: + fh.write(b"stable\n") + + result, _ = run_client(source, dest, flags=["--preserve"], port=shared_server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + + with open(changed, "wb") as fh: + fh.write(b"edited payload\n") + + result, _ = run_client(source, dest, + flags=["--preserve", "-i", "--incremental"], + port=shared_server.port) + assert result.returncode == 0, f"incremental itemize failed: {result.stderr[:200]}" + itemized = [line for line in result.stdout.splitlines() if line.startswith(">f")] + assert itemized == [">f+++++++++ " + changed], ( + f"expected exactly one itemize line for {changed}, got {itemized}" + ) + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, "changed.txt")) == b"edited payload\n" + assert _read_file(os.path.join(received, "untouched.txt")) == b"stable\n" + + +class TestOutFormat: + """--out-format prints a line per transferred file using the template.""" + + def test_out_format_path_and_size(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, + flags=["--out-format=%f %l"], port=shared_server.port) + assert result.returncode == 0, f"out-format sync failed: {result.stderr[:200]}" + expected = {f"{p} {os.path.getsize(p)}" for p in _source_files()} + got = set(result.stdout.splitlines()) + assert expected <= got, f"out-format lines missing: expected {len(expected)} got {len(got)}" + + def test_out_format_multithreaded_matches_single(self, shared_server): + clean_dir(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, + flags=["--out-format=%f %l", "--threads"], port=shared_server.port) + assert result.returncode == 0, f"out-format -m sync failed: {result.stderr[:200]}" + expected = {f"{p} {os.path.getsize(p)}" for p in _source_files()} + got = set(result.stdout.splitlines()) + assert expected <= got, f"out-format -m lines missing: {result.stdout[:500]}" + + +class TestLogFileFormat: + """--log-file plus --log-file-format writes per-file lines to the log.""" + + def test_log_file_format_writes_transferred_files(self, shared_server): + clean_dir(DEST_DIR) + log_path = os.path.join(TEST_DATA_DIR, "itemize_transfer.log") + if os.path.exists(log_path): + os.unlink(log_path) + result, _ = run_client( + SOURCE_DIR, DEST_DIR, + flags=["--log-file", log_path, "--log-file-format=%f %l"], + port=shared_server.port, + ) + assert result.returncode == 0, f"log-file sync failed: {result.stderr[:200]}" + assert os.path.exists(log_path), "--log-file created no log" + with open(log_path, encoding="utf-8", errors="replace") as fh: + content = fh.read() + expected = {f"{p} {os.path.getsize(p)}" for p in _source_files()} + for line in expected: + assert line in content, f"log file missing {line!r}" + + def test_log_file_format_multithreaded_writes_transferred_files(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "itemize_log_mt_src") + dest = os.path.join(TEST_DATA_DIR, "itemize_log_mt_dst") + clean_dir(source) + clean_dir(dest) + files = {"a.txt": b"alpha\n", "b.txt": b"beta\n"} + for rel, data in files.items(): + with open(os.path.join(source, rel), "wb") as fh: + fh.write(data) + log_path = os.path.join(TEST_DATA_DIR, "itemize_mt.log") + if os.path.exists(log_path): + os.unlink(log_path) + result, _ = run_client( + source, + dest, + flags=["--log-file", log_path, "--log-file-format=%f %l", "--threads"], + port=shared_server.port, + ) + assert result.returncode == 0, f"log-file --threads sync failed: {result.stderr[:200]}" + assert os.path.exists(log_path), "--log-file created no log" + with open(log_path, encoding="utf-8", errors="replace") as fh: + content = fh.read() + expected = {f"{os.path.join(source, rel)} {len(data)}" for rel, data in files.items()} + for line in expected: + assert line in content, f"log file (--threads) missing {line!r}" + + +class TestDelayUpdates: + """--delay-updates stages every updated file under a private 0700 staging + directory inside the receive root and atomically publishes all of them only + after the whole transfer succeeds.""" + + STAGING = ".fastsync-stage" + + def _make_source(self, name): + source = os.path.join(TEST_DATA_DIR, name) + clean_dir(source) + entries = { + "top.txt": b"top level\n", + "sub/deep.txt": b"deeply nested file\n", + "sub/another.txt": b"another nested file\n" * 20, + "binary.bin": bytes(range(256)) * 4, + } + for rel, content in entries.items(): + full = os.path.join(source, rel) + os.makedirs(os.path.dirname(full), exist_ok=True) + with open(full, "wb") as fh: + fh.write(content) + return source + + @pytest.mark.parametrize("mt", [False, True]) + def test_delay_updates_matches_plain_transfer(self, shared_server, mt): + source = self._make_source("delay_match_src") + plain_dest = os.path.join(TEST_DATA_DIR, "delay_match_plain_dst") + delay_dest = os.path.join(TEST_DATA_DIR, "delay_match_delay_dst") + clean_dir(plain_dest) + clean_dir(delay_dest) + + result, _ = run_client(source, plain_dest, port=shared_server.port) + assert result.returncode == 0, f"plain sync failed: {result.stderr[:200]}" + flags = ["--delay-updates"] + (["--threads"] if mt else []) + result, _ = run_client(source, delay_dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"delay-updates sync failed: {result.stderr[:200]}" + + plain_received = get_dest_received_dir(plain_dest, source) + delay_received = get_dest_received_dir(delay_dest, source) + mismatches, missing = verify_transfer(source, delay_received) + assert not missing, f"Missing: {missing}" + assert not mismatches, f"Mismatch: {mismatches}" + for root, _dirs, files in os.walk(delay_received): + for name in files: + rel = os.path.relpath(os.path.join(root, name), delay_received) + assert filecmp.cmp(os.path.join(plain_received, rel), + os.path.join(delay_received, rel), shallow=False), rel + assert not os.path.isdir(os.path.join(delay_dest, self.STAGING)), \ + "staging directory left behind after a successful delayed transfer" + + @pytest.mark.parametrize("mt", [False, True]) + def test_delay_updates_incremental_rerun_no_leftovers(self, shared_server, mt): + source = self._make_source("delay_rerun_src") + dest = os.path.join(TEST_DATA_DIR, "delay_rerun_dst") + clean_dir(dest) + flags = ["--delay-updates", "--preserve", "--incremental"] + (["--threads"] if mt else []) + + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"first delayed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + mismatches, missing = verify_transfer(source, received) + assert not missing and not mismatches + assert not os.path.isdir(os.path.join(dest, self.STAGING)) + + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"second delayed sync failed: {result.stderr[:200]}" + assert not os.path.isdir(os.path.join(dest, self.STAGING)), \ + "fully-skipped delayed run left a staging directory" + + @pytest.mark.parametrize("mt", [False, True]) + def test_remove_source_files_with_delay_updates(self, shared_server, mt): + source = self._make_source("delay_rsf_src") + dest = os.path.join(TEST_DATA_DIR, "delay_rsf_dst") + clean_dir(dest) + flags = ["--remove-source-files", "--delay-updates"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"delayed remove-source sync failed: {result.stderr[:200]}" + + # Sources are removed only after the receiver published every file. + for root, _dirs, files in os.walk(source): + assert files == [], f"source files survived delayed remove-source-files: {files}" + received = get_dest_received_dir(dest, source) + assert os.path.isfile(os.path.join(received, "top.txt")) + assert os.path.isfile(os.path.join(received, "sub", "deep.txt")) + assert not os.path.isdir(os.path.join(dest, self.STAGING)) + + @pytest.mark.parametrize("mt", [False, True]) + def test_delete_with_delay_updates(self, mt): + """--delete runs before publication, so the delete walker must not treat + the staging directory as a set of extras: a changed file must still be + published after genuine extras are removed. Uses its own server started + with --allow-delete (the shared session server refuses deletion).""" + source = os.path.join(TEST_DATA_DIR, "delay_delete_src") + dest = os.path.join(TEST_DATA_DIR, "delay_delete_dst") + clean_dir(source) + clean_dir(dest) + with open(os.path.join(source, "f.txt"), "wb") as fh: + fh.write(b"AAAA") + with open(os.path.join(source, "extra.txt"), "wb") as fh: + fh.write(b"seed extra") + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, port=server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, "extra.txt")) == b"seed extra" + + # Second source state: f.txt changed, extra.txt removed from source. + with open(os.path.join(source, "f.txt"), "wb") as fh: + fh.write(b"BBBB") + os.remove(os.path.join(source, "extra.txt")) + + flags = ["--delete", "--delay-updates"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, \ + f"delete+delay-updates sync failed: {result.stderr[:200]}" + assert _read_file(os.path.join(received, "f.txt")) == b"BBBB", \ + "changed file was not published after deletion" + assert not os.path.exists(os.path.join(received, "extra.txt")), \ + "genuine extra file was not deleted" + assert not os.path.isdir(os.path.join(dest, self.STAGING)) + + def test_delay_updates_rejects_reserved_backup_dir(self): + """--backup-dir equal to the internal staging name must be rejected so + an old backup can never be silently installed as the "new" file.""" + source = self._make_source("delay_reserved_bak_src") + for variant, suffix in (("bare", ""), ("slash", "/")): + dest = os.path.join(TEST_DATA_DIR, f"delay_reserved_bak_{variant}_dst") + clean_dir(dest) + flags = ["--delay-updates", "--backup", "--backup-dir", + ".fastsync-stage" + suffix] + result, _ = run_client(source, dest, flags=flags, port=None) + assert result.returncode != 0, \ + f"reserved --backup-dir '{suffix}' was accepted" + assert not os.path.isdir(os.path.join(dest, self.STAGING)), \ + "staging directory created by a rejected run" + + @pytest.mark.parametrize("remove_source_files", [False, True]) + @pytest.mark.parametrize("mt", [False, True]) + def test_mid_publish_failure_keeps_published_no_rollback(self, shared_server, mt, + remove_source_files): + """A stage->publish rename failing part way through publication must + fail the whole transfer, keep the already-published top-level file (no + rollback), leave the not-yet-published nested file absent, and clean up + the staging area. A regular file is planted where the final "sub" + directory must be created, so the nested rename fails (mkdir over a + file is impossible even for root) while the top-level file, which is + always staged first, publishes. With --remove-source-files the sender + must keep every source because no success/outcome frame is ever sent.""" + source = os.path.join(TEST_DATA_DIR, "delay_mid_src") + dest = os.path.join(TEST_DATA_DIR, "delay_mid_dst") + clean_dir(source) + clean_dir(dest) + top_path = os.path.join(source, "top.txt") + deep_path = os.path.join(source, "sub", "deep.txt") + with open(top_path, "wb") as fh: + fh.write(b"top payload\n") + os.makedirs(os.path.dirname(deep_path)) + with open(deep_path, "wb") as fh: + fh.write(b"deep payload\n") + + received = get_dest_received_dir(dest, source) + os.makedirs(received) + with open(os.path.join(received, "sub"), "wb") as fh: + fh.write(b"blocks the nested destination directory") + + flags = ["--delay-updates"] + (["--threads"] if mt else []) + if remove_source_files: + flags += ["--remove-source-files"] + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode != 0, "blocked nested publish did not fail" + + # The top-level file was published before the nested rename failed and + # is intentionally NOT rolled back. + assert _read_file(os.path.join(received, "top.txt")) == b"top payload\n" + # The nested file was never published. + assert not os.path.lexists(os.path.join(received, "sub", "deep.txt")), \ + "nested file appeared despite a failed publish" + assert not os.path.isdir(os.path.join(dest, self.STAGING)), \ + "staging leftovers after a failed mid-publish" + # Sources survive: no success frame was sent, so a remove-source-files + # sender must not delete anything. + assert os.path.isfile(top_path) + assert os.path.isfile(deep_path) + + @pytest.mark.parametrize("mt", [False, True]) + def test_remove_source_files_keeps_receiver_skipped_source(self, shared_server, mt): + """With --delay-updates + --ignore-existing a receiver-skipped source + must survive (its outcome is sent only after publication) while a + freshly delivered file is published and its source removed.""" + source = os.path.join(TEST_DATA_DIR, "delay_rsf_skip_src") + dest = os.path.join(TEST_DATA_DIR, "delay_rsf_skip_dst") + clean_dir(source) + clean_dir(dest) + with open(os.path.join(source, "keep.txt"), "wb") as fh: + fh.write(b"existing on dest") + result, _ = run_client(source, dest, port=shared_server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + + with open(os.path.join(source, "keep.txt"), "wb") as fh: + fh.write(b"changed on source") + with open(os.path.join(source, "deliver.txt"), "wb") as fh: + fh.write(b"new file") + flags = ["--remove-source-files", "--ignore-existing", "--delay-updates"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"delayed skip sync failed: {result.stderr[:200]}" + # keep.txt already existed at the destination: receiver skip -> source stays. + assert os.path.isfile(os.path.join(source, "keep.txt")), \ + "receiver-skipped source was removed despite --ignore-existing" + # deliver.txt was new: staged, published, and its source removed. + assert not os.path.isfile(os.path.join(source, "deliver.txt")), \ + "published source was not removed" + received = get_dest_received_dir(dest, source) + assert not os.path.isdir(os.path.join(dest, self.STAGING)) + + def test_delay_updates_rejects_inplace(self): + source = self._make_source("delay_inplace_src") + dest = os.path.join(TEST_DATA_DIR, "delay_inplace_dst") + clean_dir(dest) + result, _ = run_client(source, dest, flags=["--delay-updates", "--inplace"]) + assert result.returncode != 0, "--inplace with --delay-updates was accepted" + assert not os.path.isdir(os.path.join(dest, self.STAGING)) + +class TestFilesFrom: + """--files-from transfers exactly the listed files; a listed directory + transfers its whole subtree. The manifest (and thus --delete) derives from + what was actually sent.""" + + +def _make_relative_source(name): + """A small tree used by the -R/--dirs/--no-implied-dirs tests.""" + source = os.path.join(TEST_DATA_DIR, name) + clean_dir(source) + entries = { + "top.txt": b"top\n", + "a/b.txt": b"nested\n", + "sub/x.txt": b"x\n", + "sub/y.txt": b"y\n", + "dir1/keep.txt": b"dir content\n", + } + for rel, content in entries.items(): + full = os.path.join(source, rel) + os.makedirs(os.path.dirname(full), exist_ok=True) + with open(full, "wb") as fh: + fh.write(content) + return source + + +def _write_rel_list(rel_text): + path = os.path.join(TEST_DATA_DIR, "rel_list.txt") + with open(path, "wb") as fh: + fh.write(rel_text) + return path + + +class TestRelativeFilesFrom: + """-R/--relative with --files-from keeps each listed entry's bare relative + destination path below the destination root instead of mirroring the full + source path. Without -R the layout is unchanged (full source mirror).""" + + @pytest.mark.parametrize("mt", [False, True]) + def test_relative_files_from_keeps_relative_layout(self, shared_server, mt): + source = _make_relative_source("rel_src") + dest = os.path.join(TEST_DATA_DIR, "rel_dst") + clean_dir(dest) + lst = _write_rel_list(b"top.txt\nsub/x.txt\n") + flags = ["--files-from", lst, "-R"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"-R files-from sync failed: {result.stderr[:200]}" + assert _read_file(os.path.join(dest, "sub", "x.txt")) == b"x\n", \ + "listed file must land at /sub/x.txt" + assert _read_file(os.path.join(dest, "top.txt")) == b"top\n", \ + "top-level listed file must land at /top.txt" + assert not os.path.exists(os.path.join(dest, "sub", "y.txt")) + # The source-root mirror must not be reproduced under -R. + assert not os.path.exists(get_dest_received_dir(dest, source)), \ + "-R must not mirror the full source path" + + @pytest.mark.parametrize("mt", [False, True]) + def test_relative_without_files_from_has_no_effect(self, shared_server, mt): + """-R alone (no --files-from) must leave the normal full-source mirror + layout untouched.""" + source = _make_relative_source("rel_only_src") + dest = os.path.join(TEST_DATA_DIR, "rel_only_dst") + clean_dir(dest) + flags = ["-R"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"-R alone sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + mismatches, missing = verify_transfer(source, received) + assert not missing and not mismatches + + @pytest.mark.parametrize("mt", [False, True]) + def test_without_relative_layout_unchanged(self, shared_server, mt): + source = _make_relative_source("rel_noR_src") + dest = os.path.join(TEST_DATA_DIR, "rel_noR_dst") + clean_dir(dest) + lst = _write_rel_list(b"sub/x.txt\n") + flags = ["--files-from", lst] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"files-from sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, "sub", "x.txt")) == b"x\n", \ + "without -R the full source mirror layout is preserved" + assert not os.path.exists(os.path.join(dest, "sub")), \ + "bare relative layout must not appear without -R" + + def test_relative_delete_manifest_stays_consistent(self): + """--delete derives from the sent (-R) relative paths, so a later + subset run removes unlisted relative entries but keeps listed ones.""" + source = _make_relative_source("rel_del_src") + dest = os.path.join(TEST_DATA_DIR, "rel_del_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + lst = _write_rel_list(b"sub/x.txt\nsub/y.txt\n") + result, _ = run_client(source, dest, flags=["--files-from", lst, "-R"], + port=server.port) + assert result.returncode == 0, f"seed -R sync failed: {result.stderr[:200]}" + assert os.path.isfile(os.path.join(dest, "sub", "y.txt")) + + subset = _write_rel_list(b"sub/x.txt\n") + result, _ = run_client(source, dest, + flags=["--files-from", subset, "-R", "--delete"], + port=server.port) + assert result.returncode == 0, f"-R delete sync failed: {result.stderr[:200]}" + assert os.path.isfile(os.path.join(dest, "sub", "x.txt")), "listed file was deleted" + assert not os.path.exists(os.path.join(dest, "sub", "y.txt")), \ + "unlisted relative file was not deleted" + + +class TestMissingArgs: + """--ignore-missing-args / --delete-missing-args: a --files-from entry that + does not exist under the source is skipped instead of failing the run, and + (delete-missing) its destination mirror is removed receiver-side. Following + rsync, --delete-missing-args implies --ignore-missing-args but is + independent of --delete: unrelated extras stay unless --delete is also + given, and the missing-args deletion (an explicit user request) is never + blocked by filter-exclusion protection.""" + + def _make_source(self, name): + source = os.path.join(TEST_DATA_DIR, name) + clean_dir(source) + for rel, content in { + "a.txt": b"a\n", + "sub/b.txt": b"b\n", + "keep.txt": b"keep\n", + "prot/kept.txt": b"kept\n", + }.items(): + full = os.path.join(source, rel) + os.makedirs(os.path.dirname(full), exist_ok=True) + with open(full, "wb") as fh: + fh.write(content) + return source + + @pytest.mark.parametrize("mt", [False, True]) + def test_missing_entry_is_hard_error_before_transfer(self, shared_server, mt): + source = self._make_source("mg_default_src") + dest = os.path.join(TEST_DATA_DIR, "mg_default_dst") + clean_dir(dest) + lst = _write_rel_list(b"a.txt\ngone.txt\nsub/b.txt\n") + flags = ["--files-from", lst] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode != 0, "a listed-but-missing entry did not fail the run" + assert "gone.txt" in (result.stderr or result.stdout) + received = get_dest_received_dir(dest, source) + assert not os.path.isfile(os.path.join(received, "a.txt")), \ + "the transfer started despite the missing-entry hard error" + + @pytest.mark.parametrize("mt", [False, True]) + def test_ignore_missing_args_transfers_the_rest(self, shared_server, mt): + source = self._make_source("mg_ignore_src") + dest = os.path.join(TEST_DATA_DIR, "mg_ignore_dst") + clean_dir(dest) + lst = _write_rel_list(b"a.txt\ngone.txt\nsub/b.txt\n") + flags = ["--files-from", lst, "--ignore-missing-args"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"ignore-missing-args sync failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, "a.txt")) == b"a\n" + assert _read_file(os.path.join(received, "sub", "b.txt")) == b"b\n" + assert not os.path.exists(os.path.join(received, "gone.txt")), \ + "nothing was transferred for the missing entry" + assert "--ignore-missing-args" in (result.stderr or result.stdout), \ + "the skipped entry must be observable (not a silent no-op)" + + @pytest.mark.parametrize("mt", [False, True]) + def test_all_missing_entries_succeed_transferring_nothing(self, shared_server, mt): + source = self._make_source("mg_all_missing_src") + dest = os.path.join(TEST_DATA_DIR, "mg_all_missing_dst") + clean_dir(dest) + lst = _write_rel_list(b"gone1.txt\ngone2.txt\n") + flags = ["--files-from", lst, "--ignore-missing-args"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, \ + f"all-missing run should succeed (rsync parity): {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert not os.path.exists(os.path.join(received, "gone1.txt")) + + @pytest.mark.parametrize("mt", [False, True]) + def test_empty_list_stays_a_hard_error(self, shared_server, mt): + source = self._make_source("mg_empty_src") + dest = os.path.join(TEST_DATA_DIR, "mg_empty_dst") + clean_dir(dest) + lst = _write_rel_list(b"") + flags = ["--files-from", lst, "--ignore-missing-args"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode != 0, "an empty --files-from list must stay a hard error" + assert "contains no entries" in (result.stderr or result.stdout) + + @pytest.mark.parametrize("mt", [False, True]) + def test_delete_missing_removes_mirror_not_unrelated(self, mt): + """-R layout: --delete-missing-args deletes exactly the missing entry's + destination mirror (bare relative path) and leaves unrelated extras + untouched; with --delete also present the unrelated extras go too.""" + source = self._make_source("mg_del_src") + dest = os.path.join(TEST_DATA_DIR, "mg_del_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + seed = _write_rel_list(b"a.txt\nsub/b.txt\n") + result, _ = run_client(source, dest, + flags=["--files-from", seed, "-R"] + (["--threads"] if mt else []), + port=server.port) + assert result.returncode == 0, f"seed -R sync failed: {result.stderr[:200]}" + assert os.path.isfile(os.path.join(dest, "a.txt")) + assert os.path.isfile(os.path.join(dest, "sub", "b.txt")) + + # Plant the missing entry's destination mirror and an unrelated extra. + with open(os.path.join(dest, "gone.txt"), "w") as fh: + fh.write("stale mirror") + with open(os.path.join(dest, "unrelated.txt"), "w") as fh: + fh.write("unrelated") + + lst = _write_rel_list(b"a.txt\ngone.txt\nsub/b.txt\n") + flags = ["--files-from", lst, "-R", "--delete-missing-args"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, f"delete-missing sync failed: {result.stderr[:300]}" + assert not os.path.exists(os.path.join(dest, "gone.txt")), \ + "the missing entry's destination mirror was not deleted" + assert os.path.isfile(os.path.join(dest, "unrelated.txt")), \ + "--delete-missing-args removed an unrelated extra (only --delete may)" + assert os.path.isfile(os.path.join(dest, "a.txt")) + assert os.path.isfile(os.path.join(dest, "sub", "b.txt")) + + # Now with --delete the unrelated extra is an ordinary extra and must go. + lst2 = _write_rel_list(b"a.txt\ngone.txt\nsub/b.txt\n") + flags2 = ["--files-from", lst2, "-R", "--delete-missing-args", "--delete"] + \ + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags2, port=server.port) + assert result.returncode == 0, f"delete-missing + delete sync failed: {result.stderr[:300]}" + assert not os.path.exists(os.path.join(dest, "unrelated.txt")), \ + "--delete did not remove the unrelated extra" + assert not os.path.exists(os.path.join(dest, "gone.txt")) + assert os.path.isfile(os.path.join(dest, "a.txt")) + + @pytest.mark.parametrize("mt", [False, True]) + def test_delete_missing_mirror_outside_relative_layout(self, mt): + """Without -R the missing entry's mirror mirrors the full source path + below the destination root, exactly like a present sibling's.""" + source = self._make_source("mg_del_nor_src") + dest = os.path.join(TEST_DATA_DIR, "mg_del_nor_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + # Full-tree seed places every current source file in the mirrored layout. + result, _ = run_client(source, dest, flags=["--delete"], port=server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + assert os.path.isfile(os.path.join(received, "a.txt")) + + # Plant a stale mirror for an entry not (yet) on the source. + with open(os.path.join(received, "gone.txt"), "w") as fh: + fh.write("stale") + lst = _write_rel_list(b"a.txt\ngone.txt\n") + result, _ = run_client(source, dest, + flags=["--files-from", lst, "--delete-missing-args"], + port=server.port) + assert result.returncode == 0, f"delete-missing no-R sync failed: {result.stderr[:300]}" + assert not os.path.exists(os.path.join(received, "gone.txt")), \ + "the full-source-mirror path of the missing entry was not deleted" + assert os.path.isfile(os.path.join(received, "a.txt")) + + @pytest.mark.parametrize("mt", [False, True]) + def test_delete_missing_args_not_blocked_by_exclude_protection(self, mt): + """A missing-arg mirror that sits under a filter-excluded directory is an + explicit user request, so --delete-missing-args removes it even though an + ordinary --delete honours the exclusion protection (rsync parity). Uses + the non-relative layout: exclusion protection is only recorded there.""" + source = self._make_source("mg_excl_src") + dest = os.path.join(TEST_DATA_DIR, "mg_excl_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + # Full-tree seed mirrors the whole source below the destination root. + result, _ = run_client(source, dest, port=server.port) + assert result.returncode == 0, f"seed failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + assert os.path.isfile(os.path.join(received, "prot", "kept.txt")) + + # A stale mirror under the (now excluded) prot/ directory, plus an extra. + with open(os.path.join(received, "prot", "gone.txt"), "w") as fh: + fh.write("stale") + with open(os.path.join(received, "extra.txt"), "w") as fh: + fh.write("extra") + + lst = _write_rel_list(b"a.txt\nprot/gone.txt\n") + flags = ["--files-from", lst, "--filter=- prot/", "--delete-missing-args", + "--delete"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, f"delete-missing exclude sync failed: {result.stderr[:300]}" + assert not os.path.exists(os.path.join(received, "prot", "gone.txt")), \ + "the explicit missing-arg deletion was blocked by exclusion protection" + assert os.path.isfile(os.path.join(received, "prot", "kept.txt")), \ + "the excluded-but-present destination file must stay (default protection)" + assert not os.path.exists(os.path.join(received, "extra.txt")), \ + "--delete did not remove the unrelated extra" + assert os.path.isfile(os.path.join(received, "a.txt")) + + @pytest.mark.parametrize("mt", [False, True]) + def test_delete_missing_args_with_delete_before(self, mt): + """--delete-before (early delete timing) composes with --delete-missing-args: + the exact-path deletions commit with the early manifest, before data, and + --delete-before implies --delete (so unrelated extras go too).""" + source = self._make_source("mg_early_src") + dest = os.path.join(TEST_DATA_DIR, "mg_early_dst") + clean_dir(dest) + with open(os.path.join(dest, "gone.txt"), "w") as fh: + fh.write("stale") + with open(os.path.join(dest, "extra.txt"), "w") as fh: + fh.write("extra") + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + lst = _write_rel_list(b"a.txt\ngone.txt\ngone2.txt\n") + flags = ["--files-from", lst, "-R", "--delete-missing-args", "--delete-before"] + \ + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, f"early delete-missing sync failed: {result.stderr[:300]}" + assert not os.path.exists(os.path.join(dest, "gone.txt")), \ + "early timing did not remove the missing-arg mirror" + assert os.path.isfile(os.path.join(dest, "a.txt")), "a.txt was not transferred" + assert not os.path.exists(os.path.join(dest, "extra.txt")), \ + "--delete-before implies --delete: unrelated extras must go" + + @pytest.mark.parametrize("mt", [False, True]) + @pytest.mark.parametrize("relative", [False, True]) + def test_delete_missing_deep_entry_with_absent_parent(self, mt, relative): + """A missing entry whose destination mirror's parent directory does not + exist is a no-op (nothing to delete), never a run failure: the + exact-path deletions must not abort the --delete extras walk. Covers + the -R bare-relative layout and the full source-mirror layout.""" + source = self._make_source("mg_deep_src") + dest = os.path.join(TEST_DATA_DIR, "mg_deep_dst") + clean_dir(dest) + rel_flags = ["-R"] if relative else [] + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + if relative: + target_root = dest + else: + # Non-relative layout: seed a.txt so the receive-root mirror + # tree exists (its sub/ sibling deliberately does not). + seed = _write_rel_list(b"a.txt\n") + result, _ = run_client(source, dest, + flags=["--files-from", seed] + rel_flags, + port=server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + target_root = get_dest_received_dir(dest, source) + assert os.path.isfile(os.path.join(target_root, "a.txt")) + with open(os.path.join(target_root, "extra.txt"), "w") as fh: + fh.write("extra") + + lst = _write_rel_list(b"a.txt\nsub/gone.txt\n") + flags = ["--files-from", lst, "--delete-missing-args", "--delete"] + rel_flags + \ + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, \ + f"deep missing-entry sync failed: {result.stderr[:300]}" + assert _read_file(os.path.join(target_root, "a.txt")) == b"a\n" + assert not os.path.exists(os.path.join(target_root, "extra.txt")), \ + "--delete extras walk was aborted by the absent-parent missing entry" + assert not os.path.exists(os.path.join(target_root, "sub")), \ + "the absent parent directory of the missing entry was created" + + @pytest.mark.parametrize("mt", [False, True]) + def test_dirs_missing_entry_skipped_in_scanner(self, shared_server, mt): + """--dirs + --files-from: a listed-but-missing entry is skipped in the + --dirs generator (which would otherwise hard-fail), transferring the + rest of the list.""" + source = self._make_source("mg_dirs_src") + dest = os.path.join(TEST_DATA_DIR, "mg_dirs_dst") + clean_dir(dest) + lst = _write_rel_list(b"a.txt\ngone.txt\n") + flags = ["--files-from", lst, "--dirs", "-R", "--ignore-missing-args"] + \ + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"--dirs ignore-missing sync failed: {result.stderr[:300]}" + assert _read_file(os.path.join(dest, "a.txt")) == b"a\n", \ + "the listed present file was not transferred" + assert not os.path.exists(os.path.join(dest, "gone.txt")), \ + "a directory/file was created for the missing --dirs entry" + + +class TestNoImpliedDirs: + """--no-implied-dirs (only meaningful with -R + --files-from) refuses to + place a listed file whose parent directory is not itself listed.""" + + def _make(self): + return _make_relative_source("noimplied_src") + + @pytest.mark.parametrize("mt", [False, True]) + def test_implied_dir_only_fails_entry(self, shared_server, mt): + source = self._make() + dest = os.path.join(TEST_DATA_DIR, "noimplied_dst") + clean_dir(dest) + lst = _write_rel_list(b"a/b.txt\n") # "a" itself is not listed + flags = ["--files-from", lst, "-R", "--no-implied-dirs"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode != 0, "implied parent directory was not rejected" + assert "--no-implied-dirs" in (result.stderr or result.stdout) + assert not os.path.exists(os.path.join(dest, "a", "b.txt")) + + @pytest.mark.parametrize("mt", [False, True]) + def test_listed_dir_allows_file(self, shared_server, mt): + source = self._make() + dest = os.path.join(TEST_DATA_DIR, "noimplied_ok_dst") + clean_dir(dest) + lst = _write_rel_list(b"a\na/b.txt\n") + flags = ["--files-from", lst, "-R", "--no-implied-dirs"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"listed dir + file sync failed: {result.stderr[:200]}" + assert _read_file(os.path.join(dest, "a", "b.txt")) == b"nested\n" + + @pytest.mark.parametrize("mt", [False, True]) + def test_no_implied_dirs_without_relative_changes_nothing(self, shared_server, mt): + source = self._make() + dest = os.path.join(TEST_DATA_DIR, "noimplied_noR_dst") + clean_dir(dest) + lst = _write_rel_list(b"a/b.txt\n") + flags = ["--files-from", lst, "--no-implied-dirs"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, "--no-implied-dirs without -R changed behavior" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, "a", "b.txt")) == b"nested\n" + + +class TestDirs: + """-d/--dirs (and the --old-dirs/--old-d aliases) transfer directory entries + without recursing into their contents.""" + + def _make(self): + return _make_relative_source("dirs_src") + + def _assert_only_empty_mirror(self, dest, source): + mirror = get_dest_received_dir(dest, source) + assert os.path.isdir(mirror), "source-root mirror directory was not created" + files = [] + for root, _dirs, names in os.walk(mirror): + files.extend(os.path.relpath(os.path.join(root, n), mirror) for n in names) + assert files == [], f"--dirs descended into contents: {files}" + + @pytest.mark.parametrize("flag", ["--dirs", "-d", "--old-dirs", "--old-d"]) + @pytest.mark.parametrize("mt", [False, True]) + def test_dirs_transfers_empty_dir_only(self, shared_server, flag, mt): + source = self._make() + dest = os.path.join(TEST_DATA_DIR, "dirs_dst") + clean_dir(dest) + flags = [flag] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"{flag} sync failed: {result.stderr[:200]}" + self._assert_only_empty_mirror(dest, source) + + @pytest.mark.parametrize("mt", [False, True]) + def test_dirs_with_files_from(self, shared_server, mt): + source = self._make() + dest = os.path.join(TEST_DATA_DIR, "dirs_ff_dst") + clean_dir(dest) + # A listed directory is created empty; a listed file is transferred. + lst = _write_rel_list(b"dir1\nsub/x.txt\n") + flags = ["--files-from", lst, "--dirs", "-R"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"dirs files-from sync failed: {result.stderr[:200]}" + assert os.path.isdir(os.path.join(dest, "dir1")), "listed dir was not created" + assert not os.path.exists(os.path.join(dest, "dir1", "keep.txt")), \ + "--dirs must not descend into a listed directory" + assert _read_file(os.path.join(dest, "sub", "x.txt")) == b"x\n", \ + "listed file content was not transferred" + assert not os.path.exists(os.path.join(dest, "sub", "y.txt")), \ + "unlisted file appeared" + + @pytest.mark.parametrize("mt", [False, True]) + def test_dirs_with_files_from_mirror_layout(self, shared_server, mt): + """Without -R the dirs+files-from entries still mirror the source path.""" + source = self._make() + dest = os.path.join(TEST_DATA_DIR, "dirs_ff_noR_dst") + clean_dir(dest) + lst = _write_rel_list(b"dir1\n") + flags = ["--files-from", lst, "--dirs"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"dirs files-from no-R sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + assert os.path.isdir(os.path.join(received, "dir1")), "mirrored dir entry not created" + assert not os.path.exists(os.path.join(received, "dir1", "keep.txt")), \ + "--dirs must not descend into a listed directory" + assert not os.path.exists(os.path.join(received, "sub")), \ + "unlisted subtree appeared" + + @pytest.mark.parametrize("mt", [False, True]) + def test_dirs_chunk_serialization(self, shared_server, mt): + """--dirs entries survive the chunk-serialization wire path (type + marker round-trips); a listed dir lands empty and a listed file lands + with content, with no protocol desync under -s -m.""" + source = self._make() + dest = os.path.join(TEST_DATA_DIR, "dirs_s_dst") + clean_dir(dest) + lst = _write_rel_list(b"dir1\nsub/x.txt\n") + flags = ["--files-from", lst, "--dirs", "-R", "--chunk-serialization"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"dirs -s sync failed: {result.stderr[:200]}" + assert os.path.isdir(os.path.join(dest, "dir1")), "listed dir was not created" + assert not os.path.exists(os.path.join(dest, "dir1", "keep.txt")), \ + "--dirs must not descend into a listed directory" + assert _read_file(os.path.join(dest, "sub", "x.txt")) == b"x\n", \ + "listed file content was not transferred" + + def test_dirs_delete_keeps_transferred_empty_dir(self): + """Directory entries appear in the delete manifest, so the empty dir a + --dirs run just created is not pruned as an extra by --delete.""" + source = self._make() + dest = os.path.join(TEST_DATA_DIR, "dirs_del_dst") + clean_dir(dest) + extra = os.path.join(dest, "extra.txt") + with open(extra, "wb") as fh: + fh.write(b"delete me") + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, flags=["--dirs", "--delete"], + port=server.port) + assert result.returncode == 0, f"--dirs --delete sync failed: {result.stderr[:200]}" + assert not os.path.exists(extra), "--delete did not remove the extra file" + mirror = get_dest_received_dir(dest, source) + assert os.path.isdir(mirror), "transferred empty dir was pruned as an extra" + files = [] + for root, _dirs, names in os.walk(mirror): + files.extend(os.path.relpath(os.path.join(root, n), mirror) for n in names) + assert files == [], f"--dirs descended into contents: {files}" + + def test_dirs_listed_dir_colliding_with_file_fails(self, shared_server): + """A listed directory that already exists as a regular file at the + destination fails the transfer cleanly instead of clobbering the file.""" + source = self._make() + dest = os.path.join(TEST_DATA_DIR, "dirs_coll_dst") + clean_dir(dest) + blocker = os.path.join(dest, "dir1") + with open(blocker, "wb") as fh: + fh.write(b"blocking file") + lst = _write_rel_list(b"dir1\n") + result, _ = run_client(source, dest, flags=["--files-from", lst, "--dirs", "-R"], + port=shared_server.port) + assert result.returncode != 0, "dir entry over an existing file did not fail" + assert os.path.isfile(blocker), "blocking regular file was clobbered" + + +class TestMkpath: + """--mkpath tells the server to create the destination root directory (and + missing leading components) when it does not exist yet; without it a missing + destination root fails the transfer.""" + + @pytest.mark.parametrize("mt", [False, True]) + def test_missing_root_fails_without_mkpath(self, mt): + source = _make_relative_source("mkpath_fail_src") + dest = os.path.join(TEST_DATA_DIR, "mkpath_missing_dst") + shutil.rmtree(dest, ignore_errors=True) + with ServerManager() as server: + server.start() + flags = ["--threads"] if mt else [] + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode != 0, "missing destination root did not fail without --mkpath" + assert not os.path.exists(dest), "missing root was created without --mkpath" + + @pytest.mark.parametrize("mt", [False, True]) + def test_mkpath_creates_missing_root(self, mt): + source = _make_relative_source("mkpath_ok_src") + dest = os.path.join(TEST_DATA_DIR, "deep", "mkpath_dst") + shutil.rmtree(os.path.join(TEST_DATA_DIR, "deep"), ignore_errors=True) + with ServerManager() as server: + server.start() + flags = ["--mkpath"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, f"--mkpath sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, "sub", "x.txt")) == b"x\n", \ + "file not transferred into the --mkpath-created root" + + @pytest.mark.parametrize("mkpath", [False, True]) + def test_existing_dest_with_trailing_slash(self, shared_server, mkpath): + """A destination root written with a trailing slash must keep working: + an existing root is accepted both with and without --mkpath.""" + source = _make_relative_source("mkpath_trail_src") + dest = os.path.join(TEST_DATA_DIR, "mkpath_trail_dst") + clean_dir(dest) + dest_slash = dest + "/" + flags = ["--mkpath"] if mkpath else [] + result, _ = run_client(source, dest_slash, flags=flags, port=shared_server.port) + assert result.returncode == 0, \ + f"trailing-slash dest sync (mkpath={mkpath}) failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, "sub", "x.txt")) == b"x\n", \ + "file not transferred into the trailing-slash destination root" + + @pytest.mark.parametrize("mkpath", [False, True]) + def test_dest_equal_authorized_root(self, mkpath): + """A destination that is exactly the server's authorized root works + without --mkpath, and with --mkpath creates no stray / + nested directory.""" + root = os.path.join(TEST_DATA_DIR, "mkpath_eq_root") + clean_dir(root) + source = _make_relative_source("mkpath_eq_src") + with ServerManager() as server: + server.start(extra_args=["--destination-root", root]) + flags = ["--mkpath"] if mkpath else [] + result, _ = run_client(source, root, flags=flags, port=server.port) + assert result.returncode == 0, \ + f"dest==authorized-root sync (mkpath={mkpath}) failed: {result.stderr[:200]}" + received = get_dest_received_dir(root, source) + assert _read_file(os.path.join(received, "sub", "x.txt")) == b"x\n", \ + "file not transferred when the dest equals the authorized root" + basename = os.path.basename(root.rstrip(os.sep)) + assert not os.path.exists(os.path.join(root, basename)), \ + "--mkpath created a spurious nested / directory" + + +class TestFilters: + """--filter/-C/-F rule layer: excludes prune, ordering is first-match-wins, + the default with no matching rule is include, and legacy --exclude remains + an independent layer.""" + + +class TestDeleteTiming: + """rsync deletion-timing family. --delete-before/--delete-during transmit + the keep-set manifest BEFORE any file data (the receiver deletes extras and + acks first); --delete/--delete-after/--delete-delay commit deletions only + after the whole transfer succeeded. Every timing flag implies --delete.""" + + def _seed(self, tag): + source = os.path.join(TEST_DATA_DIR, f"deltiming_{tag}_src") + clean_dir(source) + entries = { + "top.txt": b"top level\n", + "sub/deep.txt": b"deeply nested file\n", + } + for rel, content in entries.items(): + full = os.path.join(source, rel) + os.makedirs(os.path.dirname(full), exist_ok=True) + with open(full, "wb") as fh: + fh.write(content) + return source + + @pytest.mark.parametrize("flag", ["--delete-before", "--delete-during", "--del", + "--delete-after", "--delete-delay"]) + @pytest.mark.parametrize("mt", [False, True]) + @pytest.mark.ci + def test_flag_removes_extras_on_success(self, flag, mt): + """Every timing flag is accepted, implies --delete, and on a successful + transfer removes the destination extras exactly like plain --delete.""" + source = self._seed("ok") + dest = os.path.join(TEST_DATA_DIR, "deltiming_ok_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, port=server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + extra = os.path.join(received, "extra.txt") + with open(extra, "wb") as fh: + fh.write(b"should be deleted") + + flags = [flag] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, \ + f"{flag} sync failed: {(result.stderr or result.stdout)[:300]}" + assert not os.path.exists(extra), f"{flag} did not remove the extra file" + mismatches, missing = verify_transfer(source, received) + assert not missing, f"{flag} missing files: {missing}" + assert not mismatches, f"{flag} mismatched files: {mismatches}" + + @pytest.mark.parametrize("flag", ["--delete-before", "--delete-during", "--del"]) + @pytest.mark.parametrize("mt", [False, True]) + def test_early_flags_delete_before_data(self, flag, mt): + """--delete-before/--delete-during remove extras (and a file blocking a + destination directory) BEFORE data is applied, so a nested write that + would fail while the blocker still exists succeeds.""" + source = self._seed("early") + dest = os.path.join(TEST_DATA_DIR, "deltiming_early_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, port=server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + extra = os.path.join(received, "extra.txt") + with open(extra, "wb") as fh: + fh.write(b"extra file") + blocker = os.path.join(received, "sub") + shutil.rmtree(blocker) + with open(blocker, "wb") as fh: + fh.write(b"blocks the nested destination directory") + + flags = [flag] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, \ + f"{flag} (early delete) did not remove the blocker in time: " \ + f"{(result.stderr or result.stdout)[:300]}" + assert not os.path.exists(extra), f"{flag} did not delete the extra before data" + assert _read_file(os.path.join(received, "sub", "deep.txt")) == b"deeply nested file\n", \ + f"{flag}: nested file was not written after the early deletion" + + @pytest.mark.parametrize("flag", ["--delete", "--delete-after", "--delete-delay"]) + @pytest.mark.parametrize("mt", [False, True]) + def test_late_flags_commit_only_after_success(self, flag, mt): + """Plain --delete/--delete-after/--delete-delay defer deletion until the + whole transfer succeeds: a mid-transfer write failure must leave every + extra in place (commit-style safety). The -m receiver must also keep + the extras: the deferred keep-set is committed by the server only after + the disk-writer thread has finished, and a failing writer means the + manifest is freed, never applied.""" + source = self._seed("late") + dest = os.path.join(TEST_DATA_DIR, "deltiming_late_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, port=server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + extra = os.path.join(received, "extra.txt") + with open(extra, "wb") as fh: + fh.write(b"extra file") + blocker = os.path.join(received, "sub") + shutil.rmtree(blocker) + with open(blocker, "wb") as fh: + fh.write(b"blocks the nested destination directory") + + flags = [flag] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode != 0, \ + f"{flag} (mt={mt}) unexpectedly succeeded (deletion must be deferred)" + assert os.path.exists(extra), \ + f"{flag} (mt={mt}) removed an extra although the transfer failed" + assert os.path.isfile(blocker), \ + f"{flag} (mt={mt}) deleted the blocker although the transfer failed" + + def test_early_flag_respected_when_server_refuses_delete(self, shared_server): + """With an --allow-delete-less server the client's early timing still + completes (no deadlock on the pre-delete ack) and simply never deletes, + exactly like the plain server policy.""" + source = self._seed("refused") + dest = os.path.join(TEST_DATA_DIR, "deltiming_refused_dst") + clean_dir(dest) + result, _ = run_client(source, dest, port=shared_server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + extra = os.path.join(received, "extra.txt") + with open(extra, "wb") as fh: + fh.write(b"extra file") + result, _ = run_client(source, dest, flags=["--delete-before"], port=shared_server.port) + assert result.returncode == 0, \ + f"--delete-before against a refuse-delete server failed: {result.stderr[:300]}" + assert os.path.exists(extra), "unauthorized delete removed an extra file" + + +def _seed_delete_tree(tag, entries, dest): + """Create a source tree and seed a full mirror at `dest`, returning + (source, received_mirror).""" + source = os.path.join(TEST_DATA_DIR, f"delpol_{tag}_src") + clean_dir(source) + for rel, content in entries.items(): + full = os.path.join(source, rel) + os.makedirs(os.path.dirname(full), exist_ok=True) + with open(full, "wb") as fh: + fh.write(content) + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, port=server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + return source, received + + +class TestDeletePolicy: + """Deletion-policy family: --delete-excluded, --max-delete, --force, + --ignore-errors and --prune-empty-dirs.""" + + def _write(self, path, content): + os.makedirs(os.path.dirname(path), exist_ok=True) + with open(path, "wb") as fh: + fh.write(content) + + @pytest.mark.parametrize("mt", [False, True]) + @pytest.mark.parametrize("timing", + ["--delete", "--delete-before", "--delete-after", "--delete-delay"]) + def test_delete_protects_excluded_by_default_and_delete_excluded_removes(self, mt, timing): + """rsync parity: with a --delete timing the destination mirror path whose + source was excluded survives (protected by default); --delete-excluded + opts back into deleting it. Verified single-threaded and -m across every + timing (commit and early).""" + source = os.path.join(TEST_DATA_DIR, f"delexcl_{timing.strip('-')}_{mt}_src") + clean_dir(source) + entries = { + "keep.txt": b"kept\n", + "secret.log": b"secret\n", + "sub/nested.log": b"nested secret\n", + } + for rel, content in entries.items(): + self._write(os.path.join(source, rel), content) + dest = os.path.join(TEST_DATA_DIR, f"delexcl_{timing.strip('-')}_{mt}_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, port=server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + self._write(os.path.join(received, "extra.txt"), b"extra\n") + + # Default: the excluded mirrors survive --delete, genuine extras die. + flags = ["--exclude", "*.log", timing] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, \ + f"default delete sync failed: {(result.stderr or result.stdout)[:300]}" + assert os.path.exists(os.path.join(received, "secret.log")), \ + "excluded dest file was deleted under plain --delete (rsync protects it)" + assert os.path.exists(os.path.join(received, "sub", "nested.log")), \ + "nested excluded dest file was deleted under plain --delete" + assert not os.path.exists(os.path.join(received, "extra.txt")), \ + "genuine extra was not deleted" + + # --delete-excluded: excluded mirrors are extras again and die. + self._write(os.path.join(received, "extra.txt"), b"extra\n") + flags = ["--exclude", "*.log", timing, "--delete-excluded"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, \ + f"--delete-excluded sync failed: {(result.stderr or result.stdout)[:300]}" + assert not os.path.exists(os.path.join(received, "secret.log")), \ + "--delete-excluded did not remove the excluded dest file" + assert not os.path.exists(os.path.join(received, "sub", "nested.log")), \ + "--delete-excluded did not remove the nested excluded dest file" + assert not os.path.exists(os.path.join(received, "extra.txt")), \ + "genuine extra survived --delete-excluded" + assert _read_file(os.path.join(received, "keep.txt")) == b"kept\n" + + @pytest.mark.parametrize("mt", [False, True]) + def test_delete_excluded_excluded_directory_subtree(self, mt): + """A whole source directory excluded by a filter rule protects its whole + destination mirror by default; --delete-excluded removes the subtree.""" + source = os.path.join(TEST_DATA_DIR, f"delexcldir_{mt}_src") + clean_dir(source) + self._write(os.path.join(source, "keep.txt"), b"kept\n") + self._write(os.path.join(source, "skipdir", "a.log"), b"a\n") + self._write(os.path.join(source, "skipdir", "deep", "b.log"), b"b\n") + dest = os.path.join(TEST_DATA_DIR, f"delexcldir_{mt}_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, port=server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + + flags = ["--filter=- skipdir/", "--delete"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, \ + f"default delete sync failed: {(result.stderr or result.stdout)[:300]}" + assert os.path.exists(os.path.join(received, "skipdir", "a.log")), \ + "excluded dir subtree was deleted under plain --delete" + assert os.path.exists(os.path.join(received, "skipdir", "deep", "b.log")), \ + "nested excluded dir content was deleted under plain --delete" + + flags = ["--filter=- skipdir/", "--delete", "--delete-excluded"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, \ + f"--delete-excluded sync failed: {(result.stderr or result.stdout)[:300]}" + assert not os.path.exists(os.path.join(received, "skipdir")), \ + "--delete-excluded did not remove the excluded dir subtree" + + @pytest.mark.parametrize("mt", [False, True]) + @pytest.mark.parametrize("timing", ["--delete", "--delete-before"]) + def test_max_delete_exceeded_fails_without_deleting(self, mt, timing): + """A run that would exceed --max-delete deletes nothing and fails.""" + source = os.path.join(TEST_DATA_DIR, f"maxdel_{timing.strip('-')}_{mt}_src") + clean_dir(source) + self._write(os.path.join(source, "keep.txt"), b"kept\n") + dest = os.path.join(TEST_DATA_DIR, f"maxdel_{timing.strip('-')}_{mt}_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, port=server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + extras = [] + for i in range(4): + name = f"e{i}.txt" + self._write(os.path.join(received, name), b"extra\n") + extras.append(os.path.join(received, name)) + + flags = ["--max-delete=2", timing] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode != 0, \ + f"--max-delete=2 with 4 extras unexpectedly succeeded: {result.stderr[:300]}" + for path in extras: + assert os.path.exists(path), \ + "--max-delete overrun deleted files (must be all-or-nothing)" + + @pytest.mark.parametrize("mt", [False, True]) + def test_max_delete_not_exceeded_deletes_exactly(self, mt): + """When the extras are at or below --max-delete the run succeeds and + removes exactly the extras.""" + source = os.path.join(TEST_DATA_DIR, f"maxdelok_{mt}_src") + clean_dir(source) + self._write(os.path.join(source, "keep.txt"), b"kept\n") + dest = os.path.join(TEST_DATA_DIR, f"maxdelok_{mt}_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, port=server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + for i in range(3): + self._write(os.path.join(received, f"e{i}.txt"), b"extra\n") + flags = ["--max-delete=3", "--delete"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, \ + f"--max-delete=3 with 3 extras failed: {(result.stderr or result.stdout)[:300]}" + for i in range(3): + assert not os.path.exists(os.path.join(received, f"e{i}.txt")), \ + f"extra e{i}.txt not deleted under --max-delete=3" + + @pytest.mark.parametrize("mt", [False, True]) + def test_force_replaces_nonempty_dir_with_file(self, mt): + """--force lets an incoming regular file replace a non-empty destination + directory; without it the write (and the run) fails.""" + source = os.path.join(TEST_DATA_DIR, f"force_{mt}_src") + clean_dir(source) + self._write(os.path.join(source, "sub", "old.txt"), b"old\n") + self._write(os.path.join(source, "keep.txt"), b"kept\n") + dest = os.path.join(TEST_DATA_DIR, f"force_{mt}_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, port=server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + + # The source path `sub` becomes a regular file (the dir is gone). + os.unlink(os.path.join(source, "sub", "old.txt")) + os.rmdir(os.path.join(source, "sub")) + self._write(os.path.join(source, "sub"), b"now a file\n") + + result, _ = run_client(source, dest, port=server.port) + assert result.returncode != 0, \ + "a file over a non-empty directory must fail without --force" + assert os.path.isdir(os.path.join(received, "sub")), \ + "directory was destroyed although the run failed without --force" + assert os.path.exists(os.path.join(received, "sub", "old.txt")), \ + "non-empty dir content was lost although the run failed without --force" + + flags = ["--force"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, \ + f"--force run failed: {(result.stderr or result.stdout)[:300]}" + assert os.path.isfile(os.path.join(received, "sub")), \ + "--force did not replace the directory with the file" + assert _read_file(os.path.join(received, "sub")) == b"now a file\n" + + def test_force_inert_under_delay_updates(self): + """Documented divergence: --force acts on the immediate-install path; a + --delay-updates run stages into its own tree and its publication renames + over regular files only, so a blocking directory is not cleared and the + run fails.""" + source = os.path.join(TEST_DATA_DIR, "force_delay_src") + clean_dir(source) + self._write(os.path.join(source, "sub", "old.txt"), b"old\n") + self._write(os.path.join(source, "keep.txt"), b"kept\n") + dest = os.path.join(TEST_DATA_DIR, "force_delay_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, port=server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + os.unlink(os.path.join(source, "sub", "old.txt")) + os.rmdir(os.path.join(source, "sub")) + self._write(os.path.join(source, "sub"), b"now a file\n") + result, _ = run_client(source, dest, flags=["--force", "--delay-updates"], + port=server.port) + assert result.returncode != 0, \ + "--force --delay-updates unexpectedly replaced the blocking directory" + assert os.path.isdir(os.path.join(received, "sub")), \ + "blocking directory was cleared although --delay-updates should keep --force inert" + assert os.path.exists(os.path.join(received, "sub", "old.txt")), \ + "blocking directory content was lost" + + @pytest.mark.parametrize("mt", [False, True]) + def test_prune_empty_dirs_dirs_mode(self, mt): + """--prune-empty-dirs omits an empty source directory's explicit entry in + --dirs mode (nothing is created, and an existing empty mirror is removed + by --delete). Recursive transfers never emit empty dirs, so the flag is + a no-op there (documented rsync -m parity).""" + source = os.path.join(TEST_DATA_DIR, f"prune_{mt}_src") + clean_dir(source) + os.makedirs(source, exist_ok=True) # physically empty source dir + + dest = os.path.join(TEST_DATA_DIR, f"prune_{mt}_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, flags=["--dirs"], port=server.port) + assert result.returncode == 0, f"-d seed failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + assert os.path.isdir(received), "-d should create the empty mirror dir" + assert os.listdir(received) == [] + + # prune-empty-dirs: the empty mirror is pruned by --delete. + flags = ["--dirs", "--prune-empty-dirs", "--delete"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, \ + f"--dirs --prune-empty-dirs --delete failed: {(result.stderr or result.stdout)[:300]}" + assert not os.path.exists(received), \ + "--prune-empty-dirs did not prune the empty dir (--delete left it)" + + # A fresh destination: prune-empty-dirs means the empty dir is never sent. + dest2 = os.path.join(TEST_DATA_DIR, f"prune2_{mt}_dst") + clean_dir(dest2) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + flags = ["--dirs", "--prune-empty-dirs", "-i"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest2, flags=flags, port=server.port) + assert result.returncode == 0, \ + f"--dirs --prune-empty-dirs failed: {(result.stderr or result.stdout)[:300]}" + received2 = get_dest_received_dir(dest2, source) + assert not os.path.exists(received2), \ + "--prune-empty-dirs transferred the empty directory" + assert result.stdout == "", \ + f"--prune-empty-dirs leaked an itemize line: {result.stdout[:200]}" + + @pytest.mark.parametrize("mt", [False, True]) + def test_prune_empty_dirs_recursion_inherent(self, mt): + """In recursive mode FastSync never transfers empty directories (rsync + -m parity): a truly-empty destination directory chain is removed by + --delete whether or not --prune-empty-dirs is given (the flag has no + additional effect there), while directories holding kept files survive. + A filter-excluded file's mirror is protected, so a directory that still + holds one is left intact (rsync default delete-excluded semantics).""" + source = os.path.join(TEST_DATA_DIR, f"prunerec_{mt}_src") + clean_dir(source) + self._write(os.path.join(source, "keep.txt"), b"kept\n") + self._write(os.path.join(source, "a", "keep.log"), b"a log\n") + self._write(os.path.join(source, "b", "deep", "kept.txt"), b"deep kept\n") + dest = os.path.join(TEST_DATA_DIR, f"prunerec_{mt}_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, port=server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + # A stray empty chain (FastSync recursion never creates such dirs, so + # this models one left by an external tool / an earlier --dirs run). + os.makedirs(os.path.join(received, "empty", "chain")) + + for prune in ([], ["--prune-empty-dirs"]): + flags = prune + ["--delete"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, \ + f"prune recursive sync failed: {(result.stderr or result.stdout)[:300]}" + assert not os.path.exists(os.path.join(received, "empty")), \ + "truly-empty dir chain was not removed by --delete" + assert os.path.exists(os.path.join(received, "b", "deep", "kept.txt")), \ + "non-empty dir subtree was wrongly removed" + assert _read_file(os.path.join(received, "keep.txt")) == b"kept\n" + + # An excluded file's mirror is protected: the dir that holds it stays. + flags = ["--exclude", "*.log", "--delete", "--prune-empty-dirs"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, \ + f"prune recursive sync failed: {(result.stderr or result.stdout)[:300]}" + assert os.path.exists(os.path.join(received, "a", "keep.log")), \ + "excluded file mirror was deleted under --delete (rsync protects it)" + + def _run_client_as_nobody(self, source, dest, port, flags): + cmd = CLIENT_CMD + ["--source-dir", source, "--dest-dir", dest, + "--save-to-disk", "--server-port", str(port)] + flags + return subprocess.run(["setpriv", "--reuid=65534", "--regid=65534", + "--clear-groups"] + cmd, text=True, capture_output=True) + + @pytest.mark.parametrize("mt", [False, True]) + @pytest.mark.setpriv + def test_ignore_errors_keeps_deletion_active_on_scan_error(self, mt): + """A source I/O error (unreadable subdirectory) aborts the run so no + deletion happens by default; --ignore-errors continues, still transfers + the readable tree and still deletes, single-threaded and under -m. Run + as an unprivileged user so the mode-000 directory is genuinely + unreadable.""" + if os.geteuid() != 0 or shutil.which("setpriv") is None: + pytest.skip("requires root + setpriv to drop privileges for the client") + tag = f"ioerr_{os.getpid()}_{mt}" + source = os.path.join(TEST_DATA_DIR, f"{tag}_src") + clean_dir(source) + self._write(os.path.join(source, "top.txt"), b"top\n") + self._write(os.path.join(source, "ok", "inside.txt"), b"inside\n") + self._write(os.path.join(source, "locked", "blocked.txt"), b"blocked\n") + dest = os.path.join(TEST_DATA_DIR, f"{tag}_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + # Seed as root (server is root too). + result, _ = run_client(source, dest, port=server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + try: + os.chmod(os.path.join(source, "locked"), 0) + + # Default: scan error aborts the run; nothing is deleted. + self._write(os.path.join(received, "extra.txt"), b"extra\n") + flags = ["--delete"] + (["--threads"] if mt else []) + result = self._run_client_as_nobody(source, dest, server.port, flags) + assert result.returncode != 0, "unreadable source dir did not fail the run" + assert os.path.exists(os.path.join(received, "extra.txt")), \ + "default run deleted although the scan hit an I/O error" + + # --ignore-errors: the readable tree transfers, deletion still runs. + self._write(os.path.join(received, "extra.txt"), b"extra\n") + flags = ["--delete", "--ignore-errors"] + (["--threads"] if mt else []) + result = self._run_client_as_nobody(source, dest, server.port, flags) + assert not os.path.exists(os.path.join(received, "extra.txt")), \ + f"--ignore-errors did not keep deletion active: {result.stderr[:300]}" + assert not os.path.exists(os.path.join(received, "locked")), \ + "mirror of the unreadable dir was left behind (should be an extra)" + finally: + os.chmod(os.path.join(source, "locked"), 0o755) + + @pytest.mark.parametrize("mt", [False, True]) + @pytest.mark.parametrize("timing", ["--delete", "--delete-before"]) + @pytest.mark.setpriv + def test_ignore_errors_unreadable_root_never_deletes(self, mt, timing): + """An unreadable SOURCE ROOT must never be treated as a skippable scan + error: with --ignore-errors the sequential scanner treats the root as + fatal (matching the -m path, which cannot even create its scanner), so + no empty keep-set manifest is sent and the destination is never wiped. + Run as an unprivileged user so the mode-000 root is genuinely + unreadable.""" + if os.geteuid() != 0 or shutil.which("setpriv") is None: + pytest.skip("requires root + setpriv to drop privileges for the client") + tag = f"rootio_{os.getpid()}_{mt}_{timing.strip('-')}" + source = os.path.join(TEST_DATA_DIR, f"{tag}_src") + clean_dir(source) + self._write(os.path.join(source, "file.txt"), b"content\n") + dest = os.path.join(TEST_DATA_DIR, f"{tag}_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, port=server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + try: + os.chmod(source, 0) + self._write(os.path.join(received, "extra.txt"), b"extra\n") + flags = [timing, "--ignore-errors"] + (["--threads"] if mt else []) + result = self._run_client_as_nobody(source, dest, server.port, flags) + assert result.returncode != 0, \ + f"unreadable source root with {timing} (mt={mt}) unexpectedly succeeded" + assert os.path.exists(os.path.join(received, "file.txt")), \ + f"{timing} (mt={mt}) wiped a kept destination file" + assert os.path.exists(os.path.join(received, "extra.txt")), \ + f"{timing} (mt={mt}) deleted the extra although the scan could not read the root" + finally: + os.chmod(source, 0o755) + + def test_delete_excluded_protection_is_sender_derived(self): + """Plain --delete protects destination mirrors of files the SOURCE scan + excluded, but a destination-only file that merely matches an exclude + rule is still an extra and is removed (protection never re-applies rules + to the destination).""" + source = os.path.join(TEST_DATA_DIR, "senderderived_src") + clean_dir(source) + self._write(os.path.join(source, "keep.txt"), b"kept\n") + self._write(os.path.join(source, "secret.log"), b"secret\n") + dest = os.path.join(TEST_DATA_DIR, "senderderived_dst") + clean_dir(dest) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, port=server.port) + assert result.returncode == 0, f"seed sync failed: {result.stderr[:200]}" + received = get_dest_received_dir(dest, source) + # A destination-only file that happens to match the exclude rule. + self._write(os.path.join(received, "stray.log"), b"never on the source\n") + result, _ = run_client(source, dest, flags=["--exclude", "*.log", "--delete"], + port=server.port) + assert result.returncode == 0, \ + f"delete sync failed: {(result.stderr or result.stdout)[:300]}" + assert os.path.exists(os.path.join(received, "secret.log")), \ + "source-excluded mirror was deleted under plain --delete" + assert not os.path.exists(os.path.join(received, "stray.log")), \ + "destination-only file matching the exclude rule was left (should be deleted)" + + +def _pin_mtime(path, ts): + os.utime(path, (ts, ts)) + + +class TestBasisDestDirs: + """--compare-dest / --copy-dest / --link-dest alternate basis directories. + + FastSync's basis directories are relative to the destination root and are + confined below it. The "unchanged" decision is receiver-side and requires + the per-file --incremental handshake (implied by these flags), so the basis + snapshot must reproduce the exact destination-relative mirror path of the + incoming files. + """ + + STAGING = ".fastsync-stage" + TS = 1577836800 # 2020-01-01 00:00:00 UTC, used to pin matching mtimes + + # fixture files: source and basis share the mtime pin, so a basis "match" + # is decided purely by content (xxHash). unchanged.txt is byte-identical; + # changed.txt is byte-DIFFERENT but has the SAME SIZE as the source (and + # the same pinned mtime), which is what forces the content-hash gate; + # added.txt does not exist in the basis at all. + UNCHANGED = "unchanged.txt" + CHANGED = "changed.txt" + ADDED = "added.txt" + + def _make_source(self, name, source_files): + src = os.path.join(TEST_DATA_DIR, name) + clean_dir(src) + for rel, content in source_files.items(): + full = os.path.join(src, rel) + os.makedirs(os.path.dirname(full), exist_ok=True) + with open(full, "wb") as fh: + fh.write(content) + _pin_mtime(full, self.TS) + return src + + def _seed_basis_file(self, dest, source, basis_dir, rel, content, ts=None): + base = os.path.join(dest, basis_dir, os.path.relpath( + get_dest_received_dir(dest, source), dest)) + full = os.path.join(base, rel) + os.makedirs(os.path.dirname(full), exist_ok=True) + with open(full, "wb") as fh: + fh.write(content) + _pin_mtime(full, self.TS if ts is None else ts) + return full + + def _seed_basis(self, dest, source, basis_dir, basis_files): + for rel, content in basis_files.items(): + self._seed_basis_file(dest, source, basis_dir, rel, content) + return os.path.join(dest, basis_dir, os.path.relpath( + get_dest_received_dir(dest, source), dest)) + + def _source_tree(self, prefix): + return { + self.UNCHANGED: b"stable content v1\n", + self.CHANGED: b"changed content now\n", + self.ADDED: b"brand new content\n", + } + + def _basis_tree(self, prefix): + # unchanged.txt is identical to the source; changed.txt has the SAME + # byte size and pinned mtime but a different body (equal size forces + # the xxHash gate); added.txt is missing from the basis. + return { + self.UNCHANGED: b"stable content v1\n", + self.CHANGED: b"CHANGED CONTENT NOW\n", + } + + def test_same_size_different_content_is_not_a_basis_match(self, shared_server): + # Core safety property: equal size + pinned mtime but different content + # must NEVER be hard-linked or copied from the basis -- the xxHash gate + # rejects it and the sender's data is transferred instead. + for flag, basis_dir in (("--link-dest", "szlb"), ("--copy-dest", "szcp"), + ("--compare-dest", "szcmp")): + source = self._make_source("basis_same_size_src", + {self.UNCHANGED: b"same length body\n"}) + dest = os.path.join(TEST_DATA_DIR, f"basis_same_size_dst_{basis_dir}") + clean_dir(dest) + basis_file = self._seed_basis_file(dest, source, basis_dir, self.UNCHANGED, + b"SAME LENGTH BODY!") + result, _ = run_client(source, dest, flags=[f"{flag}={basis_dir}"], + port=shared_server.port) + assert result.returncode == 0, \ + f"{flag} same-size mismatch failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + dest_file = os.path.join(received, self.UNCHANGED) + assert _read_file(dest_file) == b"same length body\n", \ + f"{flag}: basis content leaked into the destination on a hash mismatch" + if flag != "--compare-dest": + assert os.stat(dest_file).st_ino != os.stat(basis_file).st_ino, \ + f"{flag}: linked/copied from a content-mismatched basis file" + + @pytest.mark.ci + def test_compare_dest_skips_matching_and_transfers_missing(self, shared_server): + source = self._make_source("basis_compare_src", self._source_tree("c")) + dest = os.path.join(TEST_DATA_DIR, "basis_compare_dst") + clean_dir(dest) + self._seed_basis(dest, source, "cbasis", self._basis_tree("c")) + result, _ = run_client(source, dest, + flags=["--compare-dest=cbasis"], + port=shared_server.port) + assert result.returncode == 0, f"compare-dest failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + # compare-dest never copies: an exact basis match is skipped, leaving a + # sparse destination (rsync parity). + assert not os.path.exists(os.path.join(received, self.UNCHANGED)), \ + "compare-dest materialized the unchanged file" + # Files the destination lacks AND the basis cannot satisfy are still + # transferred normally. + assert _read_file(os.path.join(received, self.CHANGED)) == \ + self._source_tree("c")[self.CHANGED], "changed file not transferred" + assert _read_file(os.path.join(received, self.ADDED)) == \ + self._source_tree("c")[self.ADDED], "added file not transferred" + + def test_compare_dest_content_mismatch_forces_transfer(self, shared_server): + # The basis holds a file with a DIFFERENT body: even though it shares + # the mtime pin, the xxHash check fails and the data must be sent. + source = self._make_source("basis_compare_mismatch_src", {self.UNCHANGED: b"real data\n"}) + dest = os.path.join(TEST_DATA_DIR, "basis_compare_mismatch_dst") + clean_dir(dest) + basis = self._seed_basis(dest, source, "cbasis", {self.UNCHANGED: b"stale data!!\n"}) + result, _ = run_client(source, dest, flags=["--compare-dest=cbasis"], + port=shared_server.port) + assert result.returncode == 0, f"compare-dest mismatch failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.UNCHANGED)) == b"real data\n", \ + "content mismatch did not fall back to a normal transfer" + assert os.stat(os.path.join(received, self.UNCHANGED)).st_ino != \ + os.stat(os.path.join(basis, self.UNCHANGED)).st_ino + + def test_copy_dest_copies_unchanged_and_transfers_changed(self, shared_server): + source = self._make_source("basis_copy_src", self._source_tree("cp")) + dest = os.path.join(TEST_DATA_DIR, "basis_copy_dst") + clean_dir(dest) + basis = self._seed_basis(dest, source, "cpbasis", self._basis_tree("cp")) + result, _ = run_client(source, dest, flags=["--copy-dest=cpbasis"], + port=shared_server.port) + assert result.returncode == 0, f"copy-dest failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + unchanged = os.path.join(received, self.UNCHANGED) + assert _read_file(unchanged) == b"stable content v1\n", "unchanged file not materialized" + # A real local copy, NOT a hard link to the basis file. + assert os.stat(unchanged).st_ino != os.stat(os.path.join(basis, self.UNCHANGED)).st_ino + # Equal-size/different-content basis file falls back to the sender data. + assert _read_file(os.path.join(received, self.CHANGED)) == \ + self._source_tree("cp")[self.CHANGED] + assert _read_file(os.path.join(received, self.ADDED)) == \ + self._source_tree("cp")[self.ADDED] + + def test_link_dest_hardlinks_and_falls_back(self, shared_server): + source = self._make_source("basis_link_src", self._source_tree("ln")) + dest = os.path.join(TEST_DATA_DIR, "basis_link_dst") + clean_dir(dest) + basis = self._seed_basis(dest, source, "lnbasis", self._basis_tree("ln")) + result, _ = run_client(source, dest, flags=["--link-dest=lnbasis"], + port=shared_server.port) + assert result.returncode == 0, f"link-dest failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + unchanged = os.path.join(received, self.UNCHANGED) + basis_file = os.path.join(basis, self.UNCHANGED) + # Real hard link: same inode as the DIR file, nlink >= 2, no data copy. + assert os.path.exists(unchanged) + assert os.stat(unchanged).st_ino == os.stat(basis_file).st_ino, \ + "link-dest did not produce a hard link" + assert os.stat(unchanged).st_nlink >= 2 + # Equal-size/different-content basis file must fall back to a plain + # transfer (not a link). + changed = os.path.join(received, self.CHANGED) + assert _read_file(changed) == self._source_tree("ln")[self.CHANGED] + assert os.stat(changed).st_ino != os.stat(os.path.join(basis, self.CHANGED)).st_ino + + @pytest.mark.parametrize("flag", ["--compare-dest", "--copy-dest", "--link-dest"]) + def test_basis_dir_missing_is_a_clean_noop(self, shared_server, flag): + # A basis directory that does not exist must simply transfer everything. + source = self._make_source("basis_missing_src", {self.UNCHANGED: b"content\n"}) + dest = os.path.join(TEST_DATA_DIR, "basis_missing_dst") + clean_dir(dest) + result, _ = run_client(source, dest, flags=[f"{flag}=nope"], + port=shared_server.port) + assert result.returncode == 0, f"{flag} with missing dir failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.UNCHANGED)) == b"content\n" + + def test_link_dest_multithreaded(self, shared_server): + source = self._make_source("basis_link_mt_src", self._source_tree("mt")) + dest = os.path.join(TEST_DATA_DIR, "basis_link_mt_dst") + clean_dir(dest) + basis = self._seed_basis(dest, source, "mtbasis", self._basis_tree("mt")) + result, _ = run_client(source, dest, flags=["--link-dest=mtbasis", "--threads"], + port=shared_server.port) + assert result.returncode == 0, f"-m link-dest failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert os.stat(os.path.join(received, self.UNCHANGED)).st_ino == \ + os.stat(os.path.join(basis, self.UNCHANGED)).st_ino + assert _read_file(os.path.join(received, self.ADDED)) == \ + self._source_tree("mt")[self.ADDED] + + def test_link_dest_with_delay_updates_stages_and_publishes_link(self, shared_server): + source = self._make_source("basis_link_delay_src", {self.UNCHANGED: b"v1\n"}) + dest = os.path.join(TEST_DATA_DIR, "basis_link_delay_dst") + clean_dir(dest) + basis = self._seed_basis(dest, source, "delaybasis", {self.UNCHANGED: b"v1\n"}) + result, _ = run_client(source, dest, + flags=["--link-dest=delaybasis", "--delay-updates"], + port=shared_server.port) + assert result.returncode == 0, f"delay-updates link-dest failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + unchanged = os.path.join(received, self.UNCHANGED) + assert os.stat(unchanged).st_ino == \ + os.stat(os.path.join(basis, self.UNCHANGED)).st_ino + assert not os.path.isdir(os.path.join(dest, self.STAGING)), \ + "delay-updates staging tree was not cleaned up" + + def test_delete_does_not_touch_basis_dir(self): + """--delete removes genuine extras but must never treat a basis-dir + snapshot (which a --link-dest run just linked from) as destination + content.""" + source = self._make_source("basis_delete_src", {self.UNCHANGED: b"v1\n"}) + dest = os.path.join(TEST_DATA_DIR, "basis_delete_dst") + clean_dir(dest) + basis = self._seed_basis(dest, source, "delbasis", {self.UNCHANGED: b"v1\n"}) + received = get_dest_received_dir(dest, source) + os.makedirs(received, exist_ok=True) + extra = os.path.join(received, "extra.txt") + with open(extra, "wb") as fh: + fh.write(b"extra") + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, flags=["--link-dest=delbasis", "--delete"], + port=server.port) + assert result.returncode == 0, \ + f"delete+link-dest failed: {result.stderr[:300]}" + assert not os.path.exists(extra), "genuine extra file was not deleted" + assert _read_file(os.path.join(received, self.UNCHANGED)) == b"v1\n" + assert os.path.exists(os.path.join(basis, self.UNCHANGED)), \ + "basis directory was deleted by --delete" + assert os.stat(os.path.join(received, self.UNCHANGED)).st_ino == \ + os.stat(os.path.join(basis, self.UNCHANGED)).st_ino + + def test_delay_delete_keeps_nested_staging_named_dir_as_content(self): + # The real --delay-updates staging directory is protected from --delete + # only as a DIRECT child of the receive root. A nested destination + # directory that merely shares the staging name is ordinary content, so + # its extras must still be deleted (regression guard for the walker). + source = self._make_source("basis_nested_stage_src", + {"top.txt": b"top\n", "sub/real.txt": b"real\n"}) + dest = os.path.join(TEST_DATA_DIR, "basis_nested_stage_dst") + clean_dir(dest) + self._seed_basis(dest, source, "nstbasis", + {"top.txt": b"top\n", "sub/real.txt": b"real\n"}) + received = get_dest_received_dir(dest, source) + nested = os.path.join(received, "sub", self.STAGING) + os.makedirs(nested, exist_ok=True) + extra = os.path.join(nested, "extra.txt") + with open(extra, "wb") as fh: + fh.write(b"nested extra") + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, + flags=["--link-dest=nstbasis", "--delete", + "--delay-updates"], + port=server.port) + assert result.returncode == 0, \ + f"delay-delete nested staging failed: {result.stderr[:300]}" + assert not os.path.exists(extra), \ + "extra inside a nested .fastsync-stage dir was not deleted" + assert not os.path.isdir(nested), \ + "nested .fastsync-stage dir should have been removed after its extra" + assert _read_file(os.path.join(received, "sub", "real.txt")) == b"real\n" + assert not os.path.isdir(os.path.join(dest, self.STAGING)), \ + "real delay-updates staging tree was not cleaned up" + + def test_basis_priority_first_match_wins(self, shared_server): + # Two link-dest dirs both hold the exact file: the FIRST (command-line + # order) basis directory must win and supply the hard link. + source = self._make_source("basis_prio_src", {"f.txt": b"content\n"}) + dest = os.path.join(TEST_DATA_DIR, "basis_prio_dst") + clean_dir(dest) + first = self._seed_basis_file(dest, source, "b1", "f.txt", b"content\n") + self._seed_basis_file(dest, source, "b2", "f.txt", b"content\n") + result, _ = run_client(source, dest, flags=["--link-dest=b1", "--link-dest=b2"], + port=shared_server.port) + assert result.returncode == 0, f"link-dest priority failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert os.stat(os.path.join(received, "f.txt")).st_ino == os.stat(first).st_ino, \ + "first basis dir did not win over the second" + + def test_basis_priority_across_compare_and_link(self, shared_server): + # A compare-dest entry listed BEFORE a link-dest entry shadows it (the + # exact match is found first and nothing is materialized); reversing the + # order lets the link-dest entry win and materialize a hard link. + source = self._make_source("basis_prio_mixed_src", {"f.txt": b"content\n"}) + + dest = os.path.join(TEST_DATA_DIR, "basis_prio_mixed_dst") + clean_dir(dest) + self._seed_basis_file(dest, source, "cmpb", "f.txt", b"content\n") + self._seed_basis_file(dest, source, "lnb", "f.txt", b"content\n") + result, _ = run_client(source, dest, + flags=["--compare-dest=cmpb", "--link-dest=lnb"], + port=shared_server.port) + assert result.returncode == 0, \ + f"mixed priority (compare first) failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert not os.path.exists(os.path.join(received, "f.txt")), \ + "compare-dest matched first, so the file must stay sparse (no link-dest materialize)" + + dest = os.path.join(TEST_DATA_DIR, "basis_prio_mixed_dst2") + clean_dir(dest) + self._seed_basis_file(dest, source, "cmpb", "f.txt", b"content\n") + linkb2 = self._seed_basis_file(dest, source, "lnb", "f.txt", b"content\n") + result, _ = run_client(source, dest, + flags=["--link-dest=lnb", "--compare-dest=cmpb"], + port=shared_server.port) + assert result.returncode == 0, \ + f"mixed priority (link first) failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert os.stat(os.path.join(received, "f.txt")).st_ino == os.stat(linkb2).st_ino, \ + "link-dest did not materialize when listed before compare-dest" + + def test_link_dest_size_only_ignores_mtime(self, shared_server): + # --size-only drops the mtime leg of the quick check: a basis file with + # the SAME content but a DIFFERENT mtime is still an exact match. + source = self._make_source("basis_sizeonly_src", {"f.txt": b"content\n"}) + dest = os.path.join(TEST_DATA_DIR, "basis_sizeonly_dst") + clean_dir(dest) + basis_file = self._seed_basis_file(dest, source, "sob", "f.txt", b"content\n", + ts=self.TS + 500) + result, _ = run_client(source, dest, flags=["--link-dest=sob", "--size-only"], + port=shared_server.port) + assert result.returncode == 0, f"size-only link-dest failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert os.stat(os.path.join(received, "f.txt")).st_ino == os.stat(basis_file).st_ino, \ + "--size-only should link a basis file whose mtime differs" + + @pytest.mark.ci + def test_link_dest_ignore_times_never_links(self, shared_server): + # -I/--ignore-times forces every file to be updated, so a basis dir is + # never used to hard-link (rsync parity). The file is transferred and + # stored as a fresh inode even though it matches the basis exactly. + source = self._make_source("basis_igntimes_src", {"f.txt": b"content\n"}) + dest = os.path.join(TEST_DATA_DIR, "basis_igntimes_dst") + clean_dir(dest) + basis_file = self._seed_basis_file(dest, source, "itb", "f.txt", b"content\n") + result, _ = run_client(source, dest, flags=["--link-dest=itb", "--ignore-times"], + port=shared_server.port) + assert result.returncode == 0, f"ignore-times link-dest failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + dest_file = os.path.join(received, "f.txt") + assert _read_file(dest_file) == b"content\n" + assert os.stat(dest_file).st_ino != os.stat(basis_file).st_ino, \ + "--ignore-times must not hard-link to a basis file" + + def test_basis_refuses_file_above_whole_file_limit(self, shared_server): + # Every whole-file payload path in FastSync (basis dirs included) is + # bounded by MAX_RECEIVE_WHOLE_FILE_SIZE. rsync supports basis dirs for + # arbitrary sizes; FastSync refuses such a run up front with a clear + # diagnostic instead of letting the receiver abort the whole transfer + # mid-stream with no client-side explanation. + source = self._make_source("basis_oversize_src", {"small.txt": b"ok\n"}) + big = os.path.join(source, "huge.bin") + with open(big, "wb") as fh: + os.ftruncate(fh.fileno(), 256 * 1024 * 1024 + 4096) + dest = os.path.join(TEST_DATA_DIR, "basis_oversize_dst") + clean_dir(dest) + result, _ = run_client(source, dest, flags=["--link-dest=nope"], + port=shared_server.port) + assert result.returncode != 0, \ + "basis run with an over-limit file unexpectedly succeeded" + assert "larger than" in result.stderr, \ + f"no clear over-limit diagnostic: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert not os.path.exists(received), \ + "over-limit basis run transferred files before failing" + + +def _random_payloads(size=2 * 1024 * 1024, changed=64 * 1024, seed=1234): + """Return (old, new) byte strings of equal length where `new` differs from + `old` only in one contiguous `changed`-byte region. Incompressible (random) + data keeps the whole-file wire cost near the file size, so a delta transfer + is distinguishable from a whole-file one by its wire bytes.""" + r = random.Random(seed) + data = bytearray(r.randbytes(size)) + old = bytes(data) + off = size // 3 + for i in range(off, off + changed): + data[i] = r.randrange(256) + return old, bytes(data) + + +class TestFuzzy: + """-y/--fuzzy similar-file delta basis. + + Scenario modelled on rsync's --fuzzy: a file is recreated under a similar + NEW basename in the same directory. The destination still holds the + old-named file (nothing deleted it), but the new path has no content of its + own at the destination, so without --fuzzy the receiver has no delta basis + and the whole file is sent. With --fuzzy the receiver searches the + destination directory, picks the similar-named sibling as the delta basis, + sends its block signature, and the sender transmits only the differences. + The reconstructed file must be byte-identical to the source in every mode; + only the wire usage changes (observed through CountingProxy, because client + --stats report source lengths, not wire bytes). + """ + + OLD_NAME = "report-2025.dat" + NEW_NAME = "report-2026.dat" + TS = 1577836800 # 2020-01-01, used to pin stale destination mtimes + SIZE = 2 * 1024 * 1024 + + def _client_via_proxy(self, source, dest, flags, proxy): + cmd = (CLIENT_CMD + ["--source-dir", source, "--dest-dir", dest, + "--save-to-disk", "--server-port", str(proxy.port)] + flags) + return proxy.run(cmd) + + def _seed_dest(self, source, dest, files, port): + """Write `files` {rel: bytes} into source and mirror them to dest.""" + for rel, content in files.items(): + full = os.path.join(source, rel) + os.makedirs(os.path.dirname(full), exist_ok=True) + with open(full, "wb") as fh: + fh.write(content) + clean_dir(dest) + result, _ = run_client(source, dest, port=port) + assert result.returncode == 0, f"seed failed: {(result.stderr or result.stdout)[:300]}" + + def _prepare(self, tag): + source = os.path.join(TEST_DATA_DIR, f"fuzzy_{tag}_src") + dest = os.path.join(TEST_DATA_DIR, f"fuzzy_{tag}_dst") + clean_dir(source) + return source, dest + + def _run_measured(self, source, dest, flags, port): + """Run a transfer through a byte-counting proxy. Returns (result, proxy).""" + proxy = CountingProxy(port) + result = self._client_via_proxy(source, dest, flags, proxy) + return result, proxy + + def test_fuzzy_uses_similar_sibling_as_delta_basis(self, shared_server): + source, dest = self._prepare("basis") + old_bytes, new_bytes = _random_payloads() + self._seed_dest(source, dest, {self.OLD_NAME: old_bytes}, shared_server.port) + # Recreate the file under a similar new name; the old sibling stays on + # the destination (nothing deletes it). + os.unlink(os.path.join(source, self.OLD_NAME)) + with open(os.path.join(source, self.NEW_NAME), "wb") as fh: + fh.write(new_bytes) + + result, proxy = self._run_measured(source, dest, ["--fuzzy"], shared_server.port) + assert result.returncode == 0, \ + f"--fuzzy rename transfer failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.NEW_NAME)) == new_bytes, \ + "fuzzy reconstruction is not byte-exact" + # The wire carried the delta, not the 2 MiB whole file. + assert proxy.client_to_server < len(new_bytes) // 4, \ + f"fuzzy transfer sent {proxy.client_to_server} bytes; expected a delta" + + def test_without_fuzzy_sends_the_whole_file(self, shared_server): + source, dest = self._prepare("whole") + old_bytes, new_bytes = _random_payloads() + self._seed_dest(source, dest, {self.OLD_NAME: old_bytes}, shared_server.port) + os.unlink(os.path.join(source, self.OLD_NAME)) + with open(os.path.join(source, self.NEW_NAME), "wb") as fh: + fh.write(new_bytes) + + result, proxy = self._run_measured(source, dest, ["--incremental", "--delta"], shared_server.port) + assert result.returncode == 0, f"no-fuzzy rename failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.NEW_NAME)) == new_bytes + # No similar basis: the whole file goes over the wire. + assert proxy.client_to_server > len(new_bytes) // 2, \ + f"expected a whole-file transfer, got {proxy.client_to_server} bytes" + + @pytest.mark.parametrize("mt", [False, True]) + def test_fuzzy_byte_exact_single_and_multithreaded(self, shared_server, mt): + source, dest = self._prepare(f"mt{'1' if mt else '0'}") + old_bytes, new_bytes = _random_payloads() + self._seed_dest(source, dest, {self.OLD_NAME: old_bytes}, shared_server.port) + os.unlink(os.path.join(source, self.OLD_NAME)) + with open(os.path.join(source, self.NEW_NAME), "wb") as fh: + fh.write(new_bytes) + + flags = ["--fuzzy"] + (["--threads"] if mt else []) + result, proxy = self._run_measured(source, dest, flags, shared_server.port) + assert result.returncode == 0, \ + f"--fuzzy {'-m ' if mt else ''}rename failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.NEW_NAME)) == new_bytes, \ + f"fuzzy {'-m ' if mt else ''}reconstruction is not byte-exact" + assert proxy.client_to_server < len(new_bytes) // 4 + + def test_no_candidate_falls_back_to_whole_file(self, shared_server): + # A brand-new destination directory holds no sibling at all, so --fuzzy + # finds nothing and the file is transferred whole (and correctly). + source, dest = self._prepare("nocand") + clean_dir(dest) + os.makedirs(source, exist_ok=True) + _, new_bytes = _random_payloads() + with open(os.path.join(source, self.NEW_NAME), "wb") as fh: + fh.write(new_bytes) + result, proxy = self._run_measured(source, dest, ["--fuzzy"], shared_server.port) + assert result.returncode == 0, \ + f"--fuzzy no-candidate fallback failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.NEW_NAME)) == new_bytes + assert proxy.client_to_server > len(new_bytes) // 2, \ + "no-candidate fuzzy run should have sent the whole file" + + def test_dissimilar_sibling_is_not_used(self, shared_server): + # The destination holds a large sibling whose basename is too different + # from the incoming name; the name gate must reject it and fall back to + # a whole-file transfer. + source, dest = self._prepare("dissim") + old_bytes, new_bytes = _random_payloads() + self._seed_dest(source, dest, {"totally-unrelated-notes.bin": old_bytes}, + shared_server.port) + with open(os.path.join(source, self.NEW_NAME), "wb") as fh: + fh.write(new_bytes) + result, proxy = self._run_measured(source, dest, ["--fuzzy"], shared_server.port) + assert result.returncode == 0, \ + f"--fuzzy dissimilar-sibling run failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.NEW_NAME)) == new_bytes + assert proxy.client_to_server > len(new_bytes) // 2, \ + "a dissimilar-named sibling must not be used as a fuzzy basis" + + def test_fuzzy_helps_when_dest_holds_an_unsuitable_file(self, shared_server): + # The destination DOES hold the exact new name, but it is a tiny stale + # file (below the delta engine's minimum, ratio far outside its window), + # so it cannot serve as the basis. --fuzzy then falls back to the + # similar-named sibling. + source, dest = self._prepare("unsuitable") + old_bytes, new_bytes = _random_payloads() + self._seed_dest(source, dest, + {self.OLD_NAME: old_bytes, self.NEW_NAME: b"stale small file\n"}, + shared_server.port) + with open(os.path.join(source, self.NEW_NAME), "wb") as fh: + fh.write(new_bytes) + result, proxy = self._run_measured(source, dest, ["--fuzzy"], shared_server.port) + assert result.returncode == 0, \ + f"--fuzzy unsuitable-dest run failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.NEW_NAME)) == new_bytes, \ + "byte-exactness broken when the destination file was unsuitable" + assert proxy.client_to_server < len(new_bytes) // 4, \ + "fuzzy should have reused the similar sibling as the basis" + + def test_whole_file_makes_fuzzy_inert(self, shared_server): + # -W/--whole-file switches the delta machinery off, so --fuzzy has + # nothing to attach to and the file is transferred whole (rsync parity). + source, dest = self._prepare("wholefile") + old_bytes, new_bytes = _random_payloads() + self._seed_dest(source, dest, {self.OLD_NAME: old_bytes}, shared_server.port) + os.unlink(os.path.join(source, self.OLD_NAME)) + with open(os.path.join(source, self.NEW_NAME), "wb") as fh: + fh.write(new_bytes) + result, proxy = self._run_measured(source, dest, ["--fuzzy", "-W"], shared_server.port) + assert result.returncode == 0, \ + f"--fuzzy -W run failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.NEW_NAME)) == new_bytes + assert proxy.client_to_server > len(new_bytes) // 2, \ + "--whole-file must disable the fuzzy delta basis" + + def test_fuzzy_via_short_y_alias(self, shared_server): + source, dest = self._prepare("shorty") + old_bytes, new_bytes = _random_payloads() + self._seed_dest(source, dest, {self.OLD_NAME: old_bytes}, shared_server.port) + os.unlink(os.path.join(source, self.OLD_NAME)) + with open(os.path.join(source, self.NEW_NAME), "wb") as fh: + fh.write(new_bytes) + result, proxy = self._run_measured(source, dest, ["-y"], shared_server.port) + assert result.returncode == 0, f"-y rename failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.NEW_NAME)) == new_bytes + assert proxy.client_to_server < len(new_bytes) // 4, "-y did not enable fuzzy" + + def test_fuzzy_with_delay_updates_publishes_cleanly(self, shared_server): + # A fuzzy-reconstructed file goes through the normal store engine, so + # --delay-updates must stage and publish it with no staging leftovers. + source, dest = self._prepare("delay") + old_bytes, new_bytes = _random_payloads() + self._seed_dest(source, dest, {self.OLD_NAME: old_bytes}, shared_server.port) + os.unlink(os.path.join(source, self.OLD_NAME)) + with open(os.path.join(source, self.NEW_NAME), "wb") as fh: + fh.write(new_bytes) + result, _ = run_client(source, dest, + flags=["--fuzzy", "--delay-updates"], + port=shared_server.port) + assert result.returncode == 0, \ + f"--fuzzy --delay-updates failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.NEW_NAME)) == new_bytes + assert not os.path.isdir(os.path.join(dest, ".fastsync-stage")), \ + "delay-updates staging tree was not cleaned up" + + def test_fuzzy_source_removed_after_transfer(self, shared_server): + # A fuzzy transfer is a real transfer (not a skip), so + # --remove-source-files must remove the renamed source file. + source, dest = self._prepare("rm") + old_bytes, new_bytes = _random_payloads() + self._seed_dest(source, dest, {self.OLD_NAME: old_bytes}, shared_server.port) + os.unlink(os.path.join(source, self.OLD_NAME)) + with open(os.path.join(source, self.NEW_NAME), "wb") as fh: + fh.write(new_bytes) + result, _ = run_client(source, dest, + flags=["--fuzzy", "--remove-source-files"], + port=shared_server.port) + assert result.returncode == 0, \ + f"--fuzzy --remove-source-files failed: {(result.stderr or result.stdout)[:300]}" + assert not os.path.exists(os.path.join(source, self.NEW_NAME)), \ + "a fuzzy-transferred source should have been removed" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.NEW_NAME)) == new_bytes + assert _read_file(os.path.join(received, self.OLD_NAME)) == old_bytes + + @staticmethod + def _rand_bytes(size, seed): + return random.Random(seed).randbytes(size) + + def _replace_source_file(self, source, old_name, new_name, new_bytes): + """Remove old_name from source and add new_name with new_bytes.""" + os.unlink(os.path.join(source, old_name)) + with open(os.path.join(source, new_name), "wb") as fh: + fh.write(new_bytes) + + def test_worthless_fuzzy_basis_falls_back_inside_handshake(self, shared_server): + # The sibling passes the name AND size gates but shares no blocks with + # the incoming file, so the sender's delta is not worthwhile: it replies + # STATUS_NEXT and the receiver consumes the WHOLE file inside the delta + # handshake. This proves a bad fuzzy basis cannot desync the protocol + # or corrupt the result. + source, dest = self._prepare("worthless") + basis = self._rand_bytes(self.SIZE, 424242) + target = self._rand_bytes(self.SIZE, 777777) + self._seed_dest(source, dest, {self.OLD_NAME: basis}, shared_server.port) + self._replace_source_file(source, self.OLD_NAME, self.NEW_NAME, target) + result, proxy = self._run_measured(source, dest, ["--fuzzy"], shared_server.port) + assert result.returncode == 0, \ + f"--fuzzy worthless-basis run failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.NEW_NAME)) == target, \ + "whole-file fallback after a worthless fuzzy basis is not byte-exact" + assert proxy.client_to_server > self.SIZE // 2, \ + "a worthless basis should have made the sender fall back to the whole file" + + def test_existing_dest_file_preferred_over_fuzzy_sibling(self, shared_server): + # Non-displacement: the destination holds a file at the exact path that + # is inside the delta size bounds (same size, different content, older + # mtime). FastSync must delta against THAT file -- even though it + # shares nothing with the source -- and must NOT reuse a similar-named + # sibling that is byte-identical to the source. + source, dest = self._prepare("nondisp") + sibling = self._rand_bytes(self.SIZE, 111) # will equal the incoming file + stale = self._rand_bytes(self.SIZE, 333) # worthless exact-path file + self._seed_dest(source, dest, + {self.OLD_NAME: sibling, self.NEW_NAME: stale}, + shared_server.port) + # Force the exact-path destination file's mtime into the past so the + # quick check deterministically decides to transfer it. + os.utime(os.path.join(get_dest_received_dir(dest, source), self.NEW_NAME), + (self.TS, self.TS)) + os.unlink(os.path.join(source, self.OLD_NAME)) + with open(os.path.join(source, self.NEW_NAME), "wb") as fh: + fh.write(sibling) + result, proxy = self._run_measured(source, dest, ["--fuzzy"], shared_server.port) + assert result.returncode == 0, \ + f"--fuzzy non-displacement run failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.NEW_NAME)) == sibling + assert proxy.client_to_server > self.SIZE // 2, \ + "the exact-path destination file must be the delta basis, not the fuzzy sibling" + + def test_fuzzy_basis_larger_than_source(self, shared_server): + # The similar sibling is LARGER than the incoming file (within the delta + # engine's 10x ratio); the new file is an exact prefix of the basis, so + # every block matches and only a tiny delta travels. + source, dest = self._prepare("largerbasis") + big = self._rand_bytes(1536 * 1024, 1) + prefix = big[:1024 * 1024] + self._seed_dest(source, dest, {self.OLD_NAME: big}, shared_server.port) + self._replace_source_file(source, self.OLD_NAME, self.NEW_NAME, prefix) + result, proxy = self._run_measured(source, dest, ["--fuzzy"], shared_server.port) + assert result.returncode == 0, \ + f"--fuzzy larger-basis run failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.NEW_NAME)) == prefix, \ + "shrunken file reconstructed from a larger fuzzy basis is not byte-exact" + assert proxy.client_to_server < len(prefix) // 4, \ + "larger fuzzy basis should have carried most of the file as block matches" + + def test_fuzzy_basis_smaller_than_source(self, shared_server): + # The similar sibling is SMALLER than the incoming file; the new file + # appends data past the basis, so the appended tail travels as literals + # while the shared prefix is block-matched. + source, dest = self._prepare("smallerbasis") + base = self._rand_bytes(self.SIZE, 2) + tail = self._rand_bytes(64 * 1024, 3) + new_bytes = base + tail + self._seed_dest(source, dest, {self.OLD_NAME: base}, shared_server.port) + self._replace_source_file(source, self.OLD_NAME, self.NEW_NAME, new_bytes) + result, proxy = self._run_measured(source, dest, ["--fuzzy"], shared_server.port) + assert result.returncode == 0, \ + f"--fuzzy smaller-basis run failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.NEW_NAME)) == new_bytes, \ + "grown file reconstructed from a smaller fuzzy basis is not byte-exact" + assert proxy.client_to_server < len(new_bytes) // 4, \ + "smaller fuzzy basis should have block-matched the shared prefix" + + def test_no_fuzzy_end_to_end_equals_no_flag(self, shared_server): + # --no-fuzzy must not enable anything: a run with it behaves exactly + # like a run without it (whole-file transfer, byte-exact output). + source, dest = self._prepare("nofuzzye2e") + old_bytes, new_bytes = _random_payloads() + self._seed_dest(source, dest, {self.OLD_NAME: old_bytes}, shared_server.port) + self._replace_source_file(source, self.OLD_NAME, self.NEW_NAME, new_bytes) + result, proxy = self._run_measured(source, dest, ["--no-fuzzy"], shared_server.port) + assert result.returncode == 0, \ + f"--no-fuzzy run failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.NEW_NAME)) == new_bytes + assert proxy.client_to_server > len(new_bytes) // 2, \ + "--no-fuzzy should leave the default whole-file behavior intact" + + + +class TestIdentityMapping: + """Ownership-application flags (--numeric-ids / --usermap / --groupmap / + --chown). In CI the receiver usually runs unprivileged, so ownership apply + is expected to fail from lack of privilege: the transfer must STILL succeed + and exit 0 (the receiver warns and continues, rsync parity). The only + assertion that requires the ownership to actually change is gated on + os.geteuid() == 0 so it is skipped (not failed) as a non-root user.""" + + def test_numeric_ids_transfer_succeeds_unprivileged(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "identity_num_source") + dest = os.path.join(TEST_DATA_DIR, "identity_num_dest") + clean_dir(source) + clean_dir(dest) + with open(os.path.join(source, "f.txt"), "wb") as f: + f.write(b"hello identity") + result, _ = run_client(source, dest, + flags=["--preserve", "--numeric-ids"], + port=shared_server.port) + assert result.returncode == 0, \ + f"exit {result.returncode}: {(result.stderr or '')[:200]}" + received = get_dest_received_dir(dest, source) + with open(os.path.join(received, "f.txt"), "rb") as f: + assert f.read() == b"hello identity" + + def test_usermap_and_groupmap_and_chown_succeed_unprivileged(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "identity_map_source") + dest = os.path.join(TEST_DATA_DIR, "identity_map_dest") + clean_dir(source) + clean_dir(dest) + with open(os.path.join(source, "f.txt"), "wb") as f: + f.write(b"mapped") + result, _ = run_client( + source, dest, + flags=["--preserve", "--usermap=@1000:@1001", "--groupmap=@100:@101", "--chown=@2000:@2001"], + port=shared_server.port) + assert result.returncode == 0, \ + f"exit {result.returncode}: {(result.stderr or '')[:200]}" + received = get_dest_received_dir(dest, source) + with open(os.path.join(received, "f.txt"), "rb") as f: + assert f.read() == b"mapped" + + @pytest.mark.skipif(os.geteuid() != 0, reason="only root can change ownership") + def test_numeric_ids_applies_ownership_as_root(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "identity_root_source") + dest = os.path.join(TEST_DATA_DIR, "identity_root_dest") + clean_dir(source) + clean_dir(dest) + src_file = os.path.join(source, "f.txt") + with open(src_file, "wb") as f: + f.write(b"owner") + os.chown(src_file, 12345, 12346) + result, _ = run_client(source, dest, + flags=["--preserve", "--numeric-ids"], + port=shared_server.port) + assert result.returncode == 0, \ + f"exit {result.returncode}: {(result.stderr or '')[:200]}" + received = get_dest_received_dir(dest, source) + dst_file = os.path.join(received, "f.txt") + assert os.path.exists(dst_file) + st = os.stat(dst_file) + assert st.st_uid == 12345 and st.st_gid == 12346, \ + f"owner not applied: uid={st.st_uid} gid={st.st_gid}" + + @pytest.mark.skipif(os.geteuid() != 0, reason="only root can change ownership") + def test_chown_overrides_ownership_as_root(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "identity_chown_root_source") + dest = os.path.join(TEST_DATA_DIR, "identity_chown_root_dest") + clean_dir(source) + clean_dir(dest) + src_file = os.path.join(source, "f.txt") + with open(src_file, "wb") as f: + f.write(b"root chown") + os.chown(src_file, 1, 1) + result, _ = run_client(source, dest, + flags=["--preserve", "--chown=@12345:@54321"], + port=shared_server.port) + assert result.returncode == 0, \ + f"exit {result.returncode}: {(result.stderr or '')[:200]}" + received = get_dest_received_dir(dest, source) + dst_file = os.path.join(received, "f.txt") + assert os.path.exists(dst_file) + st = os.stat(dst_file) + assert st.st_uid == 12345 and st.st_gid == 54321, \ + f"--chown not applied: uid={st.st_uid} gid={st.st_gid}" + + +class TestSuperPrivilege: + """P7 Wave E: --super / --no-super control the receiver's already-confined + super-user activities (ownership application, char/block device nodes). + FastSync never elevates, so on an unprivileged receiver --super only + permits a confined attempt (which then skips); --no-super forbids the + activity even for root.""" + + def _seed(self, tag): + source = os.path.join(TEST_DATA_DIR, f"super_{tag}_source") + dest = os.path.join(TEST_DATA_DIR, f"super_{tag}_dest") + clean_dir(source) + clean_dir(dest) + with open(os.path.join(source, "f.txt"), "wb") as f: + f.write(b"super privilege\n") + return source, dest + + def test_super_and_no_super_transfer_successfully(self, shared_server): + """Both flags parse and the transfer completes normally regardless of + the receiver's privilege level.""" + for flag in ("--super", "--no-super"): + source, dest = self._seed(flag.strip("-")) + result, _ = run_client(source, dest, flags=[flag], port=shared_server.port) + assert result.returncode == 0, \ + f"{flag} exit {result.returncode}: {(result.stderr or '')[:300]}" + received = get_dest_received_dir(dest, source) + with open(os.path.join(received, "f.txt"), "rb") as f: + assert f.read() == b"super privilege\n" + + @pytest.mark.skipif(os.geteuid() != 0, reason="only root can change ownership") + def test_no_super_suppresses_ownership_as_root(self, shared_server): + """As root the default gate would apply a raw numeric id; --no-super + must suppress that ownership application entirely.""" + source, dest = self._seed("nosuper") + os.chown(os.path.join(source, "f.txt"), 12345, 12346) + result, _ = run_client(source, dest, + flags=["--preserve", "--numeric-ids", "--no-super"], + port=shared_server.port) + assert result.returncode == 0, \ + f"exit {result.returncode}: {(result.stderr or '')[:300]}" + received = get_dest_received_dir(dest, source) + st = os.stat(os.path.join(received, "f.txt")) + assert (st.st_uid, st.st_gid) != (12345, 12346), \ + f"--no-super must not apply ownership (uid={st.st_uid} gid={st.st_gid})" + + @pytest.mark.skipif(os.geteuid() != 0, reason="only root can change ownership") + def test_super_alone_does_not_apply_ownership_as_root(self, shared_server): + """A3: --super no longer implies --numeric-ids, so --super alone must NOT + apply client-chosen ownership even for root; the destination keeps the + receiver's owner (the exact ownership --no-super would also suppress).""" + source, dest = self._seed("superonly") + os.chown(os.path.join(source, "f.txt"), 12345, 12346) + result, _ = run_client(source, dest, + flags=["--preserve", "--super"], + port=shared_server.port) + assert result.returncode == 0, \ + f"exit {result.returncode}: {(result.stderr or '')[:300]}" + received = get_dest_received_dir(dest, source) + st = os.stat(os.path.join(received, "f.txt")) + assert (st.st_uid, st.st_gid) != (12345, 12346), \ + f"--super alone must not apply ownership (uid={st.st_uid} gid={st.st_gid})" + + @pytest.mark.skipif(os.geteuid() != 0, reason="only root can change ownership") + def test_super_with_numeric_ids_applies_ownership_as_root(self, shared_server): + """Control: an explicit identity policy is what enables ownership, so + --numeric-ids --super still applies the raw ids as root (the very + ownership --no-super suppresses).""" + source, dest = self._seed("supernumeric") + os.chown(os.path.join(source, "f.txt"), 12345, 12346) + result, _ = run_client(source, dest, + flags=["--preserve", "--numeric-ids", "--super"], + port=shared_server.port) + assert result.returncode == 0, \ + f"exit {result.returncode}: {(result.stderr or '')[:300]}" + received = get_dest_received_dir(dest, source) + st = os.stat(os.path.join(received, "f.txt")) + assert (st.st_uid, st.st_gid) == (12345, 12346), \ + f"--numeric-ids --super should apply raw ids: uid={st.st_uid} gid={st.st_gid}" + + @pytest.mark.ci + @pytest.mark.skipif(os.geteuid() != 0, reason="only root can change ownership") + def test_fake_super_no_super_does_not_change_owner(self, shared_server): + """--fake-super records the source owner, but --no-super must suppress the + live chown even for root: the destination keeps the receiver's owner + instead of the recorded source owner.""" + source, dest = self._seed("fakesuper_nosuper") + os.chown(os.path.join(source, "f.txt"), 12345, 12346) + result, _ = run_client(source, dest, + flags=["--fake-super", "--preserve", "--no-super"], + port=shared_server.port) + assert result.returncode == 0, \ + f"exit {result.returncode}: {(result.stderr or '')[:300]}" + received = get_dest_received_dir(dest, source) + st = os.lstat(os.path.join(received, "f.txt")) + assert (st.st_uid, st.st_gid) != (12345, 12346), \ + f"--no-super must suppress fake-super's owner replay: uid={st.st_uid} gid={st.st_gid}" + + +class TestHardLinks: + """-H/--hard-links: source files sharing an inode are re-created as hard + links to one another on the destination (dedup preserved, first copy + transferred once, the rest linked/copied). No root required.""" + + STAGING = ".fastsync-stage" + + def _make_source(self, name): + src = os.path.join(TEST_DATA_DIR, name) + clean_dir(src) + with open(os.path.join(src, "a.txt"), "wb") as fh: + fh.write(b"shared content\n" * 2000) + os.link(os.path.join(src, "a.txt"), os.path.join(src, "b.txt")) + with open(os.path.join(src, "c.txt"), "wb") as fh: + fh.write(b"independent content\n" * 2000) + return src + + @pytest.mark.parametrize("flags", [[], ["--threads"], ["--delay-updates"]]) + def test_hard_links_preserved(self, shared_server, flags): + src = self._make_source("hl_src") + dest = os.path.join(TEST_DATA_DIR, "hl_dst") + clean_dir(dest) + result, _ = run_client(src, dest, flags=["-H"] + flags, port=shared_server.port) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:300]}" + received = get_dest_received_dir(dest, src) + a = os.path.join(received, "a.txt") + b = os.path.join(received, "b.txt") + c = os.path.join(received, "c.txt") + assert os.path.isfile(a) and os.path.isfile(b) and os.path.isfile(c), \ + "all three destination files exist" + with open(a, "rb") as fa, open(b, "rb") as fb: + assert fa.read() == fb.read(), "hard-linked pair content matches" + assert os.stat(a).st_ino == os.stat(b).st_ino, \ + "source hard links were not preserved on the destination" + assert os.stat(a).st_ino != os.stat(c).st_ino, \ + "independent files were incorrectly hard linked" + with open(a, "rb") as fa, open(c, "rb") as fc: + assert fa.read() != fc.read(), "independent files must differ in content" + assert not os.path.isdir(os.path.join(dest, self.STAGING)), \ + "--delay-updates left a staging tree behind" + + def test_hard_links_rejects_chunk_serialization(self, shared_server): + src = self._make_source("hl_reject_src") + dest = os.path.join(TEST_DATA_DIR, "hl_reject_dst") + clean_dir(dest) + result, _ = run_client(src, dest, flags=["-H", "--chunk-serialization"], port=shared_server.port) + assert result.returncode != 0, "-H with -s was accepted" + + def test_hard_links_rejects_append(self, shared_server): + src = self._make_source("hl_reject_app_src") + dest = os.path.join(TEST_DATA_DIR, "hl_reject_app_dst") + clean_dir(dest) + result, _ = run_client(src, dest, flags=["-H", "--append"], port=shared_server.port) + assert result.returncode != 0, "-H with --append was accepted" + + def test_hard_links_link_to_existing_first_member(self, shared_server): + """A sibling whose first member is already up-to-date at the destination + must still be created as a hard link to that existing file.""" + src = os.path.join(TEST_DATA_DIR, "hl_exist_src") + dest = os.path.join(TEST_DATA_DIR, "hl_exist_dst") + clean_dir(src) + clean_dir(dest) + with open(os.path.join(src, "a.txt"), "wb") as fh: + fh.write(b"seed content\n" * 1500) + result, _ = run_client(src, dest, port=shared_server.port) + assert result.returncode == 0, f"seed failed: {result.stderr[:200]}" + # Introduce a hard-link sibling to the already-transferred first member. + os.link(os.path.join(src, "a.txt"), os.path.join(src, "b.txt")) + result, _ = run_client(src, dest, flags=["-H"], port=shared_server.port) + assert result.returncode == 0, f"-H sync failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, src) + a = os.path.join(received, "a.txt") + b = os.path.join(received, "b.txt") + assert os.path.isfile(a) and os.path.isfile(b) + assert os.stat(a).st_ino == os.stat(b).st_ino, \ + "new sibling was not linked to the existing first member" + with open(a, "rb") as fa, open(b, "rb") as fb: + assert fa.read() == fb.read() + + def test_hard_links_existing_asymmetric_group(self, shared_server): + """-H --existing with an asymmetric link group must succeed: when the + first member's destination is absent (so it is skipped by --existing) + but a sibling's destination already exists, the existing sibling is left + in place instead of the whole transfer aborting on the absent first + member.""" + src = os.path.join(TEST_DATA_DIR, "hl_existing_src") + dest = os.path.join(TEST_DATA_DIR, "hl_existing_dst") + clean_dir(src) + clean_dir(dest) + with open(os.path.join(src, "a.txt"), "wb") as fh: + fh.write(b"asymmetric group content\n" * 1200) + # b.txt is a hard-link sibling of a.txt on the source. + os.link(os.path.join(src, "a.txt"), os.path.join(src, "b.txt")) + with open(os.path.join(src, "c.txt"), "wb") as fh: + fh.write(b"independent\n" * 1200) + # Pre-seed the destination with ONLY the sibling's file (the first + # member has no destination entry). + received = get_dest_received_dir(dest, src) + os.makedirs(received, exist_ok=True) + with open(os.path.join(received, "b.txt"), "wb") as fh: + fh.write(b"asymmetric group content\n" * 1200) + result, _ = run_client(src, dest, flags=["-H", "--existing"], + port=shared_server.port) + assert result.returncode == 0, \ + f"-H --existing asymmetric group failed: {result.stderr[:300]}" + # The existing sibling was preserved and its content is intact. + with open(os.path.join(received, "b.txt"), "rb") as fh: + assert fh.read() == b"asymmetric group content\n" * 1200 + # Under --existing the absent first member is not created. + assert not os.path.exists(os.path.join(received, "a.txt")) + +class TestAtimes: + """-U/--atimes preserves the source access time on the destination. + + The sender captures atime during the scan (a stat, before any read for + transfer), so the value is not clobbered by reading the source. This is + verified by setting the source atime to a distinct value far in the past + and comparing the destination atime to it (with whole-second tolerance; + filesystems may round atime).""" + + PAYLOAD = b"atime preservation payload\n" + + @staticmethod + def _make_source(source, dest): + clean_dir(source) + clean_dir(dest) + path = os.path.join(source, "data.txt") + with open(path, "wb") as f: + f.write(TestAtimes.PAYLOAD) + atime = 946684800 # 2000-01-01 00:00:00 UTC (far from "now") + mtime = 951782400 + os.utime(path, ns=(atime * 10**9 + 123456789, mtime * 10**9)) + return path, atime + + @pytest.mark.ci + @pytest.mark.parametrize("mt", [False, True]) + def test_atimes_preserved(self, shared_server, mt): + source = os.path.join(TEST_DATA_DIR, f"atime_{'m' if mt else 's'}_src") + dest = os.path.join(TEST_DATA_DIR, f"atime_{'m' if mt else 's'}_dst") + src_file, atime = self._make_source(source, dest) + flags = ["-U"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, \ + f"-U failed: {(result.stderr or result.stdout)[:300]}" + + received = get_dest_received_dir(dest, source) + dst_file = os.path.join(received, "data.txt") + assert os.path.exists(dst_file) + dst_st = os.stat(dst_file) + assert abs(dst_st.st_atime - atime) < 1.5, \ + f"dest atime {dst_st.st_atime} != source atime {atime}" + + @pytest.mark.ci + @pytest.mark.parametrize("mt", [False, True]) + def test_without_atimes_dest_differs(self, shared_server, mt): + """Control: without -U the destination atime is not the source's old + value (it reflects the fresh write, i.e. now), proving -U is what + restores the source atime.""" + source = os.path.join(TEST_DATA_DIR, f"atime_ctrl_{'m' if mt else 's'}_src") + dest = os.path.join(TEST_DATA_DIR, f"atime_ctrl_{'m' if mt else 's'}_dst") + src_file, atime = self._make_source(source, dest) + now = time.time() + flags = (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, f"control run failed: {(result.stderr or '')[:200]}" + received = get_dest_received_dir(dest, source) + dst_st = os.stat(os.path.join(received, "data.txt")) + # The fresh destination atime is ~now, not the source's year-2000 value. + assert abs(dst_st.st_atime - atime) > 24 * 3600, \ + f"control dest atime {dst_st.st_atime} unexpectedly equals source atime {atime}" + assert abs(dst_st.st_atime - now) < 24 * 3600, \ + f"control dest atime {dst_st.st_atime} not ~now ({now})" + + +class TestOpenNoatime: + """--open-noatime opens the source with O_NOATIME so a transfer read does + not bump the source's access time. O_NOATIME is honoured for a file owned + by the reading process (or with CAP_FOWNER), so it works as non-root here; + where it is unavailable/refused FastSync degrades to a normal open and the + assertion below is skipped.""" + + @pytest.mark.skipif(not sys.platform.startswith("linux"), + reason="O_NOATIME is Linux-specific") + @pytest.mark.parametrize("mt", [False, True]) + def test_open_noatime_preserves_source_atime(self, shared_server, mt): + source = os.path.join(TEST_DATA_DIR, f"noatime_{'m' if mt else 's'}_src") + dest = os.path.join(TEST_DATA_DIR, f"noatime_{'m' if mt else 's'}_dst") + clean_dir(source) + clean_dir(dest) + path = os.path.join(source, "data.txt") + with open(path, "wb") as f: + f.write(b"open-noatime payload\n") + atime = 730486800 # 1993-02-11, distinct and far from now + os.utime(path, ns=(atime * 10**9, atime * 10**9)) + + flags = ["--open-noatime"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, \ + f"--open-noatime failed: {(result.stderr or result.stdout)[:300]}" + + after = os.stat(path) + assert abs(after.st_atime - atime) < 1.5, \ + f"source atime {after.st_atime} was bumped by the readable read (wanted {atime})" + + +class TestCrtimes: + """-N/--crtimes captures and transmits the source birth time. There is no + portable way to SET a birth time (utimensat only sets atime/mtime), so the + receiver deliberately does not apply it. The run must succeed without + crashing; we do not assert the destination birth time changed. When the + platform exposes a birth time (statx STATX_BTIME on Linux) we additionally + confirm a capture path exists.""" + + @pytest.mark.parametrize("mt", [False, True]) + def test_crtimes_run_succeeds(self, shared_server, mt): + source = os.path.join(TEST_DATA_DIR, f"crtime_{'m' if mt else 's'}_src") + dest = os.path.join(TEST_DATA_DIR, f"crtime_{'m' if mt else 's'}_dst") + clean_dir(source) + clean_dir(dest) + path = os.path.join(source, "data.txt") + payload = b"crtime transfer payload\n" + with open(path, "wb") as f: + f.write(payload) + + flags = ["-N"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, \ + f"-N failed: {(result.stderr or result.stdout)[:300]}" + + received = get_dest_received_dir(dest, source) + dst_file = os.path.join(received, "data.txt") + assert os.path.exists(dst_file) + with open(dst_file, "rb") as f: + assert f.read() == payload + + def test_crtimes_combines_with_atimes(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "crtime_atime_combined_src") + dest = os.path.join(TEST_DATA_DIR, "crtime_atime_combined_dst") + clean_dir(source) + clean_dir(dest) + path = os.path.join(source, "data.txt") + with open(path, "wb") as f: + f.write(b"combined U N payload\n") + atime = 946684800 + os.utime(path, ns=(atime * 10**9, 951782400 * 10**9)) + result, _ = run_client(source, dest, flags=["-U", "-N"], + port=shared_server.port) + assert result.returncode == 0, \ + f"-U -N failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + dst_st = os.stat(os.path.join(received, "data.txt")) + assert abs(dst_st.st_atime - atime) < 1.5, \ + f"combined -U -N dest atime {dst_st.st_atime} != {atime}" + + +class TestSparse: + """-S/--sparse: the receiver preserves holes by skipping long zero runs with + lseek (no wire change; the full image is in memory). The destination file + must round-trip its logical size and content byte-for-byte; on filesystems + that report holes (SEEK_HOLE/SEEK_DATA) we additionally assert the file is + genuinely sparse via st_blocks, but that check is tolerant (CI filesystems + may report no holes).""" + + def _make_sparse_source(self, name, total, zero_start, zero_len): + source = os.path.join(TEST_DATA_DIR, name) + clean_dir(source) + sfile = os.path.join(source, "blob.bin") + with open(sfile, "wb") as f: + head = os.urandom(zero_start) + tail = os.urandom(total - zero_start - zero_len) + f.write(head) + f.write(b"\x00" * zero_len) + f.write(tail) + assert f.tell() == total + return source, sfile + + @pytest.mark.parametrize("flag", ["-S", "--sparse"]) + @pytest.mark.parametrize("mt", [False, True]) + def test_sparse_transfer_round_trips(self, shared_server, flag, mt): + total = 4 * 1024 * 1024 + source = os.path.join(TEST_DATA_DIR, f"sparse_mt{mt}_{flag.lstrip('-')}_src") + dest = os.path.join(TEST_DATA_DIR, f"sparse_mt{mt}_{flag.lstrip('-')}_dst") + clean_dir(source) + clean_dir(dest) + zero_start = 1 * 1024 * 1024 + zero_len = 2 * 1024 * 1024 + _, sfile = self._make_sparse_source(os.path.basename(source), total, zero_start, zero_len) + with open(sfile, "rb") as f: + src_bytes = f.read() + + flags = [flag] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, \ + f"{flag} transfer failed: {(result.stderr or result.stdout)[:300]}" + + received = get_dest_received_dir(dest, source) + dfile = os.path.join(received, "blob.bin") + assert os.path.getsize(dfile) == total, "logical size must match data_size" + with open(dfile, "rb") as f: + assert f.read() == src_bytes, "sparse destination content must round-trip exactly" + + # Tolerant sparseness assert: if the filesystem reports holes, the file + # must actually be sparse (fewer allocated blocks than its size). + with open(dfile, "rb") as f: + off = os.lseek(f.fileno(), zero_start, os.SEEK_DATA) + if off >= 0: + hole = os.lseek(f.fileno(), off, os.SEEK_HOLE) + else: + hole = -1 + if hole > zero_start: + st = os.stat(dfile) + assert st.st_blocks * 512 < total, \ + f"-S file not sparse: {st.st_blocks} blocks for {total} bytes" + + def test_sparse_inplace(self, shared_server): + """--sparse must also preserve holes in the --inplace write path.""" + total = 2 * 1024 * 1024 + source = os.path.join(TEST_DATA_DIR, "sparse_inplace_src") + dest = os.path.join(TEST_DATA_DIR, "sparse_inplace_dst") + clean_dir(source) + clean_dir(dest) + _, sfile = self._make_sparse_source(os.path.basename(source), total, total // 2, + total // 4) + with open(sfile, "rb") as f: + src_bytes = f.read() + result, _ = run_client(source, dest, flags=["-S", "--inplace"], port=shared_server.port) + assert result.returncode == 0, \ + f"-S --inplace failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + dfile = os.path.join(received, "blob.bin") + assert os.path.getsize(dfile) == total + with open(dfile, "rb") as f: + assert f.read() == src_bytes + + +class TestBlockSize: + """--block-size / --delta-block: the checksum block size is genuinely honored + by the delta engine (both spellings parse to config->delta_block_size). An + end-to-end delta transfer with a non-default block size must still be + byte-exact.""" + + @pytest.mark.parametrize("flag", ["--block-size", "--delta-block"]) + def test_non_default_block_size_delta_transfer(self, shared_server, flag): + source = os.path.join(TEST_DATA_DIR, "blocksize_delta_src") + dest = os.path.join(TEST_DATA_DIR, "blocksize_delta_dst") + clean_dir(source) + clean_dir(dest) + payload = os.urandom(300 * 1024) # enough for several 1 KiB blocks + with open(os.path.join(source, "big.bin"), "wb") as f: + f.write(payload) + # First run installs the file as the destination basis (do NOT wipe it + # afterwards: the second run's delta must be computed against it). + result, _ = run_client(source, dest, port=shared_server.port) + assert result.returncode == 0 + received = get_dest_received_dir(dest, source) + # Extend the source so it differs from the installed basis: the second + # run with --delta must compute a real delta against that basis. + with open(os.path.join(source, "big.bin"), "ab") as f: + f.write(os.urandom(4096)) + result, _ = run_client(source, dest, + flags=["--incremental", "--delta", flag, "1024"], + port=shared_server.port) + assert result.returncode == 0, \ + f"{flag} 1024 delta transfer failed: {(result.stderr or result.stdout)[:300]}" + with open(os.path.join(received, "big.bin"), "rb") as f: + with open(os.path.join(source, "big.bin"), "rb") as expect: + assert f.read() == expect.read() + +class TestOmitTimes: + """-O/--omit-dir-times and -J/--omit-link-times are recognized and cross the + wire as receiver-side preferences. FastSync does not currently apply dir or + symlink times at all, so they are forward-compatible preferences: the run + must succeed and normal transfers must not break. A regular file's mtime + (from -M) is unaffected by -O/-J.""" + + @pytest.mark.parametrize("flag", ["-O", "-J"]) + @pytest.mark.parametrize("mt", [False, True]) + def test_omit_times_accepted(self, shared_server, flag, mt): + source = os.path.join(TEST_DATA_DIR, f"omit_{flag.strip('-')}_{'m' if mt else 's'}_src") + dest = os.path.join(TEST_DATA_DIR, f"omit_{flag.strip('-')}_{'m' if mt else 's'}_dst") + clean_dir(source) + clean_dir(dest) + with open(os.path.join(source, "a.txt"), "wb") as f: + f.write(b"omit times content\n") + flags = [flag] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, \ + f"{flag} failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + mismatches, missing = verify_transfer(source, received) + assert not missing, f"Missing: {missing}" + assert not mismatches, f"Mismatch: {mismatches}" + + @pytest.mark.ci + def test_omit_times_with_dirs_and_regular_metadata(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "omit_dirs_meta_src") + dest = os.path.join(TEST_DATA_DIR, "omit_dirs_meta_dst") + clean_dir(source) + clean_dir(dest) + os.makedirs(os.path.join(source, "subdir")) + with open(os.path.join(source, "f.txt"), "wb") as f: + f.write(b"regular mtime preserved under -O/-J\n") + result, _ = run_client(source, dest, flags=["-d", "--omit-dir-times"], + port=shared_server.port) + assert result.returncode == 0, \ + f"-d -O failed: {(result.stderr or result.stdout)[:300]}" + result, _ = run_client(source, dest, flags=["--preserve", "-O", "-J"], + port=shared_server.port) + assert result.returncode == 0, \ + f"-M -O -J failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + mismatches, missing = verify_transfer(source, received) + assert not missing and not mismatches, f"missing={missing} mismatches={mismatches}" + + +class TestSymlinkTrust: + """Phase-4 symlink trust boundaries: -k/--copy-dirlinks, -K/--keep-dirlinks + and --munge-links. Destination paths mirror the absolute source path below + the destination root (run_client uses absolute --source-dir/--dest-dir).""" + + def test_copy_dirlinks_dereferences_dir_symlink(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "symlink_trust_copy_dirlinks") + dest = os.path.join(TEST_DATA_DIR, "symlink_trust_copy_dirlinks_dst") + clean_dir(source) + clean_dir(dest) + os.makedirs(os.path.join(source, "realdir")) + with open(os.path.join(source, "realfile.txt"), "wb") as f: + f.write(b"real file\n") + with open(os.path.join(source, "realdir", "inside.txt"), "wb") as f: + f.write(b"inside dir\n") + os.symlink("realfile.txt", os.path.join(source, "link_file")) + os.symlink("realdir", os.path.join(source, "link_dir")) + + result, _ = run_client(source, dest, flags=["-k"], port=shared_server.port) + assert result.returncode == 0, f"-k failed: {(result.stderr or result.stdout)[:300]}" + + received = get_dest_received_dir(dest, source) + # link -> realdir dereferences into a real directory tree... + link_dir = os.path.join(received, "link_dir") + assert os.path.isdir(link_dir) + assert not os.path.islink(link_dir) + assert os.path.isfile(os.path.join(link_dir, "inside.txt")) + # ... while a symlink to a regular file stays a symlink. + link_file = os.path.join(received, "link_file") + assert os.path.islink(link_file) + assert os.readlink(link_file) == "realfile.txt" + + def test_keep_dirlinks_keeps_dest_symlink_to_dir(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "symlink_trust_keep_dirlinks") + dest = os.path.join(TEST_DATA_DIR, "symlink_trust_keep_dirlinks_dst") + clean_dir(source) + clean_dir(dest) + os.makedirs(os.path.join(source, "sub")) + with open(os.path.join(source, "sub", "file.txt"), "wb") as f: + f.write(b"under the kept symlinked dir\n") + + # Plant the destination's symlink-to-directory at the exact mirror path: + # sub -> realdir (relative, both siblings under the mirror parent). + parent = os.path.join(dest, os.path.abspath(source).lstrip(os.sep)) + os.makedirs(parent) + os.makedirs(os.path.join(parent, "realdir")) + os.symlink("realdir", os.path.join(parent, "sub")) + + result, _ = run_client(source, dest, flags=["-K"], port=shared_server.port) + assert result.returncode == 0, f"-K failed: {(result.stderr or result.stdout)[:300]}" + + received = get_dest_received_dir(dest, source) + sub = os.path.join(received, "sub") + # sub stays a symlink to the directory rather than being replaced... + assert os.path.islink(sub) + assert os.readlink(sub) == "realdir" + # ... and the file is written beneath it, through to the referent dir. + assert os.path.isfile(os.path.join(parent, "realdir", "file.txt")) + + def test_munge_links_unmunged_target_and_containment(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "symlink_trust_munge") + dest = os.path.join(TEST_DATA_DIR, "symlink_trust_munge_dst") + clean_dir(source) + clean_dir(dest) + with open(os.path.join(source, "a.txt"), "wb") as f: + f.write(b"a\n") + os.symlink("a.txt", os.path.join(source, "good")) + os.symlink("/etc/passwd", os.path.join(source, "abs_escape")) + os.symlink("../../escape", os.path.join(source, "dotdot_escape")) + + result, _ = run_client(source, dest, flags=["-l", "--munge-links"], + port=shared_server.port) + assert result.returncode == 0, f"--munge-links failed: {(result.stderr or result.stdout)[:300]}" + + received = get_dest_received_dir(dest, source) + # The safe symlink is created with its correct (unmunged) target. + good = os.path.join(received, "good") + assert os.path.islink(good) + assert os.readlink(good) == "a.txt" + # A target that would escape the receive root is contained (skip: never + # transmitted, so nothing is created at the destination). + assert not os.path.lexists(os.path.join(received, "abs_escape")) + assert not os.path.lexists(os.path.join(received, "dotdot_escape")) + assert os.path.isfile(os.path.join(received, "a.txt")) + + def test_links_copies_symlinks_as_symlinks(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "symlink_trust_links") + dest = os.path.join(TEST_DATA_DIR, "symlink_trust_links_dst") + clean_dir(source) + clean_dir(dest) + os.makedirs(os.path.join(source, "realdir")) + with open(os.path.join(source, "realfile.txt"), "wb") as f: + f.write(b"real\n") + with open(os.path.join(source, "realdir", "x.txt"), "wb") as f: + f.write(b"x\n") + os.symlink("realfile.txt", os.path.join(source, "lf")) + os.symlink("realdir", os.path.join(source, "ld")) + + result, _ = run_client(source, dest, flags=["-l"], port=shared_server.port) + assert result.returncode == 0, f"-l failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert os.path.islink(os.path.join(received, "lf")) + assert os.readlink(os.path.join(received, "lf")) == "realfile.txt" + assert os.path.islink(os.path.join(received, "ld")) + assert os.readlink(os.path.join(received, "ld")) == "realdir" + + def test_receiver_contains_absolute_target_even_without_munge(self, shared_server): + # The trust boundary is symmetric and enforced receiver-side: a plain -l + # (no --munge-links) run must refuse to materialize an out-of-root + # absolute symlink target, while still copying a legitimate in-root one. + source = os.path.join(TEST_DATA_DIR, "symlink_trust_abs") + dest = os.path.join(TEST_DATA_DIR, "symlink_trust_abs_dst") + clean_dir(source) + clean_dir(dest) + with open(os.path.join(source, "a.txt"), "wb") as f: + f.write(b"a\n") + os.symlink("a.txt", os.path.join(source, "good")) + os.symlink("/etc/passwd", os.path.join(source, "unsafe_abs")) + + result, _ = run_client(source, dest, flags=["-l"], port=shared_server.port) + assert result.returncode == 0, f"-l failed: {(result.stderr or result.stdout)[:300]}" + + received = get_dest_received_dir(dest, source) + good = os.path.join(received, "good") + assert os.path.islink(good) + assert os.readlink(good) == "a.txt" + # The absolute (non-contained) target was not materialized at the dest. + assert not os.path.lexists(os.path.join(received, "unsafe_abs")) + + def test_links_does_not_strip_munge_prefix_without_munge(self, shared_server): + # A source symlink whose target genuinely begins with the #SYMLINK/ marker + # must round-trip verbatim under plain -l: the receiver only unmunges when + # the negotiated --munge-links policy is on, never unconditionally. + source = os.path.join(TEST_DATA_DIR, "symlink_trust_prefix") + dest = os.path.join(TEST_DATA_DIR, "symlink_trust_prefix_dst") + clean_dir(source) + clean_dir(dest) + with open(os.path.join(source, "realfile.txt"), "wb") as f: + f.write(b"real\n") + os.symlink("#SYMLINK/realfile.txt", os.path.join(source, "prefixed")) + + result, _ = run_client(source, dest, flags=["-l"], port=shared_server.port) + assert result.returncode == 0, f"-l failed: {(result.stderr or result.stdout)[:300]}" + + received = get_dest_received_dir(dest, source) + prefixed = os.path.join(received, "prefixed") + assert os.path.islink(prefixed) + assert os.readlink(prefixed) == "#SYMLINK/realfile.txt" +def _xattr_supported(path): + """True when the filesystem hosting `path` supports user xattrs.""" + try: + os.setxattr(path, "user.fastsync-probe", b"p") + os.removexattr(path, "user.fastsync-probe") + return True + except (OSError, AttributeError): + return False + + +class TestExtendedAttributes: + """-X/--xattrs, -A/--acls, --fake-super: portable extended metadata. + + Runs unprivileged (CI is non-root). Everything is best-effort and guarded: + a filesystem without xattr support, or an ACL toolchain/POSIX-ACL + filesystem feature that is missing, is skipped rather than failed. The + security boundary (only user.* and the system.posix_acl_* namespaces are + ever applied) is asserted alongside the happy path.""" + + def _source_and_dest(self, name): + source = os.path.join(TEST_DATA_DIR, name + "_src") + dest = os.path.join(TEST_DATA_DIR, name + "_dst") + clean_dir(source) + clean_dir(dest) + return source, dest + + @pytest.mark.ci + def test_xattrs_preserves_user_namespace(self, shared_server): + source, dest = self._source_and_dest("xattr") + f = os.path.join(source, "data.txt") + with open(f, "wb") as fh: + fh.write(b"xattr payload\n") + if not _xattr_supported(f): + pytest.skip("filesystem does not support user xattrs") + os.setxattr(f, "user.foo", b"preserved-value") + + result, _ = run_client(source, dest, flags=["-X"], port=shared_server.port) + assert result.returncode == 0, \ + f"-X sync failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert os.getxattr(os.path.join(received, "data.txt"), "user.foo") == b"preserved-value" + + def test_without_xattrs_does_not_carry(self, shared_server): + source, dest = self._source_and_dest("xattr_ctrl") + f = os.path.join(source, "data.txt") + with open(f, "wb") as fh: + fh.write(b"plain\n") + if not _xattr_supported(f): + pytest.skip("filesystem does not support user xattrs") + os.setxattr(f, "user.foo", b"must-not-travel") + + result, _ = run_client(source, dest, port=shared_server.port) + assert result.returncode == 0, \ + f"control sync failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + with pytest.raises(OSError): + os.getxattr(os.path.join(received, "data.txt"), "user.foo") + + def test_reserved_fake_super_key_not_forwarded(self, shared_server): + """A source file that already carries the reserved user.fastsync.stat + record must NOT have it planted on the receiver during a plain -X run + (it is receiver-only, so it cannot be spoofed for a later privileged + restore).""" + source, dest = self._source_and_dest("xattr_reserved") + f = os.path.join(source, "data.txt") + with open(f, "wb") as fh: + fh.write(b"reserved\n") + if not _xattr_supported(f): + pytest.skip("filesystem does not support user xattrs") + os.setxattr(f, "user.fastsync.stat", b"0:0:644:0:0") + # A normal user.* attr still travels alongside. + os.setxattr(f, "user.keep", b"yes") + + result, _ = run_client(source, dest, flags=["-X"], port=shared_server.port) + assert result.returncode == 0, \ + f"-X reserved-key sync failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert os.getxattr(os.path.join(received, "data.txt"), "user.keep") == b"yes" + with pytest.raises(OSError): + os.getxattr(os.path.join(received, "data.txt"), "user.fastsync.stat") + + @pytest.mark.ci + def test_xattrs_multithreaded(self, shared_server): + source, dest = self._source_and_dest("xattr_mt") + f = os.path.join(source, "data.txt") + with open(f, "wb") as fh: + fh.write(b"mt xattr\n") + if not _xattr_supported(f): + pytest.skip("filesystem does not support user xattrs") + os.setxattr(f, "user.k", b"v") + result, _ = run_client(source, dest, flags=["-X", "--threads"], port=shared_server.port) + assert result.returncode == 0, \ + f"-X -m sync failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert os.getxattr(os.path.join(received, "data.txt"), "user.k") == b"v" + + @pytest.mark.ci + def test_acls_via_posix_acl_xattr(self, shared_server): + source, dest = self._source_and_dest("acl") + f = os.path.join(source, "data.txt") + with open(f, "wb") as fh: + fh.write(b"acl payload\n") + if not _xattr_supported(f): + pytest.skip("filesystem does not support xattrs") + acl_blob = None + if shutil.which("setfacl") is not None: + acl = subprocess.run(["setfacl", "-m", "o::r", f], capture_output=True, text=True) + if acl.returncode == 0: + try: + acl_blob = os.getxattr(f, "system.posix_acl_access") + except OSError: + acl_blob = None + if acl_blob is None: + # No setfacl (common in the minimal CI image): synthesize a valid + # non-trivial POSIX ACL ("u:current-uid:r") xattr blob directly. + import struct + try: + uid_for_acl = os.geteuid() if os.geteuid() != 0 else 65534 + struct_entry = struct.pack(" 2: + pytest.skip("filesystem does not preserve directory mtimes") + return source, dest, ("", "sub", os.path.join("sub", "deep")) + + def _link_tree(self, name): + source = os.path.join(TEST_DATA_DIR, name + "_src") + dest = os.path.join(TEST_DATA_DIR, name + "_dst") + clean_dir(source) + clean_dir(dest) + os.makedirs(os.path.join(source, "sub"), exist_ok=True) + with open(os.path.join(source, "sub", "file.txt"), "wb") as fh: + fh.write(b"target\n") + link = os.path.join(source, "sub", "link") + # A same-directory relative target (no ".."): FastSync refuses an + # escaping/ambiguous symlink target, and ".." is a deliberate divergence. + os.symlink("file.txt", link) + os.utime(link, (DISTINCT_MTIME, DISTINCT_MTIME), follow_symlinks=False) + if abs(os.lstat(link).st_mtime - DISTINCT_MTIME) > 2: + pytest.skip("filesystem does not preserve symlink mtimes") + return source, dest, os.path.join("sub", "link") + + def _run(self, source, dest, flags, shared_server): + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, \ + f"{flags} failed: {(result.stderr or result.stdout)[:400]}" + return get_dest_received_dir(dest, source) + + @pytest.mark.ci + @pytest.mark.parametrize("mt", [False, True]) + def test_directory_mtime_round_trip(self, shared_server, mt): + source, dest, rels = self._tree("dirtime") + flags = ["-a"] + (["--threads"] if mt else []) + received = self._run(source, dest, flags, shared_server) + for rel in rels: + src_m = os.stat(os.path.join(source, rel)).st_mtime + dst_m = os.stat(os.path.join(received, rel)).st_mtime + assert abs(dst_m - src_m) < 2, \ + f"dir '{rel}': source={src_m} dest={dst_m} (flags={flags})" + + @pytest.mark.ci + @pytest.mark.parametrize("mt", [False, True]) + def test_omit_dir_times_suppresses_only_dirs(self, shared_server, mt): + source, dest, rels = self._tree("omitdir") + flags = ["-a", "-O"] + (["--threads"] if mt else []) + received = self._run(source, dest, flags, shared_server) + for rel in rels: + dst_m = os.stat(os.path.join(received, rel)).st_mtime + assert abs(dst_m - DISTINCT_MTIME) > 5, \ + f"-O must not apply directory times ('{rel}' got {dst_m})" + + @pytest.mark.ci + @pytest.mark.parametrize("mt", [False, True]) + def test_symlink_mtime_round_trip(self, shared_server, mt): + source, dest, rel = self._link_tree("linktime") + flags = ["-a"] + (["--threads"] if mt else []) + received = self._run(source, dest, flags, shared_server) + src_link = os.path.join(source, rel) + dst_link = os.path.join(received, rel) + assert os.path.islink(dst_link), f"{dst_link} is not a symlink" + src_m = os.lstat(src_link).st_mtime + dst_m = os.lstat(dst_link).st_mtime + assert abs(dst_m - src_m) < 2, f"symlink times: source={src_m} dest={dst_m}" + + @pytest.mark.ci + @pytest.mark.parametrize("mt", [False, True]) + def test_omit_link_times_suppresses_only_links(self, shared_server, mt): + source, dest, rel = self._link_tree("omitlink") + flags = ["-a", "-J"] + (["--threads"] if mt else []) + received = self._run(source, dest, flags, shared_server) + dst_link = os.path.join(received, rel) + assert os.path.islink(dst_link), f"{dst_link} is not a symlink" + dst_m = os.lstat(dst_link).st_mtime + assert abs(dst_m - DISTINCT_MTIME) > 5, \ + f"-J must not apply symlink times (got {dst_m})" + + @pytest.mark.ci + def test_omit_flags_are_independent(self, shared_server): + """-O suppresses only directory times and -J only symlink times: with + -O the symlink time is still preserved, and with -J the dir times are.""" + source, dest, rel = self._link_tree("omitindep") + # Add a subdirectory mtime to check alongside the symlink. + sub = os.path.join(source, "sub") + os.utime(sub, (DISTINCT_MTIME, DISTINCT_MTIME)) + + # -O => dir times omitted, symlink time preserved. + clean_dir(dest + "_o") + recv_o = self._run(source, dest + "_o", ["-a", "-O"], shared_server) + assert abs(os.lstat(os.path.join(recv_o, rel)).st_mtime - DISTINCT_MTIME) < 2, \ + "-O must not suppress symlink times" + assert abs(os.stat(os.path.join(recv_o, "sub")).st_mtime - DISTINCT_MTIME) > 5, \ + "-O must suppress directory times" + + # -J => symlink times omitted, dir times preserved. + clean_dir(dest + "_j") + recv_j = self._run(source, dest + "_j", ["-a", "-J"], shared_server) + assert abs(os.lstat(os.path.join(recv_j, rel)).st_mtime - DISTINCT_MTIME) > 5, \ + "-J must suppress symlink times" + assert abs(os.stat(os.path.join(recv_j, "sub")).st_mtime - DISTINCT_MTIME) < 2, \ + "-J must not suppress directory times" + + @pytest.mark.ci + @pytest.mark.parametrize("mt", [False, True]) + def test_preserve_does_not_create_empty_source_dir(self, shared_server, mt): + """P7 Wave D #1: a captured-but-EMPTY source directory is never created + at the destination. The scanner records its time (it is transmitted via + STATUS_DIR_TIMES), but the receiver treats that entry as record-only, so + `-a` keeps the documented "empty dirs are never transferred" behavior.""" + source = os.path.join(TEST_DATA_DIR, f"empty_dir_{'m' if mt else 's'}_src") + dest = os.path.join(TEST_DATA_DIR, f"empty_dir_{'m' if mt else 's'}_dst") + clean_dir(source) + clean_dir(dest) + with open(os.path.join(source, "keep.txt"), "wb") as fh: + fh.write(b"regular file\n") + os.makedirs(os.path.join(source, "empty_sub")) + flags = ["-a"] + (["--threads"] if mt else []) + received = self._run(source, dest, flags, shared_server) + assert os.path.isfile(os.path.join(received, "keep.txt")), "regular file missing" + assert not os.path.lexists(os.path.join(received, "empty_sub")), \ + f"-a created an empty source directory at {received}/empty_sub" + + @pytest.mark.ci + @pytest.mark.parametrize("mt", [False, True]) + def test_prune_empty_dirs_still_does_not_create_empty_dir(self, shared_server, mt): + """P7 Wave D #1: `-a -m` (--prune-empty-dirs) keeps its semantics -- a + captured empty directory is never created even though its time is + recorded.""" + source = os.path.join(TEST_DATA_DIR, f"prune_empty_{'m' if mt else 's'}_src") + dest = os.path.join(TEST_DATA_DIR, f"prune_empty_{'m' if mt else 's'}_dst") + clean_dir(source) + clean_dir(dest) + with open(os.path.join(source, "keep.txt"), "wb") as fh: + fh.write(b"regular file\n") + os.makedirs(os.path.join(source, "empty_sub")) + flags = ["-a", "-m"] + (["--threads"] if mt else []) + received = self._run(source, dest, flags, shared_server) + assert os.path.isfile(os.path.join(received, "keep.txt")), "regular file missing" + assert not os.path.lexists(os.path.join(received, "empty_sub")), \ + f"-a -m created an empty source directory at {received}/empty_sub" + + @pytest.mark.ci + @pytest.mark.parametrize("mt", [False, True]) + def test_collision_at_dir_time_path_does_not_abort(self, shared_server, mt): + """P7 Wave D #1: a pre-existing regular file at a source-empty-dir's + mirror path must not abort the transfer (the old mkdir failed and failed + the run) and must not be clobbered.""" + source = os.path.join(TEST_DATA_DIR, f"dirtime_collide_{'m' if mt else 's'}_src") + dest = os.path.join(TEST_DATA_DIR, f"dirtime_collide_{'m' if mt else 's'}_dst") + clean_dir(source) + clean_dir(dest) + with open(os.path.join(source, "keep.txt"), "wb") as fh: + fh.write(b"regular file\n") + os.makedirs(os.path.join(source, "collide")) + # Plant a regular file at exactly the mirror path of source/collide. + received = get_dest_received_dir(dest, source) + os.makedirs(received, exist_ok=True) + blocker = os.path.join(received, "collide") + with open(blocker, "wb") as fh: + fh.write(b"pre-existing blocker\n") + flags = ["-a"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, \ + f"-a aborted on a pre-existing file at an empty-dir path: " \ + f"{(result.stderr or result.stdout)[:400]}" + assert os.path.isfile(blocker) and not os.path.islink(blocker), \ + "the pre-existing blocker was replaced by a directory" + with open(blocker, "rb") as fh: + assert fh.read() == b"pre-existing blocker\n", "the blocker file was clobbered" + assert os.path.isfile(os.path.join(received, "keep.txt")), "regular file missing" + + +class TestCopyAs: + """P7 Wave E: --copy-as=USER[:GROUP] safe subset. + + FastSync never switches the receiver's process credentials; the receiver + forces the ownership of every entry it writes to the requested ids through + the confined fd-relative identity path, which REQUIRES a privileged (root) + receiver. An unprivileged receiver refuses the whole transfer up front at + the config handshake, before any file data moves. + """ + + @pytest.mark.ci + def test_unprivileged_receiver_refuses_copy_as(self, shared_server): + """The key assertable behavior: an unprivileged receiver REFUSES a + --copy-as transfer cleanly (non-zero exit, no data written) instead of + silently writing the wrong ownership.""" + source = os.path.join(TEST_DATA_DIR, "copyas_refuse_src") + dest = os.path.join(TEST_DATA_DIR, "copyas_refuse_dst") + clean_dir(source) + clean_dir(dest) + with open(os.path.join(source, "secret.txt"), "wb") as fh: + fh.write(b"must not be written\n") + + captured = None + if os.geteuid() == 0: + if shutil.which("setpriv") is None: + pytest.skip("root runner without setpriv cannot start an unprivileged receiver") + # The unprivileged receiver must execute the server binary out of the + # test workspace, so the workspace path has to be traversable by uid + # 65534. A checkout under a 0700 directory (e.g. /root) is not; skip + # rather than fail — CI runs from a traversable workspace and still + # exercises this behavior. + probe = subprocess.run( + ["setpriv", "--reuid=65534", "--regid=65534", "--clear-groups", + "test", "-x", os.path.abspath(SERVER_CMD[0])], + stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) + if probe.returncode != 0: + pytest.skip("workspace is not traversable by the unprivileged receiver uid") + os.chmod(dest, 0o777) + proc, port = _start_captured_server( + prefix=["setpriv", "--reuid=65534", "--regid=65534", "--clear-groups"]) + captured = proc + else: + # The session server already runs unprivileged. + port = shared_server.port + + try: + result, _ = run_client(source, dest, + flags=["--copy-as=@65534:@65534"], port=port) + finally: + if captured is not None: + out, err = _stop_captured_server(captured) + else: + out, err = "", "" + + assert result.returncode != 0, ( + f"an unprivileged receiver must refuse --copy-as: rc={result.returncode} " + f"out={result.stdout[:200]!r} err={result.stderr[:200]!r}" + ) + received = get_dest_received_dir(dest, source) + assert not os.path.exists(os.path.join(received, "secret.txt")), ( + "--copy-as refusal leaked file data into the destination" + ) + if captured is not None: + assert "copy-as requires a privileged receiver" in (out + err), ( + f"refusal reason was not logged: out={out!r} err={err!r}" + ) + + @pytest.mark.ci + @pytest.mark.skipif(os.geteuid() != 0, reason="requires a root receiver to chown") + def test_root_copy_as_chowns_transferred_file(self, shared_server): + """Root-gated: --copy-as=USER:GROUP forces the transferred file's + ownership to exactly that uid/gid (numeric form for determinism).""" + source = os.path.join(TEST_DATA_DIR, "copyas_root_src") + dest = os.path.join(TEST_DATA_DIR, "copyas_root_dst") + clean_dir(source) + clean_dir(dest) + with open(os.path.join(source, "owned.txt"), "wb") as fh: + fh.write(b"owned by nobody\n") + + result, _ = run_client(source, dest, + flags=["--copy-as=@65534:@65534"], port=shared_server.port) + assert result.returncode == 0, ( + f"--copy-as root transfer failed: {(result.stderr or result.stdout)[:400]}" + ) + received = get_dest_received_dir(dest, source) + target = os.path.join(received, "owned.txt") + assert os.path.isfile(target), f"transferred file missing at {target}" + st = os.lstat(target) + assert (st.st_uid, st.st_gid) == (65534, 65534), ( + f"--copy-as did not force ownership: uid={st.st_uid} gid={st.st_gid}" + ) + + @pytest.mark.ci + @pytest.mark.skipif(os.geteuid() != 0, reason="requires a root receiver to chown") + def test_root_copy_as_owns_directory(self, shared_server): + """--copy-as must own an explicitly-created directory entry, not just the + files inside it. A listed directory (--files-from + --dirs -R) is sent + as a STATUS_MKDIR entry, exercising the directory ownership path.""" + source = os.path.join(TEST_DATA_DIR, "copyas_dir_src") + dest = os.path.join(TEST_DATA_DIR, "copyas_dir_dst") + clean_dir(source) + clean_dir(dest) + os.makedirs(os.path.join(source, "owned_dir"), exist_ok=True) + lst = os.path.join(TEST_DATA_DIR, "copyas_dir_list.txt") + with open(lst, "wb") as fh: + fh.write(b"owned_dir\n") + + result, _ = run_client( + source, dest, + flags=["--copy-as=@65534:@65534", "--files-from", lst, "--dirs", "-R"], + port=shared_server.port) + assert result.returncode == 0, ( + f"--copy-as directory transfer failed: {(result.stderr or result.stdout)[:400]}" + ) + target = os.path.join(dest, "owned_dir") + assert os.path.isdir(target), f"explicit directory missing at {target}" + st = os.stat(target) + assert (st.st_uid, st.st_gid) == (65534, 65534), ( + f"--copy-as did not own the directory: uid={st.st_uid} gid={st.st_gid}" + ) + + @pytest.mark.ci + @pytest.mark.skipif(os.geteuid() != 0, reason="requires a root receiver to chown") + def test_root_copy_as_owns_implicit_parent_dirs(self, shared_server): + """--copy-as must also own the intermediate directories that the receiver + creates implicitly while writing a nested file (the scanner does not emit + STATUS_MKDIR entries for ordinary traversal directories), not just the + file itself.""" + source = os.path.join(TEST_DATA_DIR, "copyas_nested_src") + dest = os.path.join(TEST_DATA_DIR, "copyas_nested_dst") + clean_dir(source) + clean_dir(dest) + nested = os.path.join(source, "top", "mid", "leaf") + os.makedirs(nested, exist_ok=True) + with open(os.path.join(nested, "deep.txt"), "wb") as fh: + fh.write(b"nested copy-as ownership\n") + + result, _ = run_client(source, dest, + flags=["--copy-as=@65534:@65534"], + port=shared_server.port) + assert result.returncode == 0, ( + f"--copy-as nested transfer failed: {(result.stderr or result.stdout)[:400]}" + ) + received = get_dest_received_dir(dest, source) + for rel in ("top", os.path.join("top", "mid"), os.path.join("top", "mid", "leaf")): + target = os.path.join(received, rel) + assert os.path.isdir(target), f"implicit directory missing at {target}" + st = os.stat(target) + assert (st.st_uid, st.st_gid) == (65534, 65534), ( + f"--copy-as did not own implicit directory {rel}: " + f"uid={st.st_uid} gid={st.st_gid}" + ) + + @pytest.mark.ci + @pytest.mark.skipif(os.geteuid() != 0, reason="requires a root receiver to chown") + def test_root_copy_as_owns_fifo(self, shared_server): + """--copy-as must own a recreated FIFO special node.""" + source = os.path.join(TEST_DATA_DIR, "copyas_fifo_src") + dest = os.path.join(TEST_DATA_DIR, "copyas_fifo_dst") + clean_dir(source) + clean_dir(dest) + os.mkfifo(os.path.join(source, "pipe.fifo")) + + result, _ = run_client(source, dest, + flags=["--copy-as=@65534:@65534", "--specials"], + port=shared_server.port) + assert result.returncode == 0, ( + f"--copy-as FIFO transfer failed: {(result.stderr or result.stdout)[:400]}" + ) + received = get_dest_received_dir(dest, source) + target = os.path.join(received, "pipe.fifo") + assert stat.S_ISFIFO(os.lstat(target).st_mode), f"FIFO missing at {target}" + st = os.lstat(target) + assert (st.st_uid, st.st_gid) == (65534, 65534), ( + f"--copy-as did not own the FIFO: uid={st.st_uid} gid={st.st_gid}" + ) + + @pytest.mark.ci + @pytest.mark.skipif(os.geteuid() != 0, reason="requires a root receiver to chown") + def test_root_copy_as_with_fake_super_keeps_target_owner(self, shared_server): + """--fake-super must not let the recorded source owner override the + --copy-as forced owner (copy-as is authoritative).""" + source = os.path.join(TEST_DATA_DIR, "copyas_fakesuper_src") + dest = os.path.join(TEST_DATA_DIR, "copyas_fakesuper_dst") + clean_dir(source) + clean_dir(dest) + src_file = os.path.join(source, "mixed.txt") + with open(src_file, "wb") as fh: + fh.write(b"copy-as wins over fake-super\n") + os.chown(src_file, 12345, 12346) + + result, _ = run_client(source, dest, + flags=["--copy-as=@65534:@65534", "--fake-super"], + port=shared_server.port) + assert result.returncode == 0, ( + f"--copy-as --fake-super transfer failed: " + f"{(result.stderr or result.stdout)[:400]}" + ) + received = get_dest_received_dir(dest, source) + st = os.lstat(os.path.join(received, "mixed.txt")) + assert (st.st_uid, st.st_gid) == (65534, 65534), ( + f"--fake-super overrode --copy-as: uid={st.st_uid} gid={st.st_gid}" + ) diff --git a/tests/integration/test_iconv.py b/tests/integration/test_iconv.py new file mode 100644 index 0000000..4a69aa1 --- /dev/null +++ b/tests/integration/test_iconv.py @@ -0,0 +1,250 @@ +"""--iconv=CONVERT_SPEC file-NAME charset conversion integration tests. + +The client converts every source file name from LOCAL to REMOTE before it goes +on the wire, and the receiver converts it back from REMOTE to LOCAL, so a +source tree using one charset can be written into a destination tree using +another (rsync compatibility; content bytes are never touched). +""" +import os +import shutil + +import pytest + +from common import TEST_DATA_DIR, run_client, clean_dir, ServerManager + +LATIN1_NAME = b"caf\xe9.txt" +UTF8_NAME = "caf\u00e9.txt".encode("utf-8") + + +def _make(tag): + source = os.path.join(TEST_DATA_DIR, f"iconv_{tag}_src") + dest = os.path.join(TEST_DATA_DIR, f"iconv_{tag}_dst") + clean_dir(source) + shutil.rmtree(dest, ignore_errors=True) + # The destination ROOT must pre-exist on the receiver (the --mkpath contract: + # without --mkpath the server requires the root directory to exist). + os.makedirs(dest, exist_ok=True) + return source, dest + + +def _place_bytes(root, name_bytes, data=b"latin1 payload\n"): + full = os.path.join(os.fsencode(root), name_bytes) + os.makedirs(os.path.dirname(full), exist_ok=True) + with open(full, "wb") as fh: + fh.write(data) + return full + + +def _dest_file(source, dest, name): + base = os.path.join(dest, os.path.abspath(source).lstrip(os.sep)) + return os.path.join(os.fsencode(base), name) + + +@pytest.mark.ci +def test_iconv_latin1_roundtrip(shared_server): + """A source file whose name is ISO-8859-1 bytes is transferred with + --iconv=iso-8859-1,utf-8 and lands on the destination with the ORIGINAL + latin1 name (the wire carried it as UTF-8).""" + source, dest = _make("latin1") + _place_bytes(source, LATIN1_NAME) + + result, _ = run_client( + source, dest, flags=["--iconv=iso-8859-1,utf-8"], port=shared_server.port + ) + assert result.returncode == 0, (result.stderr or result.stdout)[:400] + + dst = _dest_file(source, dest, LATIN1_NAME) + assert os.path.exists(dst), f"dest latin1-named file not found under {dest}" + + +@pytest.mark.ci +def test_iconv_to_utf8_on_wire(shared_server): + """--iconv=utf-8 (single, identity both ways) on an ascii filename transfers + cleanly with no error.""" + source, dest = _make("utf8") + src_path = os.path.join(source, "plain.txt") + with open(src_path, "wb") as fh: + fh.write(b"identity\n") + + result, _ = run_client(source, dest, flags=["--iconv=utf-8"], port=shared_server.port) + assert result.returncode == 0, (result.stderr or result.stdout)[:400] + + dst = _dest_file(source, dest, os.fsencode("plain.txt")) + assert os.path.exists(dst) + + +@pytest.mark.ci +def test_iconv_passthrough_identity(shared_server): + """No --iconv flag: the transfer is unchanged (regression guard -- the common + path must not go through iconv at all).""" + source, dest = _make("identity") + for name, data in (("a.txt", b"aaa\n"), ("sub/b.txt", b"bbb\n")): + p = os.path.join(source, name) + os.makedirs(os.path.dirname(p), exist_ok=True) + with open(p, "wb") as fh: + fh.write(data) + + result, _ = run_client(source, dest, port=shared_server.port) + assert result.returncode == 0, (result.stderr or result.stdout)[:400] + + for name in ("a.txt", "sub/b.txt"): + assert os.path.exists(_dest_file(source, dest, os.fsencode(name))) + + +@pytest.mark.ci +def test_iconv_receiver_own_charset(shared_server): + """A dedicated server started with its OWN --iconv converts received names + to ITS charset: the source holds a latin1-named file, the wire carries it + as UTF-8 (from the client's spec), and the receiver re-decodes it to UTF-8 + on disk. This discriminates a real wire conversion from a no-op passthrough + (a latin1 byte sequence is not valid UTF-8, so the receiver decoding it as + UTF-8 would fail the transfer).""" + with ServerManager() as server: + server.start(extra_args=["--iconv=utf-8"]) + source, dest = _make("recv_charset") + _place_bytes(source, LATIN1_NAME) + + result, _ = run_client( + source, dest, flags=["--iconv=iso-8859-1,utf-8"], port=server.port + ) + assert result.returncode == 0, (result.stderr or result.stdout)[:400] + + dst = _dest_file(source, dest, UTF8_NAME) + assert os.path.exists(dst), f"dest UTF-8-named file not found under {dest}" + + +@pytest.mark.ci +def test_iconv_invalid_charset_rejected(shared_server): + """An unsupported charset name is rejected at startup with a nonzero exit.""" + source, dest = _make("badcharset") + src_path = os.path.join(source, "f.txt") + with open(src_path, "wb") as fh: + fh.write(b"x") + + result, _ = run_client( + source, dest, flags=["--iconv=no-such-charset,utf-8"], port=shared_server.port + ) + assert result.returncode != 0 + + +@pytest.mark.ci +def test_iconv_garbage_spec_rejected(shared_server): + """A malformed CONVERT_SPEC is rejected at startup with a nonzero exit.""" + source, dest = _make("garbage") + src_path = os.path.join(source, "f.txt") + with open(src_path, "wb") as fh: + fh.write(b"x") + + result, _ = run_client(source, dest, flags=["--iconv=,,,"], port=shared_server.port) + assert result.returncode != 0 + + +@pytest.mark.ci +def test_iconv_expanding_name_growth(shared_server): + """A long latin1 name whose UTF-8 encoding expands past the initial output + buffer exercises the E2BIG growth path in charset_convert (each high-bit + latin1 byte doubles in UTF-8), and must land unchanged on the destination.""" + source, dest = _make("growth") + name_bytes = b"a" * 40 + bytes(range(0x80, 0x80 + 40)) + b".txt" + _place_bytes(source, name_bytes, data=b"growth\n") + + result, _ = run_client( + source, dest, flags=["--iconv=iso-8859-1,utf-8"], port=shared_server.port + ) + assert result.returncode == 0, (result.stderr or result.stdout)[:400] + + assert os.path.exists(_dest_file(source, dest, name_bytes)) + + +def test_iconv_symlink_path_and_target(shared_server): + """A latin1-named symlink pointing at a latin1-named target survives the + transfer: both the link name and the link target are wire-converted and + re-decoded on the destination (-l preserves links).""" + source, dest = _make("symlink") + target = b"target\xe9.dat" + _place_bytes(source, target, data=b"t\n") + os.symlink(target, os.path.join(os.fsencode(source), b"link\xe9")) + + result, _ = run_client( + source, dest, flags=["--iconv=iso-8859-1,utf-8", "--links"], port=shared_server.port + ) + assert result.returncode == 0, (result.stderr or result.stdout)[:400] + + dst_target = _dest_file(source, dest, target) + dst_link = _dest_file(source, dest, b"link\xe9") + assert os.path.exists(dst_target), "dest latin1 target file missing" + assert os.path.islink(dst_link), "dest latin1 symlink missing" + assert os.readlink(dst_link) == target, "symlink target not preserved/decoded" + with open(dst_link, "rb") as fh: + assert fh.read() == b"t\n" + + +def test_iconv_hardlink_path_and_target(shared_server): + """A latin1-named hard-linked pair is preserved: -H transmits later group + members as a path+target link to the first member, so both the member name + and the target wire-convert (the two destination names must stay one + inode).""" + source, dest = _make("hardlink") + a = b"hl_a\xe9.txt" + b = b"hl_b\xe9.txt" + src_a = os.path.join(os.fsencode(source), a) + with open(src_a, "wb") as fh: + fh.write(b"shared\n") + os.link(src_a, os.path.join(os.fsencode(source), b)) + + result, _ = run_client( + source, dest, flags=["--iconv=iso-8859-1,utf-8", "--hard-links"], + port=shared_server.port, + ) + assert result.returncode == 0, (result.stderr or result.stdout)[:400] + + dst_a = _dest_file(source, dest, a) + dst_b = _dest_file(source, dest, b) + assert os.path.exists(dst_a) and os.path.exists(dst_b) + assert os.stat(dst_a).st_ino == os.stat(dst_b).st_ino, \ + "hard-link relationship not preserved across the transfer" + + +def test_iconv_delete_manifest_consistent(shared_server): + """Combining --iconv with --delete: the delete manifest's keep-set paths are + wire-converted on send and disk-converted on receive, so the receiver's + delete walker compares like with like and removes exactly the missing + latin1-named file (never a wrong-named mirror).""" + source, dest = _make("delete") + keep = b"keep\xe9.txt" + gone = b"gone\xe9.txt" + _place_bytes(source, keep, data=b"k\n") + _place_bytes(source, gone, data=b"g\n") + + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + flags = ["--iconv=iso-8859-1,utf-8"] + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, (result.stderr or result.stdout)[:400] + assert os.path.exists(_dest_file(source, dest, keep)) + assert os.path.exists(_dest_file(source, dest, gone)) + + os.remove(os.path.join(os.fsencode(source), gone)) + result, _ = run_client( + source, dest, flags=flags + ["--delete"], port=server.port + ) + assert result.returncode == 0, (result.stderr or result.stdout)[:400] + assert os.path.exists(_dest_file(source, dest, keep)), "kept file deleted" + assert not os.path.exists(_dest_file(source, dest, gone)), \ + "missing file was not deleted" + + +def test_iconv_chunk_serialization_blob(shared_server): + """-s (chunk serialization) embeds paths and symlink targets inside the + serialized chunk blob rather than as separate frames; a latin1 name must + still wire-convert and re-decoded on the destination.""" + source, dest = _make("chunk") + name = b"\xe9\xe9\xe9\xe9\xe9\xe9\xe9\xe9\xe9\xe9\xe9\xe9\xe9\xe9\xe9\xe9.txt" + _place_bytes(source, name, data=b"blob\n") + + result, _ = run_client( + source, dest, flags=["--iconv=iso-8859-1,utf-8", "--chunk-serialization"], port=shared_server.port + ) + assert result.returncode == 0, (result.stderr or result.stdout)[:400] + + assert os.path.exists(_dest_file(source, dest, name)) \ No newline at end of file diff --git a/tests/integration/test_preflight.py b/tests/integration/test_preflight.py index 6acf868..cbc1449 100644 --- a/tests/integration/test_preflight.py +++ b/tests/integration/test_preflight.py @@ -2,10 +2,11 @@ import subprocess import sys import os +import shutil import pytest sys.path.insert(0, os.path.dirname(__file__)) -from common import BUILD_DIR, CLIENT_CMD, SERVER_CMD +from common import BUILD_DIR, CLIENT_CMD, SERVER_CMD, TEST_DATA_DIR, run_client, verify_transfer class TestHelp: @@ -79,3 +80,58 @@ class TestServerPort: finally: proc.terminate() proc.wait(timeout=5) + + +def _seed_protocol_source(source): + os.makedirs(source, exist_ok=True) + with open(os.path.join(source, "p.txt"), "w") as fh: + fh.write("protocol test\n") + os.makedirs(os.path.join(source, "nested"), exist_ok=True) + with open(os.path.join(source, "nested", "deep.txt"), "w") as fh: + fh.write("deep file\n") + + +class TestProtocol: + @pytest.mark.ci + def test_protocol_current_version_accepted(self, shared_server): + """--protocol=2.19.0 (the current PROTOCOL_VERSION) is accepted and the + transfer completes normally.""" + source = os.path.join(TEST_DATA_DIR, "proto_ok_src") + dest = os.path.join(TEST_DATA_DIR, "proto_ok_dst") + shutil.rmtree(dest, ignore_errors=True) + os.makedirs(dest) + _seed_protocol_source(source) + result, _ = run_client(source, dest, flags=["--protocol=2.19.0"], + port=shared_server.port) + assert result.returncode == 0, \ + f"--protocol current run failed: {(result.stderr or result.stdout)[:400]}" + received = os.path.join(dest, os.path.abspath(source).lstrip(os.sep)) + mismatches, missing = verify_transfer(source, received) + assert not mismatches and not missing, \ + f"transfer mismatch: missing={missing} mismatches={mismatches}" + + @pytest.mark.ci + def test_protocol_rejects_other_versions(self, shared_server): + """Other versions are rejected up front, before connecting.""" + source = os.path.join(TEST_DATA_DIR, "proto_reject_src") + dest = os.path.join(TEST_DATA_DIR, "proto_reject_dst") + shutil.rmtree(dest, ignore_errors=True) + os.makedirs(dest) + _seed_protocol_source(source) + for bad in ("2.18.0", "2.17.0", "2.15.0", "2.16.0", "216", "31"): + result, _ = run_client(source, dest, flags=[f"--protocol={bad}"], + port=shared_server.port) + assert result.returncode != 0, f"--protocol={bad} should be rejected" + + @pytest.mark.ci + def test_protocol_rejects_garbage(self, shared_server): + """Garbage/empty --protocol values are rejected up front.""" + source = os.path.join(TEST_DATA_DIR, "proto_garbage_src") + dest = os.path.join(TEST_DATA_DIR, "proto_garbage_dst") + shutil.rmtree(dest, ignore_errors=True) + os.makedirs(dest) + _seed_protocol_source(source) + for bad in ("abc", ""): + result, _ = run_client(source, dest, flags=[f"--protocol={bad}"], + port=shared_server.port) + assert result.returncode != 0, f"--protocol={bad} should be rejected" diff --git a/tests/integration/test_ssh.py b/tests/integration/test_ssh.py index 82b86ef..714a0a5 100644 --- a/tests/integration/test_ssh.py +++ b/tests/integration/test_ssh.py @@ -4,77 +4,79 @@ import shutil import subprocess import sys import pytest +import shlex +import tempfile +import shutil sys.path.insert(0, os.path.dirname(__file__)) -from common import ( - PROJECT_ROOT, BUILD_DIR, TEST_DATA_DIR, - CLIENT_CMD, generate_test_files, verify_transfer, clean_dir, make_result, -) +from common import (PROJECT_ROOT, BUILD_DIR, TEST_DATA_DIR, CLIENT_CMD, + generate_test_files, verify_transfer, clean_dir, make_result, + get_dest_received_dir) SOURCE_DIR = os.path.join(TEST_DATA_DIR, "ssh_source") DEST_DIR = os.path.join(TEST_DATA_DIR, "ssh_dest") SSH_AVAILABLE = False +SSH_SKIP_REASON = "SSH localhost probe was not run" +SSH_PROBE_DIR = None def _check_ssh(): - global SSH_AVAILABLE + global SSH_AVAILABLE, SSH_SKIP_REASON, SSH_PROBE_DIR + server_path = os.path.join(BUILD_DIR, "server") + if not os.path.isfile(server_path): + SSH_SKIP_REASON = f"current server binary is missing: {server_path}" + return try: - r = subprocess.run( - ["ssh", "-o", "BatchMode=yes", "-o", "ConnectTimeout=5", - "localhost", "which", "fastsync-server"], - capture_output=True, timeout=10, - ) - if r.returncode == 0: + SSH_PROBE_DIR = tempfile.mkdtemp(prefix="fastsync-ssh-probe-") + probe_server = os.path.join(SSH_PROBE_DIR, "fastsync-server") + os.symlink(server_path, probe_server) + command = f"{shlex.quote(probe_server)} --help" + path = subprocess.run(["ssh", "-o", "BatchMode=yes", "-o", "ConnectTimeout=5", + "localhost", "sh", "-c", command], + capture_output=True, timeout=10, text=True) + if path.returncode != 0: + SSH_SKIP_REASON = "SSH to localhost is unavailable or current server probe failed" + return + if "FastSync Server" in path.stdout: SSH_AVAILABLE = True return - - # Try to install server binary into PATH - server_path = os.path.join(BUILD_DIR, "server") - r = subprocess.run( - ["ssh", "-o", "BatchMode=yes", "localhost", 'echo "$PATH"'], - capture_output=True, timeout=10, text=True, - ) - if r.returncode != 0: - return - for d in r.stdout.strip().split(":"): - d = d.strip() - if not d or "wrappers" in d: - continue - test = subprocess.run( - ["ssh", "-o", "BatchMode=yes", "localhost", - f'test -w "{d}" && ln -sf {server_path} "{d}/fastsync-server" && which fastsync-server'], - capture_output=True, timeout=10, - ) - if test.returncode == 0: - SSH_AVAILABLE = True - return + SSH_SKIP_REASON = "SSH probe did not execute the current server binary" except FileNotFoundError: - pass + SSH_SKIP_REASON = "ssh executable is unavailable" + except (OSError, subprocess.TimeoutExpired) as exc: + SSH_SKIP_REASON = f"SSH setup failed: {exc}" + finally: + if SSH_PROBE_DIR: + shutil.rmtree(SSH_PROBE_DIR, ignore_errors=True) + SSH_PROBE_DIR = None + + +_check_ssh() @pytest.fixture(scope="module", autouse=True) def setup_test_data(): - _check_ssh() if SSH_AVAILABLE: generate_test_files(SOURCE_DIR, full=False) clean_dir(DEST_DIR) yield - shutil.rmtree(TEST_DATA_DIR, ignore_errors=True) + shutil.rmtree(SOURCE_DIR, ignore_errors=True) + shutil.rmtree(DEST_DIR, ignore_errors=True) -def _run_ssh_test(name, flags, expected_missing=None): - """Run an SSH test case (no server process needed, client spawns SSH).""" +def _run_ssh_test(name, flags, expected_missing=None, path_args=None): ssh_dest = f"localhost:{DEST_DIR}" clean_dir(DEST_DIR) - cmd = CLIENT_CMD + [SOURCE_DIR, ssh_dest, "--save-to-disk"] + flags + if not path_args: + path_args = ["--fastsync-server-path", os.path.join(BUILD_DIR, "server")] + cmd = CLIENT_CMD + [SOURCE_DIR, ssh_dest, "--save-to-disk"] + path_args + flags start = __import__("time").monotonic() result = subprocess.run(cmd, text=True, capture_output=True) duration = __import__("time").monotonic() - start - if result.returncode != 0: - return make_result(name, False, duration, f"Exit {result.returncode}: {(result.stderr or result.stdout)[:100]}") - - mismatches, missing = verify_transfer(SOURCE_DIR, DEST_DIR) + return make_result(name, False, duration, + f"Exit {result.returncode}: {(result.stderr or result.stdout)[:100]}") + mismatches, missing = verify_transfer(SOURCE_DIR, get_dest_received_dir(DEST_DIR, SOURCE_DIR)) if expected_missing: missing = [m for m in missing if m not in expected_missing] if missing: @@ -84,49 +86,142 @@ def _run_ssh_test(name, flags, expected_missing=None): return make_result(name, True, duration) -@pytest.mark.skipif(not SSH_AVAILABLE, reason="SSH to localhost not available") class TestSSHStandard: + @pytest.fixture(autouse=True) + def require_ssh(self): + if not SSH_AVAILABLE: + pytest.skip(SSH_SKIP_REASON) + def test_standard(self): r = _run_ssh_test("SSH (localhost)", []) assert r["status"] == "Success", r["error"] def test_multithreading(self): - r = _run_ssh_test("SSH Multithreading (-m)", ["-m"]) + r = _run_ssh_test("SSH Multithreading (--threads)", ["--threads"]) assert r["status"] == "Success", r["error"] def test_compression(self): - r = _run_ssh_test("SSH Compression (-c)", ["-c"]) + r = _run_ssh_test("SSH Compression (-z)", ["-z"]) assert r["status"] == "Success", r["error"] def test_chunk_serialization(self): - r = _run_ssh_test("SSH Chunk Serialization (-s)", ["-s"]) + r = _run_ssh_test("SSH Chunk Serialization (--chunk-serialization)", ["--chunk-serialization"]) assert r["status"] == "Success", r["error"] def test_compression_chunk(self): - r = _run_ssh_test("SSH Compression + Chunk (-c -s)", ["-c", "-s"]) + r = _run_ssh_test("SSH Compression + Chunk (-z --chunk-serialization)", ["-z", "--chunk-serialization"]) assert r["status"] == "Success", r["error"] def test_multithread_compression(self): - r = _run_ssh_test("SSH Multithread + Compression (-m -c)", ["-m", "-c"]) + r = _run_ssh_test("SSH Multithread + Compression (--threads -z)", ["--threads", "-z"]) assert r["status"] == "Success", r["error"] def test_multithread_chunk(self): - r = _run_ssh_test("SSH Multithread + Chunk (-m -s)", ["-m", "-s"]) + r = _run_ssh_test("SSH Multithread + Chunk (--threads --chunk-serialization)", ["--threads", "--chunk-serialization"]) assert r["status"] == "Success", r["error"] def test_all_flags(self): - r = _run_ssh_test("SSH All Flags (-m -c -s)", ["-m", "-c", "-s"]) + r = _run_ssh_test("SSH All Flags (--threads -z --chunk-serialization)", ["--threads", "-z", "--chunk-serialization"]) assert r["status"] == "Success", r["error"] -@pytest.mark.skipif(not SSH_AVAILABLE, reason="SSH to localhost not available") class TestSSHFeatures: + @pytest.fixture(autouse=True) + def require_ssh(self): + if not SSH_AVAILABLE: + pytest.skip(SSH_SKIP_REASON) + def test_archive(self): r = _run_ssh_test("SSH Archive (-a)", ["-a"]) assert r["status"] == "Success", r["error"] def test_exclude(self): r = _run_ssh_test("SSH Exclude (--exclude small.txt)", - ["--exclude", "small.txt"], - expected_missing=["small.txt"]) + ["--exclude", "small.txt"], expected_missing=["small.txt"]) assert r["status"] == "Success", r["error"] + + def test_preallocate(self): + r = _run_ssh_test("SSH Preallocate (--preallocate)", ["--preallocate"]) + assert r["status"] == "Success", r["error"] + +class TestSSHConnectivity: + """Phase 5 connectivity options: -e/--rsh, --rsync-path, --blocking-io, + --outbuf. These are client-side launch concerns, so each must parse and + still drive a real SSH transfer to completion.""" + + @pytest.fixture(autouse=True) + def require_ssh(self): + if not SSH_AVAILABLE: + pytest.skip(SSH_SKIP_REASON) + + def test_rsh_short_form_selects_ssh(self): + r = _run_ssh_test("SSH -e ssh", ["-e", "ssh"]) + assert r["status"] == "Success", r["error"] + + def test_rsh_long_form_selects_ssh(self): + r = _run_ssh_test("SSH --rsh=ssh", ["--rsh=ssh"]) + assert r["status"] == "Success", r["error"] + + def test_rsync_path_aliases_server_path(self): + r = _run_ssh_test("SSH --rsync-path", + [], + path_args=["--rsync-path", os.path.join(BUILD_DIR, "server")]) + assert r["status"] == "Success", r["error"] + + def test_blocking_io(self): + r = _run_ssh_test("SSH --blocking-io", ["--blocking-io"]) + assert r["status"] == "Success", r["error"] + + @pytest.mark.parametrize("mode", ["N", "L", "B"]) + def test_outbuf_mode(self, mode): + r = _run_ssh_test(f"SSH --outbuf={mode}", [f"--outbuf={mode}"]) + assert r["status"] == "Success", r["error"] + + def test_blocking_io_with_compression(self): + r = _run_ssh_test("SSH --blocking-io -c", ["--blocking-io", "-z"]) + assert r["status"] == "Success", r["error"] + + def test_trust_sender(self): + r = _run_ssh_test("SSH Trust Sender (--trust-sender)", ["--trust-sender"]) + assert r["status"] == "Success", r["error"] + + def test_remote_option_reaches_server(self): + """--remote-option=OPT appends OPT to the remote server command line and + the server honors it. Over SSH the server is launched without + --allow-delete, so a bare --delete is inert (nothing is removed). If + --remote-option=--allow-delete really reaches the remote server, the + receiver's deletion policy becomes permissive and the stale destination + file IS removed. Asserting the file is gone is therefore a positive + proof the forwarded option was honored by the server.""" + src = SOURCE_DIR + if os.path.exists(src): + shutil.rmtree(src) + os.makedirs(src) + with open(os.path.join(src, "keep.txt"), "w") as f: + f.write("kept\n") + with open(os.path.join(src, "stale.txt"), "w") as f: + f.write("stale\n") + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + + # Initial push so the destination mirrors the source. + clean_dir(DEST_DIR) + ssh_dest = f"localhost:{DEST_DIR}" + base = CLIENT_CMD + [src, ssh_dest, "--save-to-disk", + "--fastsync-server-path", os.path.join(BUILD_DIR, "server")] + first = subprocess.run(base, text=True, capture_output=True) + assert first.returncode == 0, f"initial push failed: {(first.stderr or first.stdout)[:200]}" + assert os.path.exists(os.path.join(received, "stale.txt")) + + # Remove stale.txt from the source and re-push with --delete + + # --remote-option=--allow-delete. Forwarding --allow-delete to the + # server is what makes the deletion actually happen. + os.remove(os.path.join(src, "stale.txt")) + second = subprocess.run(base + ["--delete", "--remote-option=--allow-delete"], + text=True, capture_output=True) + assert second.returncode == 0, \ + f"second push failed: {(second.stderr or second.stdout)[:200]}" + assert not os.path.exists(os.path.join(received, "stale.txt")), ( + "stale.txt still present: --allow-delete (forwarded via " + "--remote-option) did not reach the remote server" + ) + assert os.path.exists(os.path.join(received, "keep.txt")) diff --git a/tests/integration/test_stop.py b/tests/integration/test_stop.py new file mode 100644 index 0000000..28cde86 --- /dev/null +++ b/tests/integration/test_stop.py @@ -0,0 +1,241 @@ +"""--stop-after / --stop-at deadline-stop integration tests. + +These cover the client-only sender stop conditions: --stop-after=MINS stops +after N elapsed minutes, --stop-at=HH:MM[:SS] or now+N[smhd] stops at an +absolute (or relative) wall-clock time. A reached deadline ends the transfer +elegantly at the next chunk/file boundary -- whatever was already transferred is +kept, the completion tail still runs, and the exit code is 0 (like rsync's +clean "stopped early" behavior). Malformed values are rejected up front. +""" +import filecmp +import os +import shutil +import time + +import pytest + +from common import ( + TEST_DATA_DIR, + run_client, + clean_dir, + get_dest_received_dir, + verify_transfer, +) + + +def _make(self_prefix): + source = os.path.join(TEST_DATA_DIR, f"stop_{self_prefix}_src") + dest = os.path.join(TEST_DATA_DIR, f"stop_{self_prefix}_dst") + clean_dir(source) + shutil.rmtree(dest, ignore_errors=True) + os.makedirs(dest) + return source, dest + + +def _received_files(root): + """All files under `root`, relative paths.""" + if not os.path.isdir(root): + return [] + return [ + os.path.relpath(os.path.join(dirpath, name), root) + for dirpath, _, names in os.walk(root) + for name in names + ] + + +def _seed_source(source): + """Create a handful of regular and nested files.""" + files = { + "small.txt": b"hello world\n", + "medium.txt": b"the quick brown fox jumps over the lazy dog\n" * 400, + "binary.bin": bytes(range(256)) * 100, + "nested/deep.txt": b"deeply nested file\n", + "nested/another.txt": b"another nested file\n" * 40, + } + for rel, content in files.items(): + path = os.path.join(source, rel) + os.makedirs(os.path.dirname(path), exist_ok=True) + with open(path, "wb") as fh: + fh.write(content) + + +def _seed_many(source, count=40, size=32 * 1024): + """Create `count` same-size regular files (enough to span several chunks).""" + blob = os.urandom(size) + for i in range(count): + with open(os.path.join(source, f"f{i:04d}.dat"), "wb") as fh: + fh.write(blob) + + +def _seed_dest_by_transfer(source, dest, port, extra=None): + """Do a plain full transfer source->dest so dest exactly mirrors source.""" + run_client(source, dest, flags=(extra or []), port=port) + + +def _received_subset_matches(source, received): + """Every file under `received` exists under `source` with identical bytes.""" + if not os.path.isdir(received): + return not _received_files(received) + rels = _received_files(received) + for rel in rels: + src = os.path.join(source, rel) + dst = os.path.join(received, rel) + if not os.path.isfile(src) or not filecmp.cmp(src, dst, shallow=False): + return False + return True + + +class TestStopAfter: + @pytest.mark.ci + def test_stop_after_within_window(self, shared_server): + """A --stop-after set well past the run's duration lets it finish fully.""" + source, dest = _make("within") + _seed_source(source) + result, _ = run_client(source, dest, flags=["--stop-after=60"], + port=shared_server.port) + assert result.returncode == 0, \ + f"--stop-after full run failed: {(result.stderr or result.stdout)[:400]}" + received = get_dest_received_dir(dest, source) + mismatches, missing = verify_transfer(source, received) + assert not mismatches and not missing, \ + f"full transfer mismatch: missing={missing} mismatches={mismatches}" + + @pytest.mark.ci + def test_stop_after_rejects_nonpositive(self, shared_server): + """0 and negative minutes are invalid (must be a positive integer).""" + source, dest = _make("reject") + _seed_source(source) + for bad in ("0", "-1"): + result, _ = run_client(source, dest, flags=[f"--stop-after={bad}"], + port=shared_server.port) + assert result.returncode != 0, f"--stop-after={bad} should be rejected" + + +class TestStopAt: + @pytest.mark.ci + def test_stop_at_past(self, shared_server): + """A --stop-at already in the past stops the transfer immediately but + cleanly (exit 0, nothing transferred).""" + source, dest = _make("past") + _seed_source(source) + # Use a same-day HH:MM two minutes in the past when that cannot roll + # over into the previous day (which would parse as a FUTURE time today); + # otherwise fall back to now+0s which is deterministically immediate. + lt = time.localtime() + if lt.tm_hour * 60 + lt.tm_min >= 3: + past = time.localtime(time.time() - 120) + stop_value = f"{past.tm_hour:02d}:{past.tm_min:02d}" + else: + stop_value = "now+0s" + result, _ = run_client(source, dest, flags=[f"--stop-at={stop_value}"], + port=shared_server.port) + assert result.returncode == 0, \ + f"--stop-at past run failed (rc {result.returncode}): " \ + f"{(result.stderr or result.stdout)[:400]}" + received = get_dest_received_dir(dest, source) + assert _received_files(received) == [], \ + f"expected nothing transferred, got {_received_files(received)}" + + @pytest.mark.ci + def test_stop_at_now_plus_stops_immediately(self, shared_server): + """now+0s resolves to the current instant, so the transfer stops at once.""" + source, dest = _make("nowplus") + _seed_source(source) + result, _ = run_client(source, dest, flags=["--stop-at=now+0s"], + port=shared_server.port) + assert result.returncode == 0, \ + f"--stop-at=now+0s should stop cleanly: " \ + f"{(result.stderr or result.stdout)[:400]}" + received = get_dest_received_dir(dest, source) + assert _received_files(received) == [], \ + f"expected nothing transferred, got {_received_files(received)}" + + @pytest.mark.ci + def test_stop_rejects_garbage(self, shared_server): + """Malformed --stop-at/--stop-after values are rejected up front.""" + source, dest = _make("garbage") + _seed_source(source) + for flag in ("--stop-after=abc", "--stop-at=12:99", "--stop-at=12", + "--stop-at=now+5x", "--stop-at=now-5s"): + result, _ = run_client(source, dest, flags=[flag], + port=shared_server.port) + assert result.returncode != 0, f"{flag} should be rejected" + + +class TestStopPartial: + """A genuine mid-transfer stop leaves a valid, strict non-empty prefix.""" + + @pytest.mark.ci + def test_stop_mid_transfer_leaves_valid_partial(self, shared_server): + """With --bwlimit a real deadline cuts the transfer mid-way: what WAS + transferred is byte-identical, not everything is transferred, and the + run returns 0 without corrupting any file.""" + source, dest = _make("partial") + _seed_many(source, count=60, size=32 * 1024) + flags = ["--chunk-size", "262144", "--bwlimit", "100", "--stop-at=now+3s"] + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, \ + f"mid-transfer stop failed (rc {result.returncode}): " \ + f"{(result.stderr or result.stdout)[:400]}" + received = get_dest_received_dir(dest, source) + got = _received_files(received) + assert len(got) > 0, "expected an early stop to still transfer a prefix" + assert len(got) < 60, \ + f"expected a PARTIAL transfer (all 60 arrived): stopped too late" + assert _received_subset_matches(source, received), \ + f"received files are not a byte-identical subset of the source" + + +class TestStopDelete: + """--delete must never wipe the destination when the scan is cut short.""" + + def _seed(self, prefix, port, many=False): + source, dest = _make(prefix) + if many: + _seed_many(source, count=40, size=96 * 1024) + else: + _seed_source(source) + _seed_dest_by_transfer(source, dest, port) + return source, dest + + @pytest.mark.ci + def test_stop_delete_immediate_preserves_source_mirrors(self, shared_server): + """Immediate stop + --delete: the incomplete/empty keep-set must NOT + delete the seeded source mirrors (returncode 0, files survive).""" + source, dest = self._seed("del_imm", shared_server.port) + result, _ = run_client(source, dest, flags=["--delete", "--stop-at=now+0s"], + port=shared_server.port) + assert result.returncode == 0, \ + f"--delete immediate stop failed: {(result.stderr or result.stdout)[:400]}" + received = get_dest_received_dir(dest, source) + mismatches, missing = verify_transfer(source, received) + assert not mismatches and not missing, \ + f"--delete wiped source mirrors: missing={missing} mismatches={mismatches}" + + @pytest.mark.ci + def test_stop_delete_midscan_preserves_source_mirrors(self, shared_server): + """A mid-scan stop + --delete must suppress the partial keep-set so all + seeded source mirrors survive.""" + source, dest = self._seed("del_mid", shared_server.port, many=True) + flags = ["--delete", "--chunk-size", "262144", "--bwlimit", "300", "--stop-at=now+3s"] + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, \ + f"--delete mid-scan stop failed: {(result.stderr or result.stdout)[:400]}" + received = get_dest_received_dir(dest, source) + mismatches, missing = verify_transfer(source, received) + assert not mismatches and not missing, \ + f"--delete mid-scan wiped source mirrors: missing={missing} mismatches={mismatches}" + + @pytest.mark.ci + def test_stop_delete_multithreaded_preserves_source_mirrors(self, shared_server): + """-m immediate stop + --delete: the completion tail must not read the + still-appendable manifest (no race) and must not delete the mirrors.""" + source, dest = self._seed("del_mt", shared_server.port, many=True) + result, _ = run_client(source, dest, flags=["--threads", "--delete", "--stop-at=now+0s"], + port=shared_server.port) + assert result.returncode == 0, \ + f"--threads --delete immediate stop failed: {(result.stderr or result.stdout)[:400]}" + received = get_dest_received_dir(dest, source) + mismatches, missing = verify_transfer(source, received) + assert not mismatches and not missing, \ + f"--threads --delete wiped source mirrors: missing={missing} mismatches={mismatches}" \ No newline at end of file diff --git a/tests/integration/test_tcp.py b/tests/integration/test_tcp.py index 18cc07c..ffa935e 100644 --- a/tests/integration/test_tcp.py +++ b/tests/integration/test_tcp.py @@ -21,7 +21,8 @@ def setup_test_data(): generate_test_files(SOURCE_DIR, full=False) clean_dir(DEST_DIR) yield - shutil.rmtree(TEST_DATA_DIR, ignore_errors=True) + shutil.rmtree(SOURCE_DIR, ignore_errors=True) + shutil.rmtree(DEST_DIR, ignore_errors=True) def _run_tcp_test(name, port, flags, use_metadata=True, posix=False): @@ -29,11 +30,11 @@ def _run_tcp_test(name, port, flags, use_metadata=True, posix=False): clean_dir(DEST_DIR) if posix: result, dur = run_client_posix(SOURCE_DIR, DEST_DIR, - flags=(["-M"] if use_metadata else []) + flags, + flags=(["--preserve"] if use_metadata else []) + flags, port=port) else: result, dur = run_client(SOURCE_DIR, DEST_DIR, - flags=(["-M"] if use_metadata else []) + flags, + flags=(["--preserve"] if use_metadata else []) + flags, port=port) if result.returncode != 0: @@ -49,10 +50,12 @@ def _run_tcp_test(name, port, flags, use_metadata=True, posix=False): class TestTCPStandard: + @pytest.mark.ci def test_standard(self, shared_server): r = _run_tcp_test("Standard", shared_server.port, []) assert r["status"] == "Success", r["error"] + @pytest.mark.ci def test_posix_args(self, shared_server): r = _run_tcp_test("Posix Args", shared_server.port, [], posix=True) assert r["status"] == "Success", r["error"] @@ -63,40 +66,76 @@ class TestTCPStandard: class TestTCPFlags: + @pytest.mark.ci def test_multithreading(self, shared_server): - r = _run_tcp_test("Multithreading (-m)", shared_server.port, ["-m"]) + r = _run_tcp_test("Multithreading (--threads)", shared_server.port, ["--threads"]) assert r["status"] == "Success", r["error"] + @pytest.mark.ci def test_compression(self, shared_server): - r = _run_tcp_test("Compression (-c)", shared_server.port, ["-c"]) + r = _run_tcp_test("Compression (-z)", shared_server.port, ["-z"]) + assert r["status"] == "Success", r["error"] + + def test_compression_threads(self, shared_server): + r = _run_tcp_test("Compression threads (-z --compress-threads=2)", shared_server.port, + ["-z", "--compress-threads=2"]) assert r["status"] == "Success", r["error"] def test_chunk_serialization(self, shared_server): - r = _run_tcp_test("Chunk Serialization (-s)", shared_server.port, ["-s"]) + r = _run_tcp_test("Chunk Serialization (--chunk-serialization)", shared_server.port, ["--chunk-serialization"]) assert r["status"] == "Success", r["error"] def test_compression_chunk(self, shared_server): - r = _run_tcp_test("Compression + Chunk (-c -s)", shared_server.port, ["-c", "-s"]) + r = _run_tcp_test("Compression + Chunk (-z --chunk-serialization)", shared_server.port, ["-z", "--chunk-serialization"]) assert r["status"] == "Success", r["error"] def test_multithread_compression(self, shared_server): - r = _run_tcp_test("Multithreading + Compression (-m -c)", shared_server.port, ["-m", "-c"]) + r = _run_tcp_test("Multithreading + Compression (--threads -z)", shared_server.port, ["--threads", "-z"]) assert r["status"] == "Success", r["error"] def test_multithread_chunk(self, shared_server): - r = _run_tcp_test("Multithreading + Chunk (-m -s)", shared_server.port, ["-m", "-s"]) + r = _run_tcp_test("Multithreading + Chunk (--threads --chunk-serialization)", shared_server.port, ["--threads", "--chunk-serialization"]) assert r["status"] == "Success", r["error"] def test_all_flags(self, shared_server): - r = _run_tcp_test("Multithread + Compression + Chunk (-m -c -s)", shared_server.port, ["-m", "-c", "-s"]) + r = _run_tcp_test("Multithread + Compression + Chunk (--threads -z --chunk-serialization)", shared_server.port, ["--threads", "-z", "--chunk-serialization"]) assert r["status"] == "Success", r["error"] def test_sendfile(self, shared_server): - r = _run_tcp_test("Sendfile (-f)", shared_server.port, ["-f"]) + r = _run_tcp_test("Sendfile (--sendfile)", shared_server.port, ["--sendfile"]) assert r["status"] == "Success", r["error"] def test_sendfile_multithread(self, shared_server): - r = _run_tcp_test("Sendfile + Multithreading (-f -m)", shared_server.port, ["-f", "-m"]) + r = _run_tcp_test("Sendfile + Multithreading (--sendfile --threads)", shared_server.port, ["--sendfile", "--threads"]) + assert r["status"] == "Success", r["error"] + + +class TestTCPSocketOptions: + """--sockopts, -4/-6 and --address: rsync-compatible socket/bind options. + + These are purely local (client-side) socket concerns that never cross the + wire, so each is exercised by a normal transfer succeeding end-to-end.""" + + @pytest.mark.ci + def test_sockopts_apply(self, shared_server): + r = _run_tcp_test("Sockopts (TCP_NODELAY=1,SO_KEEPALIVE=1)", shared_server.port, + ["--sockopts=TCP_NODELAY=1,SO_KEEPALIVE=1"]) + assert r["status"] == "Success", r["error"] + + def test_sockopts_buffer_sizes(self, shared_server): + r = _run_tcp_test("Sockopts buffer sizes (SO_RCVBUF/SO_SNDBUF)", shared_server.port, + ["--sockopts=SO_RCVBUF=131072,SO_SNDBUF=131072"]) + assert r["status"] == "Success", r["error"] + + def test_ipv4_forced(self, shared_server): + r = _run_tcp_test("Force IPv4 (-4)", shared_server.port, ["-4"]) + assert r["status"] == "Success", r["error"] + + @pytest.mark.skipif(shutil.which("ip") is None, + reason="requires ip tooling to enumerate a usable local address") + def test_address_source_bind(self, shared_server): + r = _run_tcp_test("Source bind (--address=127.0.0.1)", shared_server.port, + ["--address", "127.0.0.1"]) assert r["status"] == "Success", r["error"] diff --git a/tests/integration/test_tls.py b/tests/integration/test_tls.py index 06e8caa..1e5c029 100644 --- a/tests/integration/test_tls.py +++ b/tests/integration/test_tls.py @@ -92,16 +92,19 @@ def setup_test_data(): generate_test_files(SOURCE_DIR, full=False) clean_dir(DEST_DIR) yield - shutil.rmtree(TEST_DATA_DIR, ignore_errors=True) + shutil.rmtree(SOURCE_DIR, ignore_errors=True) + shutil.rmtree(DEST_DIR, ignore_errors=True) class TestTLSBasic: + @pytest.mark.ci def test_tls_server_client(self, certs): """Basic TLS: server with cert/key, client with cert/key + CA.""" clean_dir(DEST_DIR) with ServerManager() as server: server.start(extra_args=[ "--tls", "--cert", certs["server_cert"], "--key", certs["server_key"], + "--ca", certs["ca"], "--client-cn", "fastsync-client", ]) result, dur = run_client( SOURCE_DIR, DEST_DIR, @@ -125,10 +128,11 @@ class TestTLSBasic: with ServerManager() as server: server.start(extra_args=[ "--tls", "--cert", certs["server_cert"], "--key", certs["server_key"], + "--ca", certs["ca"], "--client-cn", "fastsync-client", ]) result, dur = run_client( SOURCE_DIR, DEST_DIR, - flags=["-c", "--tls", + flags=["-z", "--tls", "--cert", certs["client_cert"], "--key", certs["client_key"], "--ca", certs["ca"]], port=server.port, @@ -149,10 +153,11 @@ class TestTLSBasic: with ServerManager() as server: server.start(extra_args=[ "--tls", "--cert", certs["server_cert"], "--key", certs["server_key"], + "--ca", certs["ca"], "--client-cn", "fastsync-client", ]) result, dur = run_client( SOURCE_DIR, DEST_DIR, - flags=["-m", "--tls", + flags=["--threads", "--tls", "--cert", certs["client_cert"], "--key", certs["client_key"], "--ca", certs["ca"]], port=server.port, diff --git a/tests/runner.c b/tests/runner.c index 0bba0f1..3837a6f 100644 --- a/tests/runner.c +++ b/tests/runner.c @@ -1,16 +1,24 @@ #include "test_array_list.h" +#include "test_batch.h" #include "test_chunk.h" +#include "test_change_list.h" +#include "test_checksum.h" #include "test_client_cli.h" #include "test_compression.h" #include "test_config.h" +#include "test_credentials.h" #include "test_data.h" +#include "test_daemon_conf.h" +#include "test_delay_updates.h" #include "test_delta.h" #include "test_file.h" #include "test_file_sendfile.h" #include "test_fuzz_smoke.h" #include "test_glob.h" +#include "test_iconv.h" #include "test_log.h" #include "test_metadata.h" +#include "test_motd.h" #include "test_multiprocessing.h" #include "test_property.h" #include "test_protocol.h" @@ -18,13 +26,17 @@ #include "test_robustness.h" #include "test_scanner.h" #include "test_server.h" +#include "test_server_cli.h" #include "test_shared_utils.h" #include "test_stress.h" +#include "test_stop.h" #include "test_transport_tcp.h" #include "test_transport_ssh.h" #include "test_transport_tls.h" #include "test_utils.h" +#include "test_xattr.h" #include +#include // Define global test state variables int tests_run = 0; @@ -32,33 +44,46 @@ int tests_failed = 0; bool current_test_failed = false; int main() { + signal(SIGPIPE, SIG_IGN); printf("\033[1;36m=== RUNNING UNIT TESTS ===\033[0m\n\n"); RUN_TEST(test_queue); RUN_TEST(test_array_list); RUN_TEST(test_shared_utils); RUN_TEST(test_chunk); + RUN_TEST(test_batch); + RUN_TEST(test_change_list); RUN_TEST(test_config); + RUN_TEST(test_credentials); RUN_TEST(test_compression); RUN_TEST(test_scanner); + RUN_TEST(test_checksum); RUN_TEST(test_delta); RUN_TEST(test_data); RUN_TEST(test_protocol); RUN_TEST(test_metadata); RUN_TEST(test_glob); + RUN_TEST(test_iconv); RUN_TEST(test_file); + RUN_TEST(test_trust_sender); + RUN_TEST(test_delay_updates); RUN_TEST(test_file_sendfile); RUN_TEST(test_multiprocessing); RUN_TEST(test_log); RUN_TEST(test_robustness); RUN_TEST(test_stress); + RUN_TEST(test_stop); RUN_TEST(test_property); RUN_TEST(test_transport_tcp); RUN_TEST(test_transport_ssh); RUN_TEST(test_transport_tls); RUN_TEST(test_client_cli); RUN_TEST(test_server); + RUN_TEST(test_daemon_conf); + RUN_TEST(test_motd); + RUN_TEST(test_server_cli); RUN_TEST(test_fuzz_smoke); + RUN_TEST(test_xattr); printf("\n\033[1;36m=== TEST SUMMARY ===\033[0m\n"); printf("Total Tests Run: %d\n", tests_run); diff --git a/tests/test_array_list.c b/tests/test_array_list.c index 1bf05d9..964458d 100644 --- a/tests/test_array_list.c +++ b/tests/test_array_list.c @@ -17,6 +17,8 @@ void test_array_list() { // Test adding int* val1 = malloc(sizeof(int)); + if (!val1) + return; *val1 = 42; array_list_add(list, val1); EXPECT_EQ_INT(list->size, 1); @@ -26,6 +28,8 @@ void test_array_list() { // Initial capacity is 100. Let's add 105 elements. for (int i = 0; i < 105; i++) { int* val = malloc(sizeof(int)); + if (!val) + return; *val = i; array_list_add(list, val); } diff --git a/tests/test_batch.c b/tests/test_batch.c new file mode 100644 index 0000000..a2d088e --- /dev/null +++ b/tests/test_batch.c @@ -0,0 +1,255 @@ +#include "batch.h" +#include "chunk.h" +#include "config.h" +#include "file.h" +#include "metadata.h" +#include "test_utils.h" +#include "utils.h" +#include +#include +#include +#include +#include + +static void batch_test_cleanup(void) { + unlink("batch_dest/batch_src.txt"); + rmdir("batch_dest"); + unlink("batch_src.txt"); + unlink("batch.bin"); + unlink("batch_bad.bin"); + unlink("batch_trunc.bin"); + unlink("batch_big.bin"); +} + +/* A batch round-trips a full file image byte-identically: write header+chunks, + * then apply the file to a fresh destination root and verify the content landed + * unchanged. */ +static void test_batch_roundtrip() { + batch_test_cleanup(); + EXPECT_EQ_INT(mkdir("batch_dest", 0755), 0); + + const char* content = "residual batch full image payload\nwith \x01\x02\x03 bytes\n"; + size_t content_len = strlen(content); + file_write_to_disk("batch_src.txt", content, content_len, false, false); + + struct stat st; + EXPECT_EQ_INT(stat("batch_src.txt", &st), 0); + File* f = file_create("batch_src.txt"); + EXPECT_NOT_NULL(f); + f->data->size = (unsigned long long)st.st_size; + EXPECT_TRUE(file_load_data(f)); + File* files[1] = {f}; + Chunk* chunk = chunk_create(files, 1); + EXPECT_NOT_NULL(chunk); + + Config* config = config_create(); + EXPECT_NOT_NULL(config); + + int wfd = open("batch.bin", O_WRONLY | O_CREAT | O_TRUNC, 0644); + EXPECT_TRUE(wfd >= 0); + EXPECT_TRUE(batch_write_header(wfd, config)); + EXPECT_TRUE(batch_write_chunk(wfd, chunk)); + EXPECT_EQ_INT(close(wfd), 0); + chunk_destroy(chunk); /* frees f */ + + int rfd = open("batch.bin", O_RDONLY); + EXPECT_TRUE(rfd >= 0); + EXPECT_EQ_INT(batch_read_apply(rfd, config, "batch_dest"), 0); + EXPECT_EQ_INT(close(rfd), 0); + + char* dest_path = path_cat("batch_dest", "batch_src.txt"); + EXPECT_NOT_NULL(dest_path); + FILE* df = fopen(dest_path, "rb"); + EXPECT_NOT_NULL(df); + char buf[512]; + size_t n = fread(buf, 1, sizeof(buf), df); + EXPECT_EQ_INT(fclose(df), 0); + EXPECT_EQ_INT((int)n, (int)content_len); + EXPECT_EQ_INT(n == content_len && memcmp(buf, content, content_len) == 0, 1); + free(dest_path); + + config_delete(config); + batch_test_cleanup(); +} + +static void test_batch_roundtrip_metadata() { + batch_test_cleanup(); + EXPECT_EQ_INT(mkdir("batch_dest", 0755), 0); + + const char* content = "metadata-carrying batch image\n"; + size_t content_len = strlen(content); + file_write_to_disk("batch_src.txt", content, content_len, false, false); + + struct stat st; + EXPECT_EQ_INT(stat("batch_src.txt", &st), 0); + File* f = file_create("batch_src.txt"); + EXPECT_NOT_NULL(f); + f->data->size = (unsigned long long)st.st_size; + EXPECT_TRUE(file_load_data(f)); + f->metadata = file_metadata_create("batch_src.txt", &st, false, false); + EXPECT_NOT_NULL(f->metadata); + File* files[1] = {f}; + Chunk* chunk = chunk_create(files, 1); + EXPECT_NOT_NULL(chunk); + + Config* config = config_create(); + EXPECT_NOT_NULL(config); + config->use_metadata = true; + + int wfd = open("batch.bin", O_WRONLY | O_CREAT | O_TRUNC, 0644); + EXPECT_TRUE(wfd >= 0); + EXPECT_TRUE(batch_write_header(wfd, config)); + EXPECT_TRUE(batch_write_chunk(wfd, chunk)); + EXPECT_EQ_INT(close(wfd), 0); + chunk_destroy(chunk); /* frees f */ + + int rfd = open("batch.bin", O_RDONLY); + EXPECT_TRUE(rfd >= 0); + EXPECT_EQ_INT(batch_read_apply(rfd, config, "batch_dest"), 0); + EXPECT_EQ_INT(close(rfd), 0); + + char* dest_path = path_cat("batch_dest", "batch_src.txt"); + EXPECT_NOT_NULL(dest_path); + FILE* df = fopen(dest_path, "rb"); + EXPECT_NOT_NULL(df); + char buf[512]; + size_t n = fread(buf, 1, sizeof(buf), df); + EXPECT_EQ_INT(fclose(df), 0); + EXPECT_EQ_INT((int)n, (int)content_len); + EXPECT_EQ_INT(memcmp(buf, content, content_len) == 0, 1); + free(dest_path); + + config_delete(config); + batch_test_cleanup(); +} + +/* A corrupt magic (and only 11 bytes of junk) is rejected, never applied. */ +static void test_batch_reject_bad_magic() { + Config* config = config_create(); + EXPECT_NOT_NULL(config); + int fd = open("batch_bad.bin", O_WRONLY | O_CREAT | O_TRUNC, 0644); + EXPECT_TRUE(fd >= 0); + const char* garbage = "NOTABATCHFXV"; + EXPECT_EQ_INT(write(fd, garbage, strlen(garbage)), (ssize_t)strlen(garbage)); + EXPECT_EQ_INT(close(fd), 0); + fd = open("batch_bad.bin", O_RDONLY); + EXPECT_TRUE(fd >= 0); + EXPECT_EQ_INT(batch_read_apply(fd, config, "batch_dest"), -1); + EXPECT_EQ_INT(close(fd), 0); + config_delete(config); + unlink("batch_bad.bin"); +} + +/* A clean header with a length prefix promising 100 bytes but only 12 present + * is a truncated record and is rejected (never crashes, never applies). */ +static void test_batch_reject_truncated() { + Config* config = config_create(); + EXPECT_NOT_NULL(config); + int fd = open("batch_trunc.bin", O_WRONLY | O_CREAT | O_TRUNC, 0644); + EXPECT_TRUE(fd >= 0); + EXPECT_TRUE(batch_write_header(fd, config)); + unsigned long long length = 100; + EXPECT_EQ_INT(write(fd, &length, sizeof(length)), (ssize_t)sizeof(length)); + const char* partial = "onlytwelvebytes"; + EXPECT_EQ_INT(write(fd, partial, 15), (ssize_t)15); + EXPECT_EQ_INT(close(fd), 0); + fd = open("batch_trunc.bin", O_RDONLY); + EXPECT_TRUE(fd >= 0); + EXPECT_EQ_INT(batch_read_apply(fd, config, "batch_dest"), -1); + EXPECT_EQ_INT(close(fd), 0); + config_delete(config); + unlink("batch_trunc.bin"); +} + +/* A length prefix above the 64 MB cap is refused before any allocation. */ +static void test_batch_reject_oversized() { + Config* config = config_create(); + EXPECT_NOT_NULL(config); + int fd = open("batch_big.bin", O_WRONLY | O_CREAT | O_TRUNC, 0644); + EXPECT_TRUE(fd >= 0); + EXPECT_TRUE(batch_write_header(fd, config)); + unsigned long long length = BATCH_MAX_RECORD + 16U; + EXPECT_EQ_INT(write(fd, &length, sizeof(length)), (ssize_t)sizeof(length)); + EXPECT_EQ_INT(close(fd), 0); + fd = open("batch_big.bin", O_RDONLY); + EXPECT_TRUE(fd >= 0); + EXPECT_EQ_INT(batch_read_apply(fd, config, "batch_dest"), -1); + EXPECT_EQ_INT(close(fd), 0); + config_delete(config); + unlink("batch_big.bin"); +} + +/* A clean header followed by a length prefix with NO record bytes at all (clean + * EOF on the record-body read) must be rejected as truncated — it must not feed + * an uninitialized buffer to chunk_deserialize. Regression test for a + * confirmed uninitialized-read on the untrusted read side. */ +static void test_batch_reject_eof_after_prefix() { + Config* config = config_create(); + EXPECT_NOT_NULL(config); + int fd = open("batch_eof.bin", O_WRONLY | O_CREAT | O_TRUNC, 0644); + EXPECT_TRUE(fd >= 0); + EXPECT_TRUE(batch_write_header(fd, config)); + unsigned long long length = 32; + EXPECT_EQ_INT(write(fd, &length, sizeof(length)), (ssize_t)sizeof(length)); + EXPECT_EQ_INT(close(fd), 0); + fd = open("batch_eof.bin", O_RDONLY); + EXPECT_TRUE(fd >= 0); + EXPECT_EQ_INT(batch_read_apply(fd, config, "batch_dest"), -1); + EXPECT_EQ_INT(close(fd), 0); + config_delete(config); + unlink("batch_eof.bin"); +} + +/* A malicious batch record whose chunk carries a path-traversal wire path must + * be refused by the apply path — never applied outside the destination root. + * We craft a chunk whose wire path is `../escape.txt` (the local source file + * is a benign temp file; only the transmitted path is hostile) and assert the + * apply refuses it and nothing is created outside the root. */ +static void test_batch_reject_traversal_path() { + const char* content = "hostile traversal image\n"; + size_t content_len = strlen(content); + file_write_to_disk("batch_trav_src.txt", content, content_len, false, false); + + struct stat st; + EXPECT_EQ_INT(stat("batch_trav_src.txt", &st), 0); + File* f = file_create("batch_trav_src.txt"); + EXPECT_NOT_NULL(f); + f->data->size = (unsigned long long)st.st_size; + EXPECT_TRUE(file_load_data(f)); + f->send_path = str_dup("../escape.txt"); + EXPECT_NOT_NULL(f->send_path); + File* files[1] = {f}; + Chunk* chunk = chunk_create(files, 1); + EXPECT_NOT_NULL(chunk); + + Config* config = config_create(); + EXPECT_NOT_NULL(config); + + int wfd = open("batch_trav.bin", O_WRONLY | O_CREAT | O_TRUNC, 0644); + EXPECT_TRUE(wfd >= 0); + EXPECT_TRUE(batch_write_header(wfd, config)); + EXPECT_TRUE(batch_write_chunk(wfd, chunk)); + EXPECT_EQ_INT(close(wfd), 0); + chunk_destroy(chunk); /* frees f and f->send_path */ + + int rfd = open("batch_trav.bin", O_RDONLY); + EXPECT_TRUE(rfd >= 0); + EXPECT_EQ_INT(batch_read_apply(rfd, config, "batch_dest"), -1); + EXPECT_EQ_INT(close(rfd), 0); + unlink("../escape.txt"); /* clear any stale file so the probe below is clean */ + EXPECT_TRUE(access("../escape.txt", F_OK) != 0); + + config_delete(config); + unlink("batch_trav.bin"); + unlink("batch_trav_src.txt"); +} + +void test_batch() { + test_batch_roundtrip(); + test_batch_roundtrip_metadata(); + test_batch_reject_bad_magic(); + test_batch_reject_truncated(); + test_batch_reject_oversized(); + test_batch_reject_eof_after_prefix(); + test_batch_reject_traversal_path(); +} \ No newline at end of file diff --git a/tests/test_batch.h b/tests/test_batch.h new file mode 100644 index 0000000..84f2252 --- /dev/null +++ b/tests/test_batch.h @@ -0,0 +1,6 @@ +#ifndef TEST_BATCH_H +#define TEST_BATCH_H + +void test_batch(); + +#endif \ No newline at end of file diff --git a/tests/test_change_list.c b/tests/test_change_list.c new file mode 100644 index 0000000..a5fe016 --- /dev/null +++ b/tests/test_change_list.c @@ -0,0 +1,101 @@ +#include "test_change_list.h" +#include "change_list.h" +#include "test_utils.h" +#include "utils.h" +#include +#include +#include + +static ChangeEvent sample_event(void) { + ChangeEvent event; + memset(&event, 0, sizeof(event)); + event.path = "/srv/root/sub/file.txt"; + event.decision = CHANGE_SENT; + event.is_directory = false; + event.size = 12345; + event.bytes_sent = 999; + event.mtime_sec = 1700000000; + return event; +} + +static void test_format_tokens() { + ChangeEvent event = sample_event(); + char* line = change_render_format("%f %n %l %b %M %%", &event); + EXPECT_NOT_NULL(line); + EXPECT_EQ_STR(line, "/srv/root/sub/file.txt file.txt 12345 999 1700000000 %"); + free(line); +} + +static void test_format_unknown_tokens_preserved() { + ChangeEvent event = sample_event(); + char* line = change_render_format("x%q=%f%z", &event); + EXPECT_NOT_NULL(line); + EXPECT_EQ_STR(line, "x%q=/srv/root/sub/file.txt%z"); + free(line); +} + +static void test_format_leaf_name() { + ChangeEvent event = sample_event(); + event.path = "bare.txt"; + char* line = change_render_format("%n|%f", &event); + EXPECT_NOT_NULL(line); + EXPECT_EQ_STR(line, "bare.txt|bare.txt"); + free(line); +} + +static void test_render_itemize_sent_file() { + ChangeEvent event = sample_event(); + char* line = change_render_itemize(&event); + EXPECT_NOT_NULL(line); + EXPECT_EQ_STR(line, ">f+++++++++ /srv/root/sub/file.txt"); + free(line); +} + +static void test_render_itemize_up_to_date_is_empty() { + ChangeEvent event = sample_event(); + event.decision = CHANGE_UP_TO_DATE; + char* line = change_render_itemize(&event); + EXPECT_NOT_NULL(line); + EXPECT_EQ_STR(line, ""); + free(line); +} + +static void test_render_list_line() { + char* line = change_render_list_line(0100644, 4096, 1700000000, "/srv/x.txt"); + EXPECT_NOT_NULL(line); + EXPECT_TRUE(strncmp(line, "-rw-r--r--", 10) == 0); + EXPECT_TRUE(strstr(line, "4096") != NULL); + EXPECT_TRUE(strstr(line, "/srv/x.txt") != NULL); + free(line); +} + +static void test_change_list_enabled() { + Config* config = config_create(); + EXPECT_NOT_NULL(config); + EXPECT_FALSE(change_list_enabled(config)); + config->itemize_changes = true; + EXPECT_TRUE(change_list_enabled(config)); + config->itemize_changes = false; + config->out_format = str_dup("%f"); + EXPECT_TRUE(change_list_enabled(config)); + free(config->out_format); + config->out_format = NULL; + EXPECT_FALSE(change_list_enabled(config)); + /* config_delete() closes log_file, so use a throwaway tmpfile. */ + config->log_file = tmpfile(); + EXPECT_NOT_NULL(config->log_file); + EXPECT_FALSE(change_list_enabled(config)); /* needs a format too */ + config->log_file_format = str_dup("%n"); + EXPECT_TRUE(change_list_enabled(config)); + config_delete(config); /* closes config->log_file */ +} + +void test_change_list() { + test_format_tokens(); + test_format_unknown_tokens_preserved(); + test_format_leaf_name(); + test_render_itemize_sent_file(); + test_render_itemize_up_to_date_is_empty(); + test_render_list_line(); + test_change_list_enabled(); +} diff --git a/tests/test_change_list.h b/tests/test_change_list.h new file mode 100644 index 0000000..8887e43 --- /dev/null +++ b/tests/test_change_list.h @@ -0,0 +1,6 @@ +#ifndef TEST_CHANGE_LIST_H +#define TEST_CHANGE_LIST_H + +void test_change_list(void); + +#endif diff --git a/tests/test_checksum.c b/tests/test_checksum.c new file mode 100644 index 0000000..a816271 --- /dev/null +++ b/tests/test_checksum.c @@ -0,0 +1,147 @@ +#include "test_checksum.h" +#include "checksum.h" +#include "test_utils.h" +#include + +/* Known xxHash64 vector (seed 0) for the empty string and a literal. + * The md5 vectors are the standard NIST/RFC1321 test strings. These pin the + * digest selection to genuinely distinct algorithm outputs so a --checksum- + * choice change is observable, not a silent no-op. */ + +static void test_checksum_xxh64_seed0() { + uint8_t out[CHECKSUM_MAX_DIGEST_LEN]; + size_t len = 0; + EXPECT_TRUE(checksum_digest(CHECKSUM_ALGO_XXH64, 0, "hello", 5, out, sizeof(out), &len)); + EXPECT_TRUE(len == (size_t)8); + /* Hard-coded: XXH64("hello", 5, 0). */ + const uint8_t expect[8] = {0xa3, 0x6d, 0x9f, 0x88, 0x7d, 0x82, 0xc7, 0x26}; + for (int i = 0; i < 8; i++) + EXPECT_EQ_INT(out[i], expect[i]); +} + +static void test_checksum_xxh64_empty() { + uint8_t out[CHECKSUM_MAX_DIGEST_LEN]; + size_t len = 0; + EXPECT_TRUE(checksum_digest(CHECKSUM_ALGO_XXH64, 0, "", 0, out, sizeof(out), &len)); + EXPECT_TRUE(len == (size_t)8); + /* XXH64("", 0, 0). */ + const uint8_t expect[8] = {0x99, 0xe9, 0xd8, 0x51, 0x37, 0xdb, 0x46, 0xef}; + for (int i = 0; i < 8; i++) + EXPECT_EQ_INT(out[i], expect[i]); +} + +/* A nonzero seed must change the xxh64 digest: the algorithm is genuinely + * seed-aware, deterministic, and distinct from seed 0. */ +static void test_checksum_xxh64_seed_changes_digest() { + uint8_t a[CHECKSUM_MAX_DIGEST_LEN], b[CHECKSUM_MAX_DIGEST_LEN]; + size_t alen = 0, blen = 0; + EXPECT_TRUE(checksum_digest(CHECKSUM_ALGO_XXH64, 7, "payload", 7, a, sizeof(a), &alen)); + EXPECT_TRUE(checksum_digest(CHECKSUM_ALGO_XXH64, 0, "payload", 7, b, sizeof(b), &blen)); + EXPECT_TRUE(alen == blen); + EXPECT_TRUE(memcmp(a, b, alen) != 0); +} + +static void test_checksum_xxh64_seed_deterministic() { + uint8_t a[CHECKSUM_MAX_DIGEST_LEN], b[CHECKSUM_MAX_DIGEST_LEN]; + size_t alen = 0, blen = 0; + EXPECT_TRUE(checksum_digest(CHECKSUM_ALGO_XXH64, 12345, "same", 4, a, sizeof(a), &alen)); + EXPECT_TRUE(checksum_digest(CHECKSUM_ALGO_XXH64, 12345, "same", 4, b, sizeof(b), &blen)); + EXPECT_TRUE(alen == blen); + EXPECT_TRUE(memcmp(a, b, alen) == 0); +} + +static void test_checksum_md5_vectors() { + uint8_t out[CHECKSUM_MAX_DIGEST_LEN]; + size_t len = 0; + EXPECT_TRUE(checksum_digest(CHECKSUM_ALGO_MD5, 0, "", 0, out, sizeof(out), &len)); + EXPECT_TRUE(len == (size_t)16); + const uint8_t expect_empty[16] = {0xd4, 0x1d, 0x8c, 0xd9, 0x8f, 0x00, 0xb2, 0x04, + 0xe9, 0x80, 0x09, 0x98, 0xec, 0xf8, 0x42, 0x7e}; + EXPECT_TRUE(memcmp(out, expect_empty, 16) == 0); + + /* MD5("abc") */ + EXPECT_TRUE(checksum_digest(CHECKSUM_ALGO_MD5, 0, "abc", 3, out, sizeof(out), &len)); + const uint8_t expect_abc[16] = {0x90, 0x01, 0x50, 0x98, 0x3c, 0xd2, 0x4f, 0xb0, + 0xd6, 0x96, 0x3f, 0x7d, 0x28, 0xe1, 0x7f, 0x72}; + EXPECT_TRUE(memcmp(out, expect_abc, 16) == 0); +} + +/* md5 is 16 bytes and differs from the 8-byte xxh64 for the same input, so the + * choice is observably different both in length and in content. */ +static void test_checksum_algo_lengths_distinct() { + EXPECT_EQ_INT((int)checksum_digest_len(CHECKSUM_ALGO_XXH64), 8); + EXPECT_EQ_INT((int)checksum_digest_len(CHECKSUM_ALGO_MD5), 16); + + uint8_t x[CHECKSUM_MAX_DIGEST_LEN], m[CHECKSUM_MAX_DIGEST_LEN]; + size_t xl = 0, ml = 0; + EXPECT_TRUE(checksum_digest(CHECKSUM_ALGO_XXH64, 0, "same content", 12, x, sizeof(x), &xl)); + EXPECT_TRUE(checksum_digest(CHECKSUM_ALGO_MD5, 0, "same content", 12, m, sizeof(m), &ml)); + EXPECT_TRUE(xl == (size_t)8); + EXPECT_TRUE(ml == (size_t)16); + EXPECT_TRUE(memcmp(x, m, 8) != 0); +} + +/* md5 has no seed: two distinct seeds give the same md5 digest (documented); + * the seed is only honored by xxh64 and the delta block hash (low 32 bits). */ +static void test_checksum_md5_seed_ignored() { + uint8_t a[CHECKSUM_MAX_DIGEST_LEN], b[CHECKSUM_MAX_DIGEST_LEN]; + size_t alen = 0, blen = 0; + EXPECT_TRUE(checksum_digest(CHECKSUM_ALGO_MD5, 0, "data", 4, a, sizeof(a), &alen)); + EXPECT_TRUE(checksum_digest(CHECKSUM_ALGO_MD5, 99, "data", 4, b, sizeof(b), &blen)); + EXPECT_TRUE(memcmp(a, b, alen) == 0); +} + +static void test_checksum_algo_name_mapping() { + EXPECT_EQ_INT(checksum_algo_from_name("xxh64"), (int)CHECKSUM_ALGO_XXH64); + EXPECT_EQ_INT(checksum_algo_from_name("XXH64"), (int)CHECKSUM_ALGO_XXH64); + EXPECT_EQ_INT(checksum_algo_from_name("xxhash"), (int)CHECKSUM_ALGO_XXH64); + EXPECT_EQ_INT(checksum_algo_from_name("XXHASH"), (int)CHECKSUM_ALGO_XXH64); + EXPECT_EQ_INT(checksum_algo_from_name("md5"), (int)CHECKSUM_ALGO_MD5); + EXPECT_EQ_INT(checksum_algo_from_name("MD5"), (int)CHECKSUM_ALGO_MD5); + EXPECT_TRUE(checksum_algo_from_name("sha256") < 0); + EXPECT_TRUE(checksum_algo_from_name("crc32") < 0); + EXPECT_TRUE(checksum_algo_from_name("none") < 0); + EXPECT_TRUE(checksum_algo_from_name("xxh3") < 0); + EXPECT_TRUE(checksum_algo_from_name("") < 0); + EXPECT_TRUE(checksum_algo_from_name(NULL) < 0); + + EXPECT_TRUE(checksum_algo_valid((int)CHECKSUM_ALGO_XXH64)); + EXPECT_TRUE(checksum_algo_valid((int)CHECKSUM_ALGO_MD5)); + EXPECT_FALSE(checksum_algo_valid(99)); + EXPECT_EQ_STR(checksum_algo_name(CHECKSUM_ALGO_XXH64), "xxh64"); + EXPECT_EQ_STR(checksum_algo_name(CHECKSUM_ALGO_MD5), "md5"); +} + +static void test_checksum_truncated_buffer_rejected() { + uint8_t small[4]; + size_t len = 0; + /* The digest cannot fit in a 4-byte buffer. */ + EXPECT_FALSE(checksum_digest(CHECKSUM_ALGO_XXH64, 0, "x", 1, small, sizeof(small), &len)); + EXPECT_FALSE(checksum_digest(CHECKSUM_ALGO_MD5, 0, "x", 1, small, sizeof(small), &len)); + EXPECT_FALSE(checksum_digest(CHECKSUM_ALGO_XXH64, 0, NULL, 5, small, sizeof(small), &len)); + EXPECT_FALSE(checksum_digest(CHECKSUM_ALGO_XXH64, 0, "x", 1, NULL, 0, &len)); + EXPECT_FALSE(checksum_digest(CHECKSUM_ALGO_XXH64, 0, "x", 1, small, sizeof(small), NULL)); +} + +/* A NULL data pointer with size 0 is the empty input, not an error. */ +static void test_checksum_null_empty_digest() { + uint8_t a[CHECKSUM_MAX_DIGEST_LEN], b[CHECKSUM_MAX_DIGEST_LEN]; + size_t alen = 0, blen = 0; + EXPECT_TRUE(checksum_digest(CHECKSUM_ALGO_XXH64, 0, NULL, 0, a, sizeof(a), &alen)); + EXPECT_TRUE(checksum_digest(CHECKSUM_ALGO_XXH64, 0, "", 0, b, sizeof(b), &blen)); + EXPECT_TRUE(alen == blen); + EXPECT_TRUE(memcmp(a, b, alen) == 0); +} + +void test_checksum(void) { + test_checksum_xxh64_seed0(); + test_checksum_xxh64_empty(); + test_checksum_xxh64_seed_changes_digest(); + test_checksum_xxh64_seed_deterministic(); + test_checksum_md5_vectors(); + test_checksum_algo_lengths_distinct(); + test_checksum_md5_seed_ignored(); + test_checksum_algo_name_mapping(); + test_checksum_truncated_buffer_rejected(); + test_checksum_null_empty_digest(); +} \ No newline at end of file diff --git a/tests/test_checksum.h b/tests/test_checksum.h new file mode 100644 index 0000000..8849100 --- /dev/null +++ b/tests/test_checksum.h @@ -0,0 +1,6 @@ +#ifndef TEST_CHECKSUM_H +#define TEST_CHECKSUM_H + +void test_checksum(void); + +#endif \ No newline at end of file diff --git a/tests/test_chunk.c b/tests/test_chunk.c index 63400d4..b1d6cf6 100644 --- a/tests/test_chunk.c +++ b/tests/test_chunk.c @@ -11,7 +11,7 @@ static void test_file_operations() { char* test_content = "Hello, Chunk System!"; unsigned long long test_len = strlen(test_content); - to_disk(test_path, test_content, test_len); + file_write_to_disk(test_path, test_content, test_len, false, false); File* f = file_create(test_path); EXPECT_NOT_NULL(f); @@ -43,8 +43,8 @@ static void test_chunk_operations() { char* content2 = "chunk item number 2"; unsigned long long len2 = strlen(content2); - to_disk(path1, content1, len1); - to_disk(path2, content2, len2); + file_write_to_disk(path1, content1, len1, false, false); + file_write_to_disk(path2, content2, len2, false, false); struct stat st1, st2; stat(path1, &st1); @@ -89,7 +89,201 @@ static void test_chunk_operations() { unlink(path2); } +/* A chunk mixing a regular file and an explicit directory entry (--dirs, with + * or without metadata) must round-trip through serialize/deserialize with the + * is_dir flag and the entry type marker preserved. */ +static void test_chunk_dir_entry_roundtrip() { + const char* file_path = "temp_chunk_dir_file.txt"; + const char* dir_path = "temp_chunk_dir_entry"; + const char* content = "regular file payload"; + + /* A failed earlier run can leave artifacts behind; start clean. */ + rmdir(dir_path); + unlink(file_path); + + file_write_to_disk(file_path, content, strlen(content), false, false); + EXPECT_EQ_INT(mkdir(dir_path, 0755), 0); + + for (int use_metadata = 0; use_metadata <= 1; use_metadata++) { + struct stat st; + EXPECT_EQ_INT(stat(file_path, &st), 0); + + File* reg = file_create(file_path); + EXPECT_NOT_NULL(reg); + reg->data->size = (unsigned long long)st.st_size; + EXPECT_TRUE(file_load_data(reg)); + + File* dir = file_create(dir_path); + EXPECT_NOT_NULL(dir); + dir->is_dir = true; + + if (use_metadata) { + reg->metadata = file_metadata_create(file_path, &st, false, false); + EXPECT_NOT_NULL(reg->metadata); + struct stat dst; + EXPECT_EQ_INT(stat(dir_path, &dst), 0); + dir->metadata = file_metadata_create(dir_path, &dst, false, false); + EXPECT_NOT_NULL(dir->metadata); + } + + File* files[2] = {reg, dir}; + Chunk* chunk = chunk_create(files, 2); + EXPECT_NOT_NULL(chunk); + + Data* serialized = chunk_serialize(chunk, use_metadata != 0); + EXPECT_NOT_NULL(serialized); + Chunk* deserialized = chunk_deserialize(serialized, use_metadata != 0); + EXPECT_NOT_NULL(deserialized); + EXPECT_EQ_INT(deserialized->element_count, 2); + EXPECT_FALSE(deserialized->items[0]->is_dir); + EXPECT_EQ_STR(deserialized->items[0]->path, file_path); + EXPECT_EQ_INT((int)deserialized->items[0]->data->size, (int)strlen(content)); + EXPECT_EQ_INT(memcmp(deserialized->items[0]->data->data, content, strlen(content)), 0); + EXPECT_TRUE(deserialized->items[1]->is_dir); + EXPECT_EQ_STR(deserialized->items[1]->path, dir_path); + EXPECT_EQ_INT((int)deserialized->items[1]->data->size, 0); + if (use_metadata) { + EXPECT_NOT_NULL(deserialized->items[0]->metadata); + EXPECT_NOT_NULL(deserialized->items[1]->metadata); + } else { + EXPECT_NULL(deserialized->items[0]->metadata); + EXPECT_NULL(deserialized->items[1]->metadata); + } + + data_destroy(serialized); + chunk_destroy(deserialized); + chunk_destroy(chunk); /* frees reg and dir */ + } + + unlink(file_path); + rmdir(dir_path); +} + +static void test_chunk_symlink_roundtrip() { + const char* file_path = "temp_chunk_symlink_file.txt"; + const char* link_path = "temp_chunk_symlink"; + const char* content = "regular payload"; + const char* target = "temp_chunk_symlink_file.txt"; + + rmdir(link_path); + unlink(file_path); + + file_write_to_disk(file_path, content, strlen(content), false, false); + + for (int use_metadata = 0; use_metadata <= 1; use_metadata++) { + struct stat st; + EXPECT_EQ_INT(stat(file_path, &st), 0); + + File* reg = file_create(file_path); + EXPECT_NOT_NULL(reg); + reg->data->size = (unsigned long long)st.st_size; + EXPECT_TRUE(file_load_data(reg)); + + File* link = file_create(link_path); + EXPECT_NOT_NULL(link); + link->is_symlink = true; + link->symlink_target = str_dup(target); + EXPECT_NOT_NULL(link->symlink_target); + + if (use_metadata) { + reg->metadata = file_metadata_create(file_path, &st, false, false); + EXPECT_NOT_NULL(reg->metadata); + link->metadata = file_metadata_create(file_path, &st, false, false); + EXPECT_NOT_NULL(link->metadata); + } + + File* files[2] = {reg, link}; + Chunk* chunk = chunk_create(files, 2); + EXPECT_NOT_NULL(chunk); + + Data* serialized = chunk_serialize(chunk, use_metadata != 0); + EXPECT_NOT_NULL(serialized); + Chunk* deserialized = chunk_deserialize(serialized, use_metadata != 0); + EXPECT_NOT_NULL(deserialized); + EXPECT_EQ_INT(deserialized->element_count, 2); + EXPECT_FALSE(deserialized->items[0]->is_symlink); + EXPECT_TRUE(deserialized->items[1]->is_symlink); + EXPECT_NULL(deserialized->items[0]->symlink_target); + EXPECT_EQ_STR(deserialized->items[1]->symlink_target, target); + EXPECT_EQ_INT((int)deserialized->items[1]->data->size, 0); + + data_destroy(serialized); + chunk_destroy(deserialized); + chunk_destroy(chunk); /* frees reg and link */ + } + + unlink(file_path); + rmdir(link_path); +} + +/* A --devices/--specials special entry (is_special + rdev) must round-trip + * through the chunk wire with a legal rdev. */ +static void test_chunk_special_rdev_roundtrip() { + const char* path = "temp_chunk_special_node"; + unlink(path); + File* special = file_create(path); + EXPECT_NOT_NULL(special); + special->is_special = true; + special->rdev_major = 1; + special->rdev_minor = 3; + struct stat st; + EXPECT_EQ_INT(stat("/dev/null", &st), 0); + special->metadata = file_metadata_create(path, &st, false, false); + EXPECT_NOT_NULL(special->metadata); + + File* files[1] = {special}; + Chunk* chunk = chunk_create(files, 1); + EXPECT_NOT_NULL(chunk); + Data* serialized = chunk_serialize(chunk, true); + EXPECT_NOT_NULL(serialized); + Chunk* deserialized = chunk_deserialize(serialized, true); + EXPECT_NOT_NULL(deserialized); + EXPECT_EQ_INT(deserialized->element_count, 1); + EXPECT_TRUE(deserialized->items[0]->is_special); + EXPECT_FALSE(deserialized->items[0]->is_dir); + EXPECT_EQ_INT((int)deserialized->items[0]->data->size, 0); + EXPECT_EQ_INT(deserialized->items[0]->rdev_major, 1); + EXPECT_EQ_INT(deserialized->items[0]->rdev_minor, 3); + EXPECT_NOT_NULL(deserialized->items[0]->metadata); + + data_destroy(serialized); + chunk_destroy(deserialized); + chunk_destroy(chunk); +} + +/* A special entry carrying an out-of-range rdev is a malformed chunk and must be + * rejected at deserialize (bounded by the same 0xffff / 0x00ffffff limits + * file_special_rdev_valid uses on the per-file wire), not deferred to the + * creation site. */ +static void test_chunk_special_rdev_out_of_range_rejected() { + const char* path = "temp_chunk_special_bad_rdev"; + unlink(path); + File* special = file_create(path); + EXPECT_NOT_NULL(special); + special->is_special = true; + special->rdev_major = 0x10000; /* > 0xffff */ + special->rdev_minor = 3; + struct stat st; + EXPECT_EQ_INT(stat("/dev/null", &st), 0); + special->metadata = file_metadata_create(path, &st, false, false); + EXPECT_NOT_NULL(special->metadata); + + File* files[1] = {special}; + Chunk* chunk = chunk_create(files, 1); + EXPECT_NOT_NULL(chunk); + Data* serialized = chunk_serialize(chunk, true); + EXPECT_NOT_NULL(serialized); + const Chunk* deserialized = chunk_deserialize(serialized, true); + EXPECT_NULL(deserialized); + data_destroy(serialized); + chunk_destroy(chunk); +} + void test_chunk() { test_file_operations(); test_chunk_operations(); + test_chunk_dir_entry_roundtrip(); + test_chunk_symlink_roundtrip(); + test_chunk_special_rdev_roundtrip(); + test_chunk_special_rdev_out_of_range_rejected(); } diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c index ab44023..44d6de7 100644 --- a/tests/test_client_cli.c +++ b/tests/test_client_cli.c @@ -1,11 +1,127 @@ #include "test_client_cli.h" +#include "checksum.h" +#include "client_validation.h" +#include "chmod.h" #include "config.h" +#include "delta.h" +#include "file_list.h" +#include "log.h" #include "test_utils.h" #include "utils.h" +#include +#include +#include #include #include #include +/* Declaration of parse_args from client_cli.c */ +int parse_args(Config* config, int argc, char* argv[], int* positional_args, int* positional_count); + +static Config* valid_client_config() { + Config* cfg = config_create(); + if (!cfg) + return NULL; + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + return cfg; +} + +static void test_validate_config_required_paths() { + Config* cfg = config_create(); + EXPECT_FALSE(validate_config(cfg)); + cfg->send_directory = str_dup("/src"); + EXPECT_FALSE(validate_config(cfg)); + config_delete(cfg); +} + +static void test_validate_config_incompatible_options() { + Config* cfg = valid_client_config(); + cfg->use_sendfile = true; + cfg->use_compression = true; + EXPECT_FALSE(validate_config(cfg)); + cfg->use_compression = false; + cfg->use_incremental = true; + cfg->use_chunk_serialization = true; + EXPECT_FALSE(validate_config(cfg)); + cfg->use_incremental = false; + cfg->skip_compress_set = true; + EXPECT_FALSE(validate_config(cfg)); + cfg->skip_compress_set = false; + cfg->compression_threads = 2; + EXPECT_FALSE(validate_config(cfg)); + cfg->use_compression = true; + cfg->use_sendfile = false; + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); +} + +static void test_validate_config_tls_requirements() { + Config* cfg = valid_client_config(); + cfg->use_tls = true; + EXPECT_FALSE(validate_config(cfg)); + cfg->tls_cert = str_dup("cert.pem"); + EXPECT_FALSE(validate_config(cfg)); + cfg->tls_key = str_dup("key.pem"); + EXPECT_FALSE(validate_config(cfg)); + cfg->tls_ca = str_dup("ca.pem"); + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); +} + +/* A7-3/S1: --password-file sends daemon credentials, so it is only allowed + over TLS (which itself mandates a verified --cert/--key/--ca) or to a + loopback destination. A remote plaintext daemon is refused up front. */ +static void test_validate_config_credentials_require_tls_or_loopback() { + /* Default host is 127.0.0.1 (loopback), so plaintext credentials are fine. */ + Config* cfg = valid_client_config(); + cfg->password_file = str_dup("creds.pw"); + EXPECT_TRUE(validate_config(cfg)); + + /* localhost is loopback too. */ + free(cfg->server_host); + cfg->server_host = str_dup("localhost"); + EXPECT_TRUE(validate_config(cfg)); + + /* A clearly remote host over plaintext is refused before any network I/O. */ + free(cfg->server_host); + cfg->server_host = str_dup("192.0.2.1"); + EXPECT_FALSE(validate_config(cfg)); + + /* TLS makes the remote destination acceptable (cert/key/ca are required). */ + cfg->use_tls = true; + EXPECT_FALSE(validate_config(cfg)); + cfg->tls_cert = str_dup("cert.pem"); + cfg->tls_key = str_dup("key.pem"); + cfg->tls_ca = str_dup("ca.pem"); + EXPECT_TRUE(validate_config(cfg)); + + /* No credentials: the remote plaintext rule does not apply. */ + cfg->use_tls = false; + char* creds = cfg->password_file; + cfg->password_file = NULL; + EXPECT_TRUE(validate_config(cfg)); + cfg->password_file = creds; + config_delete(cfg); +} + +static void test_validate_config_delta_sendfile_constraints() { + Config* cfg = valid_client_config(); + cfg->use_delta = true; + EXPECT_FALSE(validate_config(cfg)); + cfg->use_incremental = true; + cfg->use_sendfile = true; + EXPECT_FALSE(validate_config(cfg)); + + /* Whole-file makes delta selection inactive, so these combinations are valid. */ + cfg->whole_file = true; + EXPECT_TRUE(validate_config(cfg)); + + cfg->use_incremental = false; + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); +} + /* Test main() with --help flag (early return path, no server connection needed) */ static void test_cli_help() { /* We can't easily call main() because it calls send_files which needs a server. @@ -24,19 +140,24 @@ static void test_cli_help() { config_delete(cfg); } -/* Test that --archive sets compression, multithreading, and metadata */ +/* Test that the -a short spelling applies --archive's config bundle, matching + * rsync -rlptgoD semantics: links + metadata + devices + specials, and NOT + * compression/multithreading. (--archive itself is covered by + * test_parse_args_archive; this guards the short alias.) */ static void test_cli_archive_flags() { Config* cfg = config_create(); EXPECT_NOT_NULL(cfg); + char* argv[] = {"fastsync", "-a", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; - /* Simulate --archive flag */ - cfg->use_compression = true; - cfg->use_multithreading = true; - cfg->use_metadata = true; - - EXPECT_TRUE(cfg->use_compression); - EXPECT_TRUE(cfg->use_multithreading); + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->follow_symlinks); EXPECT_TRUE(cfg->use_metadata); + EXPECT_TRUE(cfg->preserve_devices); + EXPECT_TRUE(cfg->preserve_specials); + EXPECT_FALSE(cfg->use_compression); + EXPECT_FALSE(cfg->use_multithreading); config_delete(cfg); } @@ -51,6 +172,27 @@ static void test_cli_dry_run() { config_delete(cfg); } +static void test_cli_remove_source_files() { + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + EXPECT_FALSE(cfg->remove_source_files); + cfg->remove_source_files = true; + EXPECT_TRUE(cfg->remove_source_files); + config_delete(cfg); +} + +static void test_parse_args_remove_source_files() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--remove-source-files", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + int ret = parse_args(cfg, 4, argv, positional_args, &positional_count); + EXPECT_EQ_INT(ret, 0); + EXPECT_TRUE(cfg->remove_source_files); + config_delete(cfg); +} + /* Test that --delete sets use_delete */ static void test_cli_delete_flag() { Config* cfg = config_create(); @@ -80,10 +222,3060 @@ static void test_cli_exclude_patterns() { config_delete(cfg); } +/* Test parse_args with --help returns 1 (clean exit) */ +static void test_parse_args_help() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--help"}; + int positional_args[2]; + int positional_count = 0; + + int ret = parse_args(cfg, 2, argv, positional_args, &positional_count); + EXPECT_EQ_INT(ret, 1); + + config_delete(cfg); +} + +/* Test parse_args with -V/--version returns 1 */ +static void test_parse_args_version() { + Config* cfg = config_create(); + char* argv_short[] = {"fastsync", "-V"}; + char* argv_long[] = {"fastsync", "--version"}; + int positional_args[2]; + int positional_count = 0; + + int ret = parse_args(cfg, 2, argv_short, positional_args, &positional_count); + EXPECT_EQ_INT(ret, 1); + + ret = parse_args(cfg, 2, argv_long, positional_args, &positional_count); + EXPECT_EQ_INT(ret, 1); + + config_delete(cfg); +} + +/* --protocol=NUM forces the wire protocol version: the current PROTOCOL_VERSION + * is accepted (stored into config->version, which the config frame transmits), + * and any other value is rejected. Client-only: no server-side flag exists. */ +static void test_parse_args_protocol_accept_current() { + Config* cfg = valid_client_config(); + EXPECT_NOT_NULL(cfg); + char* argv_equals[] = {"fastsync", "--source-dir", "/src", + "--dest-dir", "/dst", "--protocol=2.19.0"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 6, argv_equals, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->version, PROTOCOL_VERSION); + config_delete(cfg); + + cfg = valid_client_config(); + EXPECT_NOT_NULL(cfg); + char* argv_space[] = {"fastsync", "--source-dir", "/src", "--dest-dir", + "/dst", "--protocol", "2.19.0"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 7, argv_space, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->version, PROTOCOL_VERSION); + config_delete(cfg); +} + +/* Any --protocol value other than the current PROTOCOL_VERSION must end in + * failure (parse_args simply stores it; validate_config rejects it up front). */ +static void test_parse_args_protocol_rejects_other_versions() { + static const char* const bad_versions[] = {"2.17", "2.16", "2.15.0", "2.16.0", "2.17.0", + "2.18.0", "216", "31", "abc", ""}; + for (size_t i = 0; i < sizeof(bad_versions) / sizeof(bad_versions[0]); i++) { + Config* cfg = valid_client_config(); + EXPECT_NOT_NULL(cfg); + char arg[64]; + snprintf(arg, sizeof(arg), "--protocol=%s", bad_versions[i]); + char* argv[] = {"fastsync", "--source-dir", "/src", "--dest-dir", "/dst", arg}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 6, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(strcmp(cfg->version, PROTOCOL_VERSION) != 0); + EXPECT_FALSE(validate_config(cfg)); + config_delete(cfg); + } +} + +/* validate_config accepts the current PROTOCOL_VERSION (the default) and rejects + * a version that does not equal it -- the honest post-parse enforcement. */ +static void test_validate_config_protocol_version() { + Config* cfg = valid_client_config(); + EXPECT_NOT_NULL(cfg); + EXPECT_EQ_STR(cfg->version, PROTOCOL_VERSION); + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); + + cfg = valid_client_config(); + EXPECT_NOT_NULL(cfg); + free(cfg->version); + cfg->version = str_dup("2.15.0"); + EXPECT_NOT_NULL(cfg->version); + EXPECT_FALSE(validate_config(cfg)); + config_delete(cfg); +} + +/* --xattrs/-X and --acls/-A preserve per-file xattrs and both imply metadata + * transmission (the xattr block rides the metadata/per-file frame); each is + * individually negatable and the derived use_xattrs follows the flags. */ +static void test_parse_args_xattrs_acls() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "-X", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->preserve_xattrs); + EXPECT_FALSE(cfg->preserve_acls); + EXPECT_TRUE(cfg->use_xattrs); + EXPECT_TRUE(cfg->use_metadata); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv_long[] = {"fastsync", "--acls", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_long, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->preserve_acls); + EXPECT_TRUE(cfg->use_xattrs); + EXPECT_TRUE(cfg->use_metadata); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv_neg[] = {"fastsync", "-X", "-A", "--no-xattrs", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 6, argv_neg, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->preserve_xattrs); + EXPECT_TRUE(cfg->preserve_acls); + EXPECT_TRUE(cfg->use_xattrs); + config_delete(cfg); +} + +/* --fake-super is a receiver-side preference that parks the source + * uid/gid/mode/mtime in a reserved xattr; it implies metadata transmission. */ +static void test_parse_args_fake_super() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--fake-super", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->fake_super); + EXPECT_TRUE(cfg->use_metadata); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv_neg[] = {"fastsync", "--fake-super", "--no-fake-super", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv_neg, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->fake_super); + config_delete(cfg); +} + +/* P7 Wave E: --super / --no-super set the receiver-side privilege tri-state + * (they take no argument). The default is AUTO, the last of either flag wins, + * and a malformed inline value ("--super=x") is rejected rather than silently + * treated as --super. */ +static void test_parse_args_super() { + Config* cfg = config_create(); + EXPECT_EQ_INT(cfg->super_mode, SUPER_MODE_AUTO); + char* argv_on[] = {"fastsync", "--super", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_on, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->super_mode, SUPER_MODE_ON); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv_off[] = {"fastsync", "--no-super", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_off, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->super_mode, SUPER_MODE_OFF); + config_delete(cfg); + + /* Tri-state, not a boolean pair: the last flag wins. */ + cfg = config_create(); + positional_count = 0; + char* argv_both[] = {"fastsync", "--super", "--no-super", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv_both, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->super_mode, SUPER_MODE_OFF); + config_delete(cfg); + + /* A malformed inline value is a hard unknown-option error. */ + cfg = config_create(); + positional_count = 0; + char* argv_bad[] = {"fastsync", "--super=x", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_bad, positional_args, &positional_count), -1); + config_delete(cfg); +} + +/* Test parse_args with valid SSH port (long form; -p is now rsync --perms) */ +static void test_parse_args_valid_port() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--ssh-port", "2222", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + int ret = parse_args(cfg, 5, argv, positional_args, &positional_count); + EXPECT_EQ_INT(ret, 0); + EXPECT_EQ_INT(cfg->ssh_port, 2222); + EXPECT_EQ_INT(positional_count, 2); + + config_delete(cfg); +} + +/* Test parse_args with --size-only. */ +static void test_parse_args_size_only() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--size-only", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->size_only); + EXPECT_EQ_INT(positional_count, 2); + + config_delete(cfg); +} + +static void test_parse_args_ignore_existing() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--ignore-existing", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + int ret = parse_args(cfg, 4, argv, positional_args, &positional_count); + EXPECT_EQ_INT(ret, 0); + EXPECT_TRUE(cfg->ignore_existing); + + config_delete(cfg); +} + +static void test_parse_args_executability() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "-E", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_executability); + EXPECT_TRUE(cfg->use_metadata); + + config_delete(cfg); +} + +static void test_parse_args_chmod() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--chmod=u=rw,go=r", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->chmod_spec, "u=rw,go=r"); + EXPECT_TRUE(cfg->use_metadata); + mode_t result; + EXPECT_TRUE(chmod_apply(0777, cfg->chmod_spec, &result)); + EXPECT_EQ_INT(result, 0644); + config_delete(cfg); +} + +static void test_parse_args_numeric_chmod() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--chmod", "7777", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->chmod_spec, "7777"); + EXPECT_TRUE(cfg->use_metadata); + config_delete(cfg); +} + +static void test_parse_args_rejects_invalid_chmod() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--chmod=a+X", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), -1); + config_delete(cfg); +} + +/* Test parse_args rejects port > 65535 */ +static void test_parse_args_invalid_port() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--ssh-port", "99999", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + int ret = parse_args(cfg, 5, argv, positional_args, &positional_count); + EXPECT_EQ_INT(ret, -1); + + config_delete(cfg); +} + +/* Test parse_args rejects non-numeric port */ +static void test_parse_args_non_numeric_port() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--ssh-port", "abc", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + int ret = parse_args(cfg, 5, argv, positional_args, &positional_count); + EXPECT_EQ_INT(ret, -1); + + config_delete(cfg); +} + +/* Test parse_args rejects server port > 65535 */ +static void test_parse_args_invalid_server_port() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--server-port", "70000", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + int ret = parse_args(cfg, 5, argv, positional_args, &positional_count); + EXPECT_EQ_INT(ret, -1); + + config_delete(cfg); +} + +/* Test parse_args rejects invalid compression level (-z/--compress) */ +static void test_parse_args_invalid_compression_level() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "-z", "25", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + int ret = parse_args(cfg, 5, argv, positional_args, &positional_count); + EXPECT_EQ_INT(ret, -1); + + config_delete(cfg); +} + +/* Test parse_args accepts valid compression level (-z/--compress) */ +static void test_parse_args_valid_compression_level() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "-z", "10", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + int ret = parse_args(cfg, 5, argv, positional_args, &positional_count); + EXPECT_EQ_INT(ret, 0); + EXPECT_EQ_INT(cfg->compression_level, 10); + + config_delete(cfg); +} + +static void test_parse_args_debug_flags() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--debug=io,proto,pack,util", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->debug_level, LOG_DEBUG_ALL); + EXPECT_EQ_INT(get_log_debug_flags(), LOG_DEBUG_ALL); + config_delete(cfg); +} + +static void test_parse_args_debug_help() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--debug=help"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 2, argv, positional_args, &positional_count), 1); + config_delete(cfg); +} + +static void test_parse_args_debug_flags_validation() { + static const char* const values[] = {"", "io,", ",io", "io,,proto", "acl", "tls", "unknown"}; + for (size_t i = 0; i < sizeof(values) / sizeof(values[0]); i++) { + Config* cfg = config_create(); + char option[64]; + snprintf(option, sizeof(option), "--debug=%s", values[i]); + char* argv[] = {"fastsync", option, "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), -1); + config_delete(cfg); + } +} + +static void test_parse_args_modify_window() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--modify-window=3", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->modify_window, 3); + config_delete(cfg); + + cfg = config_create(); + char* short_argv[] = {"fastsync", "-@", "7", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, short_argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->modify_window, 7); + config_delete(cfg); + + cfg = config_create(); + char* attached_argv[] = {"fastsync", "-@11", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, attached_argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->modify_window, 11); + config_delete(cfg); +} + +static void test_parse_args_rejects_invalid_modify_window() { + const char* values[] = {"-1", "not-a-number", ""}; + for (size_t i = 0; i < sizeof(values) / sizeof(values[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--modify-window", (char*)values[i], "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1); + config_delete(cfg); + } +} + +static void test_parse_args_max_alloc_sizes() { + const char* values[] = {"1", "4K", "2m", "3G", "1T", "1P", "1E", "512B"}; + const unsigned long long expected[] = {1, + 4ULL * 1024, + 2ULL * 1024 * 1024, + 3ULL * 1024 * 1024 * 1024, + 1ULL * 1024 * 1024 * 1024 * 1024, + 1ULL * 1024 * 1024 * 1024 * 1024 * 1024, + 1ULL * 1024 * 1024 * 1024 * 1024 * 1024 * 1024, + 512}; + for (size_t i = 0; i < sizeof(values) / sizeof(values[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--max-alloc", (char*)values[i], "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->max_alloc == expected[i]); + config_delete(cfg); + } + + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--max-alloc=8M", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->max_alloc == 8ULL * 1024 * 1024); + config_delete(cfg); +} + +static void test_parse_args_rejects_invalid_max_alloc() { + const char* values[] = {"0", "-1", "+1", " 1", "1 ", + "1Z", "1K2", "1 K", "1\tK", "18446744073709551615K"}; + for (size_t i = 0; i < sizeof(values) / sizeof(values[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--max-alloc", (char*)values[i], "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1); + config_delete(cfg); + } +} + +static void test_parse_args_skip_compress() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--skip-compress=.ZIP, .GZ", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->skip_compress_set); + EXPECT_EQ_INT(cfg->skip_compress_count, 2); + EXPECT_EQ_STR(cfg->skip_compress_suffixes[0], ".ZIP"); + EXPECT_EQ_STR(cfg->skip_compress_suffixes[1], ".GZ"); + config_delete(cfg); +} + +static void test_parse_args_empty_skip_compress() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--skip-compress=", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->skip_compress_set); + EXPECT_EQ_INT(cfg->skip_compress_count, 0); + config_delete(cfg); +} + +static void test_parse_args_compression_threads() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--compress-threads", "4", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->compression_threads, 4); + config_delete(cfg); + + cfg = config_create(); + char* equals_argv[] = {"fastsync", "--compress-threads=3", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, equals_argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->compression_threads, 3); + config_delete(cfg); + + cfg = config_create(); + char* invalid_argv[] = {"fastsync", "--compress-threads=0", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, invalid_argv, positional_args, &positional_count), -1); + config_delete(cfg); + + cfg = config_create(); + char* excessive_argv[] = {"fastsync", "--compress-threads=65", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, excessive_argv, positional_args, &positional_count), -1); + config_delete(cfg); +} + +/* Test parse_args unknown option returns error */ +static void test_parse_args_unknown_option() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--nonexistent", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + int ret = parse_args(cfg, 4, argv, positional_args, &positional_count); + EXPECT_EQ_INT(ret, -1); + + config_delete(cfg); +} + +/* -d/--dirs and the rsync --old-dirs/--old-d aliases all enable directory-only + * transfers (--dirs maps every spelling onto the same config field). */ +static void test_parse_args_dirs_aliases() { + static const char* const options[] = {"--dirs", "-d", "--old-dirs", "--old-d"}; + for (size_t i = 0; i < sizeof(options) / sizeof(options[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", (char*)options[i], "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->dirs); + config_delete(cfg); + } +} + +/* -R/--relative, --no-implied-dirs and --mkpath are plain boolean flags. */ +static void test_parse_args_relative_no_implied_mkpath() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "-R", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->relative); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* long_argv[] = {"fastsync", "--relative", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, long_argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->relative); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* noimplied_argv[] = {"fastsync", "--no-implied-dirs", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, noimplied_argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->no_implied_dirs); + EXPECT_FALSE(cfg->relative); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* mkpath_argv[] = {"fastsync", "--mkpath", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, mkpath_argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->mkpath); + config_delete(cfg); +} + +/* Parse --compare-dest/--copy-dest/--link-dest, including the =value and + separate-argument forms, and verify the ordered (repeatable) basis list. */ +static void test_parse_args_basis_dirs() { + Config* cfg = config_create(); + int positional_args[2]; + int positional_count = 0; + char* argv[] = {"fastsync", "--link-dest=prior", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(config_has_basis(cfg)); + EXPECT_EQ_INT(cfg->basis_count, 1); + EXPECT_EQ_INT(cfg->basis_dirs[0].type, BASIS_DEST_LINK); + EXPECT_EQ_STR(cfg->basis_dirs[0].path, "prior"); + /* Basis dirs are honored by the receiver-side per-file check, so they imply + --incremental (and, unless disabled, metadata) on the sender. */ + EXPECT_TRUE(cfg->use_incremental); + EXPECT_TRUE(cfg->use_metadata); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv2[] = {"fastsync", "--compare-dest", "cmp", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv2, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->basis_count, 1); + EXPECT_EQ_INT(cfg->basis_dirs[0].type, BASIS_DEST_COMPARE); + EXPECT_EQ_STR(cfg->basis_dirs[0].path, "cmp"); + config_delete(cfg); + + /* Repetition is supported: entries keep command-line order and type. */ + cfg = config_create(); + positional_count = 0; + char* argv3[] = {"fastsync", "--link-dest=a", "--compare-dest=b", + "--link-dest=c", "--copy-dest=d", "/src", + "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 7, argv3, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->basis_count, 4); + EXPECT_EQ_INT(cfg->basis_dirs[0].type, BASIS_DEST_LINK); + EXPECT_EQ_STR(cfg->basis_dirs[0].path, "a"); + EXPECT_EQ_INT(cfg->basis_dirs[1].type, BASIS_DEST_COMPARE); + EXPECT_EQ_STR(cfg->basis_dirs[1].path, "b"); + EXPECT_EQ_INT(cfg->basis_dirs[2].type, BASIS_DEST_LINK); + EXPECT_EQ_STR(cfg->basis_dirs[2].path, "c"); + EXPECT_EQ_INT(cfg->basis_dirs[3].type, BASIS_DEST_COPY); + EXPECT_EQ_STR(cfg->basis_dirs[3].path, "d"); + config_delete(cfg); + + /* Nested relative basis dirs are allowed (they resolve below the root). */ + cfg = config_create(); + positional_count = 0; + char* argv4[] = {"fastsync", "--copy-dest=snap/2026-01", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv4, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->basis_count, 1); + EXPECT_EQ_STR(cfg->basis_dirs[0].path, "snap/2026-01"); + config_delete(cfg); +} + +/* Absolute, escaping, or degenerate basis-dir values must be rejected up + front: they would resolve outside the destination root on the receiver. */ +static void test_parse_args_basis_invalid_paths() { + static const char* const invalid[] = {"/abs", "..", "a/../b", "."}; + for (size_t i = 0; i < sizeof(invalid) / sizeof(invalid[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--link-dest", (char*)invalid[i], "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1); + config_delete(cfg); + } +} + +/* Basis dirs require the per-file incremental handshake, which -s disables. */ +static void test_validate_config_basis_rejects_chunk_serialization() { + Config* cfg = valid_client_config(); + EXPECT_EQ_INT(config_basis_append(cfg, BASIS_DEST_LINK, "prior"), 0); + cfg->use_chunk_serialization = true; + EXPECT_FALSE(validate_config(cfg)); + cfg->use_chunk_serialization = false; + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); +} + +/* --del is accepted as the rsync alias for --delete-during: it enables + * deletion with the during (early) timing. */ +static void test_parse_args_delete_during_alias() { + static const char* const options[] = {"--del", "--delete-during"}; + + for (size_t i = 0; i < sizeof(options) / sizeof(options[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", (char*)options[i], "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_delete); + EXPECT_TRUE(cfg->delete_during); + EXPECT_FALSE(cfg->delete_before); + EXPECT_FALSE(cfg->delete_delay); + EXPECT_FALSE(cfg->delete_after); + config_delete(cfg); + } +} + +/* Each rsync deletion-timing flag is accepted and implies --delete. */ +static void test_parse_args_delete_timing_flags() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--delete-before", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_delete); + EXPECT_TRUE(cfg->delete_before); + config_delete(cfg); + + cfg = config_create(); + char* argv_after[] = {"fastsync", "--delete-after", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_after, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_delete); + EXPECT_TRUE(cfg->delete_after); + EXPECT_FALSE(cfg->delete_before); + config_delete(cfg); + + cfg = config_create(); + char* argv_delay[] = {"fastsync", "--delete-delay", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_delay, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_delete); + EXPECT_TRUE(cfg->delete_delay); + EXPECT_FALSE(cfg->delete_before); + EXPECT_FALSE(cfg->delete_after); + config_delete(cfg); +} + +/* Two different delete-timing flags on one command line are a conflict, not a + * silent last-one-wins choice. */ +static void test_parse_args_delete_timing_conflict_rejected() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--delete-before", "--delete-after", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_delete); + EXPECT_FALSE(validate_config(cfg)); + config_delete(cfg); + + cfg = config_create(); + char* argv2[] = {"fastsync", "--delete-during", "--delete-delay", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv2, positional_args, &positional_count), 0); + EXPECT_FALSE(validate_config(cfg)); + config_delete(cfg); +} + +/* A timing flag whose --delete was then negated away must be rejected: timing + * without deletion is meaningless. */ +static void test_parse_args_delete_timing_without_delete_rejected() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--delete-before", "--no-delete", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->use_delete); + EXPECT_TRUE(cfg->delete_before); + EXPECT_FALSE(validate_config(cfg)); + config_delete(cfg); +} + +/* Parsed-but-unimplemented options must fail instead of being silently accepted. */ +static void test_parse_args_rejects_unimplemented_options() { + static const char* const options[] = {"--silent", + "--queue-size", + "-A", + "--acls", + "-X", + "--xattrs", + "-D", + "--devices", + "--delete-excluded", + "--max-delete", + "--prune-empty-dirs", + "--bind-address", + "--daemon", + "--config", + "--server"}; + + for (size_t i = 0; i < sizeof(options) / sizeof(options[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", (char*)options[i], "dummy", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1); + config_delete(cfg); + } +} + +/* Test both rsync-compatible quiet spellings and option ordering. */ +static void test_parse_args_quiet() { + static const char* const options[][2] = { + {"-q", "-v"}, {"-v", "-q"}, {"--quiet", "-v"}, {"-v", "--quiet"}}; + for (size_t i = 0; i < sizeof(options) / sizeof(options[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", (char*)options[i][0], (char*)options[i][1], "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->quiet); + config_delete(cfg); + } +} + +static void test_parse_args_human_readable() { + static const char* const options[] = {"-h", "--human-readable"}; + for (size_t i = 0; i < sizeof(options) / sizeof(options[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", (char*)options[i], "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->human_readable); + config_delete(cfg); + } +} + +static void test_parse_args_update() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "-u", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->update); + EXPECT_TRUE(cfg->use_metadata); + EXPECT_EQ_INT(positional_count, 2); + + config_delete(cfg); +} + +static void test_parse_args_hard_links() { + Config* cfg = config_create(); + char* argv_H[] = {"fastsync", "-H", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_H, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->preserve_hard_links); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv_long[] = {"fastsync", "--hard-links", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_long, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->preserve_hard_links); + config_delete(cfg); + + /* --no-hard-links clears the flag. */ + cfg = config_create(); + positional_count = 0; + char* argv_neg[] = {"fastsync", "--hard-links", "--no-hard-links", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv_neg, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->preserve_hard_links); + config_delete(cfg); +} + +/* -H/--hard-links violates the per-file streaming requirement of + * --chunk-serialization and the payload-bearing tail-resume of --append: both + * combos are rejected up front. */ +static void test_validate_config_hard_links_incompatible_modes() { + Config* cfg = config_create(); + char* argv_s[] = {"fastsync", "-H", "--chunk-serialization", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv_s, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->preserve_hard_links); + EXPECT_FALSE(validate_config(cfg)); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv_append[] = {"fastsync", "-H", "--append", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv_append, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->preserve_hard_links); + EXPECT_FALSE(validate_config(cfg)); + config_delete(cfg); +} + +static void test_parse_args_info_flags() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--info=copy,skip", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->info_level, LOG_INFO_COPY | LOG_INFO_SKIP); + EXPECT_EQ_INT(get_log_info_flags(), LOG_INFO_COPY | LOG_INFO_SKIP); + config_delete(cfg); +} + +static void test_parse_args_info_verbose_order() { + char* argv_info_first[] = {"fastsync", "--info=none", "--verbose", "/src", "/dst"}; + char* argv_verbose_first[] = {"fastsync", "--verbose", "--info=none", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + Config* cfg = config_create(); + EXPECT_EQ_INT(parse_args(cfg, 5, argv_info_first, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->info_level, 0); + EXPECT_EQ_INT(get_log_info_flags(), 0); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv_verbose_first, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->info_level, 0); + EXPECT_EQ_INT(get_log_info_flags(), 0); + config_delete(cfg); +} + +static void test_parse_args_rejects_invalid_info_flag() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--info=copy,unknown", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), -1); + config_delete(cfg); +} + +/* Test parse_args with --archive flag */ +static void test_parse_args_archive() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--archive", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + int ret = parse_args(cfg, 4, argv, positional_args, &positional_count); + EXPECT_EQ_INT(ret, 0); + EXPECT_TRUE(cfg->follow_symlinks); + EXPECT_TRUE(cfg->use_metadata); + EXPECT_TRUE(cfg->preserve_devices); + EXPECT_TRUE(cfg->preserve_specials); + EXPECT_FALSE(cfg->use_compression); + EXPECT_FALSE(cfg->use_multithreading); + + config_delete(cfg); +} + +/* Negations must override archive's implied options in argument order. Note: + * archive implies devices+specials, and device/special preservation itself + * forces metadata transmission (re-creating a node needs the metadata mode), so + * --no-preserve cannot turn metadata back off while archive keeps devices/specials + * on -- that is the correct interaction, not a bug. A link negation does work. */ +static void test_parse_args_negations() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--archive", "--no-links", "--no-preserve", + "--no-dry-run", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 7, argv, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->follow_symlinks); + EXPECT_TRUE(cfg->use_metadata); + EXPECT_FALSE(cfg->dry_run); + EXPECT_EQ_INT(positional_count, 2); + config_delete(cfg); +} + +/* --no-preserve negates an explicit --preserve when nothing forces metadata back + * on (no devices/specials). */ +static void test_parse_args_negate_preserve_without_devices() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--preserve", "--no-preserve", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->use_metadata); + config_delete(cfg); +} + +static void test_parse_args_negation_order() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--no-z", "-z", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_compression); + config_delete(cfg); +} + +static void test_parse_args_no_preserve_blocks_implicit_metadata() { + static const char* const options[][3] = { + {"--incremental", "--no-preserve", "/src"}, + {"--no-preserve", "--incremental", "/src"}, + {"--delta", "--no-preserve", "/src"}, + {"--no-preserve", "--delta", "/src"}, + }; + + for (size_t i = 0; i < sizeof(options) / sizeof(options[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", (char*)options[i][0], (char*)options[i][1], (char*)options[i][2], + "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->use_metadata); + EXPECT_TRUE(cfg->metadata_explicitly_disabled); + config_delete(cfg); + } +} + +/* --checksum-choice and its --cc alias select the whole-file digest algorithm + (default xxh64; both "xxh64" and the rsync "xxhash" spelling accepted). */ +static void test_parse_args_checksum_choice_aliases() { + static const char* const options[] = {"--checksum-choice", "--cc"}; + + for (size_t i = 0; i < sizeof(options) / sizeof(options[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", (char*)options[i], "xxh64", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->checksum_algo, (int)CHECKSUM_ALGO_XXH64); + config_delete(cfg); + } +} + +/* Both the "--checksum-choice=ALG" and "--cc=ALG" inline forms parse. */ +static void test_parse_args_checksum_choice_equals_forms() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--checksum-choice=md5", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->checksum_algo, (int)CHECKSUM_ALGO_MD5); + config_delete(cfg); + + cfg = config_create(); + char* argv2[] = {"fastsync", "--cc=xxhash", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv2, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->checksum_algo, (int)CHECKSUM_ALGO_XXH64); + config_delete(cfg); +} + +/* An algorithm FastSync does not support must be rejected, never a silent + no-op. */ +static void test_parse_args_checksum_choice_rejects_unsupported() { + static const char* const bad[] = {"md4", "sha256", "crc32", "none", "bogus"}; + for (size_t i = 0; i < sizeof(bad) / sizeof(bad[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--checksum-choice", (char*)bad[i], "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1); + config_delete(cfg); + } +} + +/* --checksum-seed parses as a 64-bit non-negative integer (space and = forms); + invalid values are rejected. */ +static void test_parse_args_checksum_seed() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--checksum-seed=42", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->checksum_seed == 42ULL); + config_delete(cfg); + + cfg = config_create(); + char* argv2[] = {"fastsync", "--checksum-seed", "12345", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv2, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->checksum_seed == 12345ULL); + config_delete(cfg); + + /* 0 is a valid (and default) seed. */ + cfg = config_create(); + char* argv3[] = {"fastsync", "--checksum-seed=0", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv3, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->checksum_seed == 0ULL); + config_delete(cfg); + + /* Non-numeric and negative seeds are rejected. */ + static const char* const bad[] = {"abc", "-5", "1.5", ""}; + for (size_t i = 0; i < sizeof(bad) / sizeof(bad[0]); i++) { + cfg = config_create(); + char* argv4[] = {"fastsync", "--checksum-seed", (char*)bad[i], "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv4, positional_args, &positional_count), -1); + config_delete(cfg); + } + + /* Missing value is rejected. */ + cfg = config_create(); + char* argv5[] = {"fastsync", "--checksum-seed"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 2, argv5, positional_args, &positional_count), -1); + config_delete(cfg); +} + +static void test_parse_args_rejects_unsafe_negation() { + static const char* const options[] = {"--no-archive", "--no-timeout", "--no-unknown"}; + for (size_t i = 0; i < sizeof(options) / sizeof(options[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", (char*)options[i], "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), -1); + config_delete(cfg); + } +} + +/* Both checksum-choice spellings require a value. */ +static void test_parse_args_checksum_choice_requires_value() { + static const char* const options[] = {"--checksum-choice", "--cc"}; + + for (size_t i = 0; i < sizeof(options) / sizeof(options[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", (char*)options[i]}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 2, argv, positional_args, &positional_count), -1); + config_delete(cfg); + } +} + +/* --temp-dir accepts both the "--temp-dir=DIR" and "--temp-dir DIR" forms. */ +static void test_parse_args_temp_dir() { + Config* cfg = config_create(); + char* equals_argv[] = {"fastsync", "--temp-dir=scratch", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, equals_argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->temp_dir, "scratch"); + EXPECT_EQ_INT(positional_count, 2); + config_delete(cfg); + + cfg = config_create(); + char* space_argv[] = {"fastsync", "--temp-dir", "scratch/sub", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, space_argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->temp_dir, "scratch/sub"); + EXPECT_EQ_INT(positional_count, 2); + config_delete(cfg); + + /* A value-taking option may not be passed without a value. */ + cfg = config_create(); + char* missing_argv[] = {"fastsync", "--temp-dir"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 2, missing_argv, positional_args, &positional_count), -1); + config_delete(cfg); +} + +static void test_parse_args_old_args() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--old-args", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->old_args); + config_delete(cfg); +} + +/* Phase 5 connectivity: -e/--rsh select the remote-shell program. Both the + * short (space-separated value) and long (=value and space) forms parse, and + * a multi-word command line is preserved verbatim for the transport layer. */ +static void test_parse_args_rsh() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "-e", "ssh -p 2222", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->rsh_command, "ssh -p 2222"); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv_eq[] = {"fastsync", "--rsh=customsh", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_eq, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->rsh_command, "customsh"); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv_space[] = {"fastsync", "--rsh", "ssh -l bob", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv_space, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->rsh_command, "ssh -l bob"); + config_delete(cfg); + + /* A missing value is a hard error. */ + cfg = config_create(); + positional_count = 0; + char* argv_missing[] = {"fastsync", "-e"}; + EXPECT_EQ_INT(parse_args(cfg, 2, argv_missing, positional_args, &positional_count), -1); + config_delete(cfg); +} + +/* --rsync-path is rsync's spelling for the server program path: it aliases + * fastsync_server_path exactly like --fastsync-server-path. */ +static void test_parse_args_rsync_path_alias() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--rsync-path", "/usr/bin/fastsync-server", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->fastsync_server_path, "/usr/bin/fastsync-server"); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv_eq[] = {"fastsync", "--rsync-path=/opt/bin/srv", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_eq, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->fastsync_server_path, "/opt/bin/srv"); + config_delete(cfg); +} + +/* --blocking-io is a plain boolean flag that leaves the SSH socket with no + * timeouts; the default is off. */ +static void test_parse_args_blocking_io() { + Config* cfg = config_create(); + EXPECT_FALSE(cfg->blocking_io); + char* argv[] = {"fastsync", "--blocking-io", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->blocking_io); + config_delete(cfg); +} + +/* --outbuf=N|L|B maps onto the OUTBUF_* modes (default: block). Garbage is + * rejected, never silently coerced. */ +static void test_parse_args_outbuf() { + Config* cfg = config_create(); + EXPECT_EQ_INT(cfg->outbuf, OUTBUF_BLOCK); + int positional_args[2]; + int positional_count = 0; + + char* argv_n[] = {"fastsync", "--outbuf=N", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_n, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->outbuf, OUTBUF_NONE); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv_l[] = {"fastsync", "--outbuf", "L", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv_l, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->outbuf, OUTBUF_LINE); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv_b[] = {"fastsync", "--outbuf=b", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_b, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->outbuf, OUTBUF_BLOCK); + config_delete(cfg); + + static const char* const bad[] = {"G", "X", ""}; + for (size_t i = 0; i < sizeof(bad) / sizeof(bad[0]); i++) { + cfg = config_create(); + positional_count = 0; + char option[32]; + snprintf(option, sizeof(option), "--outbuf=%s", bad[i]); + char* argv_bad[] = {"fastsync", option, "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_bad, positional_args, &positional_count), -1); + config_delete(cfg); + } + + cfg = config_create(); + positional_count = 0; + char* argv_missing[] = {"fastsync", "--outbuf"}; + EXPECT_EQ_INT(parse_args(cfg, 2, argv_missing, positional_args, &positional_count), -1); + config_delete(cfg); +} + +static void test_parse_args_fsync() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--fsync", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_fsync); + config_delete(cfg); +} + +static void test_parse_args_existing() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--existing", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->existing); + config_delete(cfg); +} + +static void test_parse_args_ignore_times() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "-I", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->ignore_times); + config_delete(cfg); + + cfg = config_create(); + char* long_argv[] = {"fastsync", "--ignore-times", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, long_argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->ignore_times); + config_delete(cfg); +} + +/* --secluded-args is accepted for compatibility but has no effect. */ +static void test_parse_args_secluded_args() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--secluded-args", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->use_chunk_serialization); + config_delete(cfg); +} + +static void test_parse_args_chunk_serialization_long_form() { + Config* cfg = config_create(); + /* Chunk serialization is now long-form-only (the short -s is rsync's + * --secluded-args no-op). */ + char* argv[] = {"fastsync", "--chunk-serialization", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_chunk_serialization); + config_delete(cfg); +} + +/* Phase 4 symlink-trust flags: -k/--copy-dirlinks, -K/--keep-dirlinks and + --munge-links must parse into their Config fields. */ +static void test_parse_args_symlink_trust() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "-k", "-K", "--munge-links", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 6, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->copy_dirlinks); + EXPECT_TRUE(cfg->keep_dirlinks); + EXPECT_TRUE(cfg->munge_links); + config_delete(cfg); + + cfg = config_create(); + char* long_argv[] = {"fastsync", "--copy-dirlinks", "--keep-dirlinks", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, long_argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->copy_dirlinks); + EXPECT_TRUE(cfg->keep_dirlinks); + EXPECT_FALSE(cfg->munge_links); + config_delete(cfg); + + /* Without any of the flags they stay off (additive, opt-in). */ + cfg = config_create(); + char* plain_argv[] = {"fastsync", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 3, plain_argv, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->copy_dirlinks); + EXPECT_FALSE(cfg->keep_dirlinks); + EXPECT_FALSE(cfg->munge_links); + config_delete(cfg); +} + +static void test_parse_args_8_bit_output() { + Config* cfg = config_create(); + char* long_argv[] = {"fastsync", "--8-bit-output", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, long_argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->eight_bit_output); + config_delete(cfg); + + cfg = config_create(); + char* short_argv[] = {"fastsync", "-8", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, short_argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->eight_bit_output); + config_delete(cfg); +} + +static void test_parse_args_stderr_modes() { + static const char* const modes[] = {"errors", "all", "e", "a"}; + static const LogStderrMode expected[] = {LOG_STDERR_ERRORS, LOG_STDERR_ALL, LOG_STDERR_ERRORS, + LOG_STDERR_ALL}; + for (size_t i = 0; i < sizeof(modes) / sizeof(modes[0]); i++) { + Config* cfg = config_create(); + char option[32]; + snprintf(option, sizeof(option), "--stderr=%s", modes[i]); + char* argv[] = {"fastsync", option, "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(log_get_stderr_mode(), expected[i]); + config_delete(cfg); + } + log_set_stderr_mode(LOG_STDERR_ERRORS); +} + +static void test_parse_args_rejects_unsupported_stderr_modes() { + static const char* const modes[] = {"client", "c", "invalid"}; + for (size_t i = 0; i < sizeof(modes) / sizeof(modes[0]); i++) { + Config* cfg = config_create(); + char option[32]; + snprintf(option, sizeof(option), "--stderr=%s", modes[i]); + char* argv[] = {"fastsync", option, "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), -1); + config_delete(cfg); + } + log_set_stderr_mode(LOG_STDERR_ERRORS); +} + +/* Test both whole-file spellings and its precedence over delta selection. */ +static void test_parse_args_whole_file() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--delta", "--incremental", "-W", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 6, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->whole_file); + EXPECT_TRUE(cfg->use_delta); + + config_delete(cfg); + cfg = config_create(); + char* long_argv[] = {"fastsync", "--whole-file", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, long_argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->whole_file); + config_delete(cfg); +} + +/* -y/--fuzzy reuses a similar destination file as a delta basis, so it implies + * the receiver-driven delta path (--incremental + --delta): FastSync's delta + * machinery is OFF by default, so without the implication a bare --fuzzy would + * be a silent no-op. Both spellings behave identically. */ +static void test_parse_args_fuzzy_implies_delta() { + static const char* const spellings[] = {"--fuzzy", "-y"}; + for (size_t i = 0; i < sizeof(spellings) / sizeof(spellings[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", (char*)spellings[i], "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->fuzzy); + EXPECT_TRUE(cfg->use_incremental); + EXPECT_TRUE(cfg->use_delta); + EXPECT_TRUE(cfg->use_metadata); + config_delete(cfg); + } +} + +/* --no-fuzzy turns the flag back off; the incremental/delta implication must + * only fire when the FINAL value of the flag is true (order-independent). */ +static void test_parse_args_fuzzy_negation() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--fuzzy", "--no-fuzzy", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->fuzzy); + EXPECT_FALSE(cfg->use_delta); + EXPECT_FALSE(cfg->use_incremental); + config_delete(cfg); + + cfg = config_create(); + char* reordered[] = {"fastsync", "--no-fuzzy", "--fuzzy", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, reordered, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->fuzzy); + EXPECT_TRUE(cfg->use_incremental); + EXPECT_TRUE(cfg->use_delta); + config_delete(cfg); +} + +/* -W/--whole-file switches the delta machinery off, so --fuzzy is inert (the + * per-file quick check still needs --incremental, which stays implied). */ +static void test_parse_args_fuzzy_with_whole_file() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--fuzzy", "-W", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->fuzzy); + EXPECT_TRUE(cfg->whole_file); + EXPECT_TRUE(cfg->use_incremental); + EXPECT_FALSE(cfg->use_delta); + config_delete(cfg); +} + +/* An explicit --no-delta is respected by the --fuzzy implication in either + * argument order (a user who switched delta off does not want it forced on). */ +static void test_parse_args_fuzzy_respects_no_delta() { + static const char* const combos[][2] = { + {"--fuzzy", "--no-delta"}, + {"--no-delta", "--fuzzy"}, + }; + for (size_t i = 0; i < sizeof(combos) / sizeof(combos[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", (char*)combos[i][0], (char*)combos[i][1], "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->fuzzy); + EXPECT_FALSE(cfg->use_delta); + config_delete(cfg); + } +} + +/* An explicit --no-incremental is respected by the --fuzzy implication in + * either argument order (unlike the basis-dir options, --fuzzy does not force + * the incremental handshake back on). Because delta needs the handshake, the + * delta implication is suppressed too, so the run is a plain (default-mode) + * transfer rather than an invalid "--delta requires --incremental" config. */ +static void test_parse_args_fuzzy_respects_no_incremental() { + static const char* const combos[][2] = { + {"--fuzzy", "--no-incremental"}, + {"--no-incremental", "--fuzzy"}, + }; + for (size_t i = 0; i < sizeof(combos) / sizeof(combos[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", (char*)combos[i][0], (char*)combos[i][1], "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->fuzzy); + EXPECT_FALSE(cfg->use_incremental); + EXPECT_FALSE(cfg->use_delta); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); + } +} + +/* --fuzzy requires the delta machinery, which the chunk-serialization (-s) and + * sendfile (-f) modes reject -- mirroring the --delta constraint checks. */ +static void test_validate_config_fuzzy_incompatible_modes() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--fuzzy", "--chunk-serialization", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + EXPECT_TRUE(cfg->use_incremental); + EXPECT_FALSE(validate_config(cfg)); + config_delete(cfg); + + cfg = config_create(); + char* sendfile_argv[] = {"fastsync", "--fuzzy", "--sendfile", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, sendfile_argv, positional_args, &positional_count), 0); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + EXPECT_TRUE(cfg->use_delta); + EXPECT_FALSE(validate_config(cfg)); + config_delete(cfg); + + /* A plain --fuzzy run is a valid configuration. */ + cfg = config_create(); + char* ok_argv[] = {"fastsync", "--fuzzy", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, ok_argv, positional_args, &positional_count), 0); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + EXPECT_TRUE(cfg->use_delta); + EXPECT_TRUE(cfg->use_incremental); + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); +} + +/* -x and --one-file-system enable client-side single-filesystem scanning. */ +static void test_parse_args_one_file_system() { + Config* cfg = config_create(); + EXPECT_FALSE(cfg->one_file_system); + + char* argv[] = {"fastsync", "-x", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->one_file_system); + EXPECT_EQ_INT(positional_count, 2); + + config_delete(cfg); + cfg = config_create(); + char* long_argv[] = {"fastsync", "--one-file-system", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, long_argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->one_file_system); + + config_delete(cfg); + cfg = config_create(); + /* Flags never take a value: the "=value" form must be rejected. */ + char* bad_argv[] = {"fastsync", "--one-file-system=yes", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, bad_argv, positional_args, &positional_count), -1); + config_delete(cfg); +} + +/* Test rsync-compatible compression-choice and compression-level aliases. */ +static void test_parse_args_compression_aliases() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--zc", "zstd", "--zl", "10", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + int ret = parse_args(cfg, 7, argv, positional_args, &positional_count); + EXPECT_EQ_INT(ret, 0); + EXPECT_EQ_STR(cfg->compress_choice, "zstd"); + EXPECT_EQ_INT(cfg->compression_level, 10); + EXPECT_TRUE(cfg->use_compression); + EXPECT_EQ_INT(positional_count, 2); + + config_delete(cfg); +} + +/* Test rsync-compatible -P parsing; resumable partial-file retention is not implied. */ +static void test_parse_args_partial_progress() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "-P", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + int ret = parse_args(cfg, 4, argv, positional_args, &positional_count); + EXPECT_EQ_INT(ret, 0); + EXPECT_TRUE(cfg->partial); + EXPECT_TRUE(cfg->show_progress); + EXPECT_EQ_INT(positional_count, 2); + + config_delete(cfg); +} + +static void test_parse_args_compression_equals_and_none() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "-z", "--zc=none", "--zl=7", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + int ret = parse_args(cfg, 6, argv, positional_args, &positional_count); + EXPECT_EQ_INT(ret, 0); + EXPECT_EQ_STR(cfg->compress_choice, "none"); + EXPECT_EQ_INT(cfg->compression_level, 7); + EXPECT_FALSE(cfg->use_compression); + config_delete(cfg); +} + +static void test_parse_args_compression_canonical_equals() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--compress-choice=zstd", "--compress-level=7", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->compress_choice, "zstd"); + EXPECT_EQ_INT(cfg->compression_level, 7); + EXPECT_TRUE(cfg->use_compression); + EXPECT_EQ_INT(positional_count, 2); + + config_delete(cfg); +} + +static void test_parse_args_compression_alias_equals() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--zc=zstd", "--zl=7", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->compress_choice, "zstd"); + EXPECT_EQ_INT(cfg->compression_level, 7); + EXPECT_TRUE(cfg->use_compression); + EXPECT_EQ_INT(positional_count, 2); + + config_delete(cfg); +} + +static void test_parse_args_rejects_invalid_compression_level_equals() { + static const char* const values[] = {"0", "23", "invalid"}; + + for (size_t i = 0; i < sizeof(values) / sizeof(values[0]); i++) { + Config* cfg = config_create(); + char option[32]; + snprintf(option, sizeof(option), "--compress-level=%s", values[i]); + char* argv[] = {"fastsync", option, "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), -1); + config_delete(cfg); + } +} + +static void test_parse_args_rejects_invalid_compression_choice() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--compress-choice=bogus", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), -1); + config_delete(cfg); +} + +/* Every value-taking table option accepts an inline "--opt=value" form. */ +static void test_parse_args_table_equals_size_options() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--max-size=2G", "--min-size=1K", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->max_size == 2ULL * 1024 * 1024 * 1024); + EXPECT_TRUE(cfg->min_size == 1024ULL); + EXPECT_EQ_INT(positional_count, 2); + config_delete(cfg); +} + +static void test_parse_args_table_equals_string_and_int_options() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--suffix=.bak", "--timeout=30", "--max-depth=5", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 6, argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->suffix, ".bak"); + EXPECT_EQ_INT(cfg->timeout, 30); + EXPECT_EQ_INT(cfg->max_depth, 5); + EXPECT_EQ_INT(positional_count, 2); + config_delete(cfg); + + cfg = config_create(); + char* backup_argv[] = {"fastsync", "--backup-dir=/tmp/bak", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, backup_argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->backup_dir, "/tmp/bak"); + config_delete(cfg); +} + +/* Options that take a separate value must report "missing argument", not the + * generic "Unknown option", when they are the final argv entry. */ +static void test_parse_args_missing_argument_diagnostic() { + static const char* const options[] = {"--exclude", "--server-port", "--skip-compress", + "-T", "--out-format", "--log-file-format", + "--protocol"}; + + for (size_t i = 0; i < sizeof(options) / sizeof(options[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", (char*)options[i]}; + int positional_args[2]; + int positional_count = 0; + FILE* log_file = tmpfile(); + char log_buffer[512] = {0}; + + EXPECT_NOT_NULL(log_file); + log_set_file(log_file); + + EXPECT_EQ_INT(parse_args(cfg, 2, argv, positional_args, &positional_count), -1); + fflush(log_file); + rewind(log_file); + EXPECT_TRUE(fread(log_buffer, 1, sizeof(log_buffer) - 1, log_file) > 0); + EXPECT_TRUE(strstr(log_buffer, "missing argument") != NULL); + EXPECT_TRUE(strstr(log_buffer, "Unknown option") == NULL); + + log_set_file(NULL); + fclose(log_file); + config_delete(cfg); + } +} + +static void test_parse_args_itemize_changes() { + static const char* const flags[] = {"-i", "--itemize-changes"}; + for (size_t i = 0; i < sizeof(flags) / sizeof(flags[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", (char*)flags[i], "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->itemize_changes); + EXPECT_EQ_INT(positional_count, 2); + config_delete(cfg); + } +} + +static void test_parse_args_list_only() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--list-only", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->list_only); + config_delete(cfg); +} + +static void test_parse_args_out_format() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--out-format=%f %l", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->out_format, "%f %l"); + config_delete(cfg); + + cfg = config_create(); + char* separate_argv[] = {"fastsync", "--out-format", "%f %l", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, separate_argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->out_format, "%f %l"); + config_delete(cfg); +} + +static void test_parse_args_log_file_format() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--log-file-format=%n %M", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->log_file_format, "%n %M"); + config_delete(cfg); + + cfg = config_create(); + char* separate_argv[] = {"fastsync", "--log-file-format", "%n %M", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, separate_argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->log_file_format, "%n %M"); + config_delete(cfg); +} + +/* --delay-updates is a plain boolean receiver option. */ +static void test_parse_args_delay_updates() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--delay-updates", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->delay_updates); + EXPECT_EQ_INT(positional_count, 2); + config_delete(cfg); +} + +/* rsync rejects --delay-updates with --inplace; FastSync must too. */ +static void test_validate_config_delay_updates_rejects_inplace() { + Config* cfg = valid_client_config(); + cfg->delay_updates = true; + cfg->inplace = true; + EXPECT_FALSE(validate_config(cfg)); + config_delete(cfg); +} + +/* --backup-dir may not collide with the internal --delay-updates staging + directory (with or without a trailing slash), or old backups would silently + be installed as the "new" file. */ +static void test_validate_config_delay_updates_rejects_reserved_backup_dir() { + static const char* const reserved[] = {".fastsync-stage", ".fastsync-stage/"}; + for (size_t i = 0; i < sizeof(reserved) / sizeof(reserved[0]); i++) { + Config* cfg = valid_client_config(); + cfg->delay_updates = true; + cfg->backup_dir = str_dup(reserved[i]); + EXPECT_FALSE(validate_config(cfg)); + config_delete(cfg); + } + + /* A non-colliding backup dir is fine alongside --delay-updates. */ + Config* ok = valid_client_config(); + ok->delay_updates = true; + ok->backup_dir = str_dup("backups"); + EXPECT_TRUE(validate_config(ok)); + config_delete(ok); +} + +static void test_parse_args_filter_rules() { + Config* cfg = config_create(); + int positional_args[2]; + int positional_count = 0; + char* argv[] = {"fastsync", "--filter", "- *.tmp", "--filter=+ /keep.txt", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 6, argv, positional_args, &positional_count), 0); + EXPECT_NOT_NULL(cfg->filters); + EXPECT_EQ_INT(cfg->filters->size, 2); + EXPECT_EQ_STR((char*)cfg->filters->items[0], "- *.tmp"); + EXPECT_EQ_STR((char*)cfg->filters->items[1], "+ /keep.txt"); + config_delete(cfg); + + /* An unsupported rsync rule type is rejected with a clear error. */ + cfg = config_create(); + positional_count = 0; + char* bad_argv[] = {"fastsync", "--filter=merge /tmp/excludes", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, bad_argv, positional_args, &positional_count), -1); + config_delete(cfg); + + /* A trailing --filter with no rule is a missing-argument error. */ + cfg = config_create(); + positional_count = 0; + char* missing_argv[] = {"fastsync", "/src", "/dst", "--filter"}; + EXPECT_EQ_INT(parse_args(cfg, 4, missing_argv, positional_args, &positional_count), -1); + config_delete(cfg); + + /* rsync shorthands/modifiers we do not support are rejected instead of being + * silently parsed as literal patterns. */ + static const char* const unsupported[] = { + ": .rsync-filter", ". /tmp/rules", "-s foo", "-p bar", "-C", "-! *.o", "!", + }; + for (size_t i = 0; i < sizeof(unsupported) / sizeof(unsupported[0]); i++) { + cfg = config_create(); + positional_count = 0; + char* rule_argv[] = {"fastsync", "--filter", (char*)unsupported[i], "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, rule_argv, positional_args, &positional_count), -1); + config_delete(cfg); + } + + /* Supported spellings still parse: space- or slash-separated, attached + * wildcards, and anchored rules. */ + cfg = config_create(); + positional_count = 0; + char* ok_argv[] = {"fastsync", "--filter=-*.o", "--filter=- /foo", + "--filter=+ /bar/", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 6, ok_argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->filters->size, 3); + config_delete(cfg); +} + +static void test_parse_args_from0_cvs_filter_file_flags() { + static const struct { + const char* arg; + bool from0; + bool cvs; + bool per_dir; + } cases[] = { + {"--from0", true, false, false}, + {"-0", true, false, false}, + {"--cvs-exclude", false, true, false}, + {"-C", false, true, false}, + {"-F", false, false, true}, + }; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", (char*)cases[i].arg, "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->from0, cases[i].from0); + EXPECT_EQ_INT(cfg->cvs_exclude, cases[i].cvs); + EXPECT_EQ_INT(cfg->per_dir_filter, cases[i].per_dir); + config_delete(cfg); + } + + /* The plain booleans are negatable (--no-* simply clears the flag). */ + static const char* const on[][2] = {{"--from0", "--no-from0"}, {"-C", "--no-cvs-exclude"}}; + for (size_t i = 0; i < sizeof(on) / sizeof(on[0]); i++) { + Config* cfg = config_create(); + int positional_args[2]; + int positional_count = 0; + char* argv[] = {"fastsync", (char*)on[i][0], (char*)on[i][1], "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->from0); + EXPECT_FALSE(cfg->cvs_exclude); + config_delete(cfg); + } +} + +static void write_file_bytes(const char* path, const char* bytes, size_t len) { + FILE* fp = fopen(path, "wb"); + EXPECT_NOT_NULL(fp); + EXPECT_EQ_INT((int)fwrite(bytes, 1, len, fp), (int)len); + fclose(fp); +} + +static void test_parse_args_files_from() { + const char* list_path = "cli_files_from_list.txt"; + write_file_bytes(list_path, "a.txt\nsub/b.bin\n\n./c.txt\n", 25); + Config* cfg = config_create(); + int positional_args[2]; + int positional_count = 0; + char* argv[] = {"fastsync", "--files-from", (char*)list_path, "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->files_from, list_path); + EXPECT_NOT_NULL(cfg->files_from_set); + FileListSet* set = (FileListSet*)cfg->files_from_set; + EXPECT_TRUE(file_list_affects(set, "a.txt")); + EXPECT_TRUE(file_list_affects(set, "sub/b.bin")); + EXPECT_TRUE(file_list_affects(set, "sub/b.bin/x")); + EXPECT_TRUE(file_list_affects(set, "sub")); + EXPECT_TRUE(file_list_affects(set, "c.txt")); + EXPECT_FALSE(file_list_affects(set, "other.txt")); + config_delete(cfg); + remove(list_path); + + /* -0 switches the separator to NUL regardless of argument order, and NUL + * mode preserves entry bytes exactly (a trailing CR/LF is part of the name). */ + write_file_bytes(list_path, "x.txt\0y/z.bin\0", 14); + cfg = config_create(); + positional_count = 0; + char* nul_argv[] = {"fastsync", + "--files-from=" + "cli_files_from_list.txt", + "-0", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, nul_argv, positional_args, &positional_count), 0); + set = (FileListSet*)cfg->files_from_set; + EXPECT_NOT_NULL(set); + EXPECT_TRUE(file_list_affects(set, "x.txt")); + EXPECT_TRUE(file_list_affects(set, "y/z.bin")); + EXPECT_TRUE(file_list_affects(set, "y")); + EXPECT_FALSE(file_list_affects(set, "z.txt")); + config_delete(cfg); + remove(list_path); + + write_file_bytes(list_path, "crlf\n\0tail\0", 11); + cfg = config_create(); + positional_count = 0; + char* nul_nl_argv[] = {"fastsync", + "--files-from=" + "cli_files_from_list.txt", + "-0", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, nul_nl_argv, positional_args, &positional_count), 0); + set = (FileListSet*)cfg->files_from_set; + EXPECT_NOT_NULL(set); + EXPECT_TRUE(file_list_affects(set, "crlf\n")); + EXPECT_TRUE(file_list_affects(set, "tail")); + config_delete(cfg); + remove(list_path); + + /* A missing list file is a hard parse-time error. */ + cfg = config_create(); + positional_count = 0; + char* missing_argv[] = {"fastsync", "--files-from", "does_not_exist_ff.txt", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, missing_argv, positional_args, &positional_count), -1); + config_delete(cfg); + + /* Absolute and traversal entries are rejected. */ + write_file_bytes(list_path, "/abs/path\n", 10); + cfg = config_create(); + positional_count = 0; + char* abs_argv[] = {"fastsync", + "--files-from=" + "cli_files_from_list.txt", + "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, abs_argv, positional_args, &positional_count), -1); + config_delete(cfg); + write_file_bytes(list_path, "../escape\n", 10); + cfg = config_create(); + positional_count = 0; + char* trav_argv[] = {"fastsync", + "--files-from=" + "cli_files_from_list.txt", + "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, trav_argv, positional_args, &positional_count), -1); + config_delete(cfg); + remove(list_path); +} + +/* The deletion-policy family parses onto the config fields: --delete-excluded, + * --ignore-errors and --force are flags, --max-delete takes a non-negative + * number, and --prune-empty-dirs is the long-only spelling (FastSync's -m stays + * multithreading). None of them implies --delete by itself. */ +static void test_parse_args_delete_policy_flags() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", + "--delete", + "--delete-excluded", + "--max-delete=5", + "--ignore-errors", + "--force", + "--prune-empty-dirs", + "/src", + "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 9, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_delete); + EXPECT_TRUE(cfg->delete_excluded); + EXPECT_EQ_INT(cfg->max_delete, 5); + EXPECT_TRUE(cfg->ignore_errors); + EXPECT_TRUE(cfg->force_delete); + EXPECT_TRUE(cfg->prune_empty_dirs); + EXPECT_FALSE(cfg->delete_before); + config_delete(cfg); + + /* --max-delete accepts the separated-argument and zero forms. */ + cfg = config_create(); + char* argv2[] = {"fastsync", "--max-delete", "0", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv2, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->max_delete, 0); + config_delete(cfg); +} + +static void test_parse_args_delete_policy_invalid_values() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--max-delete=abc", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), -1); + config_delete(cfg); + + cfg = config_create(); + char* argv2[] = {"fastsync", "--max-delete=-3", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv2, positional_args, &positional_count), -1); + config_delete(cfg); +} + +/* --max-delete without --delete is inert (it only bounds a --delete run); the + * config stays valid. */ +static void test_parse_args_max_delete_inert_without_delete() { + Config* cfg = config_create(); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + char* argv[] = {"fastsync", "--max-delete=5", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->use_delete); + EXPECT_EQ_INT(cfg->max_delete, 5); + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); +} + +/* --ignore-missing-args / --delete-missing-args parse onto their config fields. + * --delete-missing-args implies --ignore-missing-args (order-independent), + * does NOT imply --delete (rsync: independent of other delete processing), and + * the config stays valid in every combination. */ +static void test_parse_args_missing_args_flags() { + Config* cfg = config_create(); + int positional_args[2]; + int positional_count = 0; + char* argv[] = {"fastsync", "--ignore-missing-args", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->ignore_missing_args); + EXPECT_FALSE(cfg->delete_missing_args); + EXPECT_FALSE(cfg->use_delete); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv2[] = {"fastsync", "--delete-missing-args", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv2, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->delete_missing_args); + EXPECT_TRUE(cfg->ignore_missing_args); + EXPECT_FALSE(cfg->use_delete); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); + + /* The implication is order-independent: even with the explicit flag first. */ + cfg = config_create(); + positional_count = 0; + char* argv3[] = {"fastsync", "--ignore-missing-args", "--delete-missing-args", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv3, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->ignore_missing_args); + EXPECT_TRUE(cfg->delete_missing_args); + config_delete(cfg); + + /* --delete-missing-args composes with --delete (both active) and with a + delete-timing flag (which implies --delete); timing stays valid. */ + cfg = config_create(); + positional_count = 0; + char* argv4[] = {"fastsync", "--delete", "--delete-missing-args", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv4, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_delete); + EXPECT_TRUE(cfg->delete_missing_args); + EXPECT_TRUE(cfg->ignore_missing_args); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv5[] = {"fastsync", "--delete-before", "--delete-missing-args", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv5, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_delete); + EXPECT_TRUE(cfg->delete_before); + EXPECT_TRUE(cfg->delete_missing_args); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); +} +/* --append is accepted and implies the per-file incremental check a tail resume + * needs; it validates cleanly on its own. */ +static void test_parse_args_append() { + Config* cfg = config_create(); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + char* argv[] = {"fastsync", "--append", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->append); + EXPECT_FALSE(cfg->append_verify); + EXPECT_TRUE(cfg->use_incremental); + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); +} + +static void test_parse_args_append_verify() { + Config* cfg = config_create(); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + char* argv[] = {"fastsync", "--append-verify", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->append_verify); + EXPECT_FALSE(cfg->append); + EXPECT_TRUE(cfg->use_incremental); + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); +} + +/* Both spellings are accepted; the safer --append-verify semantics win on the + * wire (the sender checks append_verify first), so neither flag is silently + * dropped but the run is still valid. */ +static void test_parse_args_append_both() { + Config* cfg = config_create(); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + char* argv[] = {"fastsync", "--append", "--append-verify", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->append); + EXPECT_TRUE(cfg->append_verify); + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); +} + +static void test_validate_config_append_rejects_chunk_serialization() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--append", "--chunk-serialization", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + EXPECT_FALSE(validate_config(cfg)); + config_delete(cfg); +} + +static void test_validate_config_append_verify_rejects_chunk_serialization() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--append-verify", "--chunk-serialization", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + EXPECT_FALSE(validate_config(cfg)); + config_delete(cfg); +} + +static void test_validate_config_append_rejects_whole_file() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--append", "-W", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_FALSE(validate_config(cfg)); + config_delete(cfg); +} + +static void test_validate_config_append_verify_rejects_whole_file() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--append-verify", "-W", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_FALSE(validate_config(cfg)); + config_delete(cfg); +} +/* --numeric-ids is a plain boolean flag. */ +static void test_parse_args_numeric_ids() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--numeric-ids", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->numeric_ids); + config_delete(cfg); +} + +/* --usermap / --groupmap resolve an rsync subset into numeric FROM:TO pairs and + * imply metadata preservation (so the source uid/gid travel on the wire). */ +static void test_parse_args_usermap() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--usermap=@1000:@1001", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_metadata); + EXPECT_EQ_INT(cfg->usermap_count, 1); + EXPECT_EQ_INT(cfg->usermap[0].from, 1000); + EXPECT_EQ_INT(cfg->usermap[0].to, 1001); + config_delete(cfg); + + /* Space form, multiple rules, comma-separated. */ + cfg = config_create(); + positional_count = 0; + char* argv2[] = {"fastsync", "--usermap", "@1:@2,@3:@4", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv2, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->usermap_count, 2); + EXPECT_EQ_INT(cfg->usermap[0].from, 1); + EXPECT_EQ_INT(cfg->usermap[0].to, 2); + EXPECT_EQ_INT(cfg->usermap[1].from, 3); + EXPECT_EQ_INT(cfg->usermap[1].to, 4); + config_delete(cfg); + + /* '*' FROM means match any id; '*' TO means current user. */ + cfg = config_create(); + positional_count = 0; + char* argv3[] = {"fastsync", "--usermap=*:@2000", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv3, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->usermap[0].from, IDENTITY_MATCH_ANY); + EXPECT_EQ_INT(cfg->usermap[0].to, 2000); + config_delete(cfg); +} + +static void test_parse_args_groupmap() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--groupmap=@100:@101", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_metadata); + EXPECT_EQ_INT(cfg->groupmap_count, 1); + EXPECT_EQ_INT(cfg->groupmap[0].from, 100); + EXPECT_EQ_INT(cfg->groupmap[0].to, 101); + config_delete(cfg); +} + +/* A name in a map can be resolved to a number via the local user database. */ +static void test_parse_args_usermap_name_resolution() { + struct passwd* self = getpwuid(geteuid()); + if (!self) + return; /* cannot construct a resolvable name deterministically */ + char map_value[128]; + snprintf(map_value, sizeof(map_value), "%s:@0", self->pw_name); + Config* cfg = config_create(); + int positional_args[2]; + int positional_count = 0; + char* argv[] = {"fastsync", (char*)"--usermap", map_value, "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->usermap_count, 1); + EXPECT_EQ_INT(cfg->usermap[0].from, (int32_t)self->pw_uid); + config_delete(cfg); +} + +/* --chown parses USER:GROUP / USER / :GROUP, numeric ids, and '*'. */ +static void test_parse_args_chown() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--chown=@1000:@1001", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_metadata); + EXPECT_TRUE(cfg->chown_uid_set); + EXPECT_EQ_INT(cfg->chown_uid, 1000); + EXPECT_TRUE(cfg->chown_gid_set); + EXPECT_EQ_INT(cfg->chown_gid, 1001); + config_delete(cfg); + + /* --chown=:GROUP sets only the group. */ + cfg = config_create(); + positional_count = 0; + char* argv2[] = {"fastsync", "--chown=:@1001", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv2, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->chown_uid_set); + EXPECT_TRUE(cfg->chown_gid_set); + EXPECT_EQ_INT(cfg->chown_gid, 1001); + config_delete(cfg); + + /* --chown=USER sets only the owner. */ + cfg = config_create(); + positional_count = 0; + char* argv3[] = {"fastsync", "--chown=@1000", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv3, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->chown_uid_set); + EXPECT_EQ_INT(cfg->chown_uid, 1000); + EXPECT_FALSE(cfg->chown_gid_set); + config_delete(cfg); + + /* '*' means current user/group. */ + cfg = config_create(); + positional_count = 0; + char* argv4[] = {"fastsync", "--chown=*:*", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv4, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->chown_uid_set); + EXPECT_EQ_INT(cfg->chown_uid, IDENTITY_CURRENT); + EXPECT_TRUE(cfg->chown_gid_set); + EXPECT_EQ_INT(cfg->chown_gid, IDENTITY_CURRENT); + config_delete(cfg); +} + +/* --copy-as=USER[:GROUP] (P7 Wave E): resolve the user/group against the local + * databases, imply metadata, and apply the documented group-default rule. */ +static void test_parse_args_copy_as() { + /* Explicit numeric user and group. */ + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--copy-as=@1000:@1001", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->copy_as_set); + EXPECT_TRUE(cfg->use_metadata); + EXPECT_EQ_INT(cfg->copy_as_uid, 1000); + EXPECT_EQ_INT(cfg->copy_as_gid, 1001); + config_delete(cfg); + + /* Space form. */ + cfg = config_create(); + positional_count = 0; + char* argv2[] = {"fastsync", "--copy-as", "@2000:3000", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv2, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->copy_as_uid, 2000); + EXPECT_EQ_INT(cfg->copy_as_gid, 3000); + config_delete(cfg); + + /* Group omitted: a resolvable user uses its primary gid. */ + struct passwd* self = getpwuid(geteuid()); + if (self) { + cfg = config_create(); + positional_count = 0; + char* argv3[] = {"fastsync", (char*)"--copy-as", (char*)self->pw_name, "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv3, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->copy_as_uid, (int32_t)self->pw_uid); + EXPECT_EQ_INT(cfg->copy_as_gid, (int32_t)self->pw_gid); + config_delete(cfg); + } + + /* Group omitted with a numeric id that has no passwd entry: gid falls back + * to uid (documented divergence). */ + if (!getpwuid((uid_t)4242)) { + cfg = config_create(); + positional_count = 0; + char* argv4[] = {"fastsync", "--copy-as=@4242", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv4, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->copy_as_uid, 4242); + EXPECT_EQ_INT(cfg->copy_as_gid, 4242); + config_delete(cfg); + } + + /* '*' means the client's current euid/egid. */ + cfg = config_create(); + positional_count = 0; + char* argv5[] = {"fastsync", "--copy-as=*:*", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv5, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->copy_as_uid, (int32_t)geteuid()); + EXPECT_EQ_INT(cfg->copy_as_gid, (int32_t)getegid()); + config_delete(cfg); +} + +/* Malformed identity specs are rejected, never silently ignored. */ +static void test_parse_args_rejects_malformed_identity() { + struct { + const char* opt; + const char* val; + } bad[] = { + {"--usermap", "@1000"}, + {"--usermap", ":1000"}, + {"--usermap", "definitely_not_a_real_user_zzz:@1"}, + {"--groupmap", "@1"}, + {"--groupmap", "no_such_group_qqq:x"}, + {"--chown", "a:b:c"}, + {"--chown", "no_such_user_zzz:"}, + {"--copy-as", ""}, + {"--copy-as", ":"}, + {"--copy-as", "a:b:c"}, + {"--copy-as", "@1000:"}, + {"--copy-as", "definitely_not_a_real_user_zzz"}, + {"--copy-as", "no_such_group_qqq_group"}, + }; + for (size_t i = 0; i < sizeof(bad) / sizeof(bad[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", (char*)bad[i].opt, (char*)bad[i].val, "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1); + config_delete(cfg); + } + + /* An option with a missing value fails at the CLI layer. */ + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--chown"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 2, argv, positional_args, &positional_count), -1); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv2[] = {"fastsync", "--copy-as"}; + EXPECT_EQ_INT(parse_args(cfg, 2, argv2, positional_args, &positional_count), -1); + config_delete(cfg); +} + +/* --preallocate parses as a boolean flag and validates cleanly. */ +static void test_parse_args_preallocate() { + Config* cfg = config_create(); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + char* argv[] = {"fastsync", "--preallocate", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->preallocate); + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); +} + +/* Phase 4 metadata-time flags parse and set the expected config fields. -U and + * -N imply metadata transmission (they carry their times inside the metadata + * payload); -O/-J and --open-noatime do not. */ +static void test_parse_args_metadata_times() { + Config* cfg = config_create(); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + char* argv[] = {"fastsync", "-U", "-N", "-O", "-J", "--open-noatime", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 8, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->preserve_atimes); + EXPECT_TRUE(cfg->preserve_crtimes); + EXPECT_TRUE(cfg->omit_dir_times); + EXPECT_TRUE(cfg->omit_link_times); + EXPECT_TRUE(cfg->open_noatime); + /* -U/-N carry their times inside the metadata payload, so they imply it. */ + EXPECT_TRUE(cfg->use_metadata); + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); +} + +static void test_parse_args_atimes_long_and_short() { + Config* cfg = config_create(); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + char* argv[] = {"fastsync", "--atimes", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->preserve_atimes); + EXPECT_TRUE(cfg->use_metadata); + config_delete(cfg); +} + +static void test_parse_args_omit_link_times_long() { + Config* cfg = config_create(); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + char* argv[] = {"fastsync", "--omit-link-times", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->omit_link_times); + EXPECT_FALSE(cfg->use_metadata); + config_delete(cfg); +} + +/* --devices / --specials / -D / --copy-devices / --write-devices parse into the + config, and the preserved flags imply metadata transmission. */ +static void test_parse_args_devices_specials() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--devices", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->preserve_devices); + EXPECT_FALSE(cfg->preserve_specials); + EXPECT_TRUE(cfg->use_metadata); + config_delete(cfg); + + cfg = config_create(); + char* argv2[] = {"fastsync", "--specials", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv2, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->preserve_specials); + EXPECT_FALSE(cfg->preserve_devices); + EXPECT_TRUE(cfg->use_metadata); + config_delete(cfg); + + cfg = config_create(); + char* argv3[] = {"fastsync", "-D", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv3, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->preserve_devices); + EXPECT_TRUE(cfg->preserve_specials); + EXPECT_TRUE(cfg->use_metadata); + config_delete(cfg); + + cfg = config_create(); + char* argv4[] = {"fastsync", "--copy-devices", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv4, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->copy_devices); + config_delete(cfg); + + cfg = config_create(); + char* argv5[] = {"fastsync", "--write-devices", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv5, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->write_devices); + config_delete(cfg); +} + +/* --address binds the outgoing client socket; it is a plain string option. */ +static void test_parse_args_address() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--address", "192.0.2.10", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->address, "192.0.2.10"); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* eq_argv[] = {"fastsync", "--address=10.0.0.5", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, eq_argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->address, "10.0.0.5"); + config_delete(cfg); +} + +/* -4/--ipv4 and -6/--ipv6 set the resolution family; both together are + * rejected by validate_config (an address cannot be both v4 and v6). */ +static void test_parse_args_ipv4_ipv6() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "-4", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->ipv4); + EXPECT_FALSE(cfg->ipv6); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* longv6[] = {"fastsync", "--ipv6", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, longv6, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->ipv4); + EXPECT_TRUE(cfg->ipv6); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* both[] = {"fastsync", "-4", "-6", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, both, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->ipv4); + EXPECT_TRUE(cfg->ipv6); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + EXPECT_FALSE(validate_config(cfg)); + config_delete(cfg); +} + +/* --sockopts parses and stores the allowlist; unknown options and bad values + * are rejected at the CLI layer (never silently ignored). */ +static void test_parse_args_sockopts() { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--sockopts=TCP_NODELAY=1,SO_KEEPALIVE=1", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->sockopt_count, 2); + EXPECT_EQ_INT(cfg->sockopts[0].id, SOCKOPT_TCP_NODELAY); + EXPECT_EQ_INT(cfg->sockopts[0].value, 1); + EXPECT_EQ_INT(cfg->sockopts[1].id, SOCKOPT_SO_KEEPALIVE); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* sep_argv[] = {"fastsync", "--sockopts", "SO_RCVBUF=65536", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, sep_argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->sockopt_count, 1); + EXPECT_EQ_INT(cfg->sockopts[0].id, SOCKOPT_SO_RCVBUF); + EXPECT_EQ_INT(cfg->sockopts[0].value, 65536); + config_delete(cfg); + + static const char* const bad[] = {"--sockopts=IP_TTL=1", "--sockopts=TCP_NODELAY=2", + "--sockopts=SO_KEEPALIVE"}; + for (size_t i = 0; i < sizeof(bad) / sizeof(bad[0]); i++) { + cfg = config_create(); + positional_count = 0; + char* b[] = {"fastsync", (char*)bad[i], "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, b, positional_args, &positional_count), -1); + config_delete(cfg); + } +} + +/* --trust-sender parses; default is false (receiver-local policy, off). */ +static void test_parse_args_trust_sender_default_false() { + Config* cfg = valid_client_config(); + EXPECT_NOT_NULL(cfg); + char* argv[] = {"fastsync", "--source-dir", "/src", "--dest-dir", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->trust_sender); + config_delete(cfg); +} + +static void test_parse_args_trust_sender() { + Config* cfg = valid_client_config(); + EXPECT_NOT_NULL(cfg); + char* argv[] = {"fastsync", "--trust-sender", "--source-dir", "/src", "--dest-dir", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 6, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->trust_sender); + config_delete(cfg); +} + +/* --remote-option=OPT is repeatable and stores each value in order. */ +static void test_parse_args_remote_option_multiple() { + Config* cfg = valid_client_config(); + EXPECT_NOT_NULL(cfg); + EXPECT_EQ_INT(cfg->remote_option_count, 0); + char* argv[] = {"fastsync", + "--source-dir", + "/src", + "--dest-dir", + "/dst", + "--remote-option=--allow-delete", + "--remote-option=--verbose"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 7, argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->remote_option_count, 2); + EXPECT_EQ_STR(cfg->remote_options[0], "--allow-delete"); + EXPECT_EQ_STR(cfg->remote_options[1], "--verbose"); + config_delete(cfg); +} + +/* Space-separated form "--remote-option OPT" also parses. */ +static void test_parse_args_remote_option_space_form() { + Config* cfg = valid_client_config(); + EXPECT_NOT_NULL(cfg); + char* argv[] = {"fastsync", "--source-dir", "/src", "--dest-dir", + "/dst", "--remote-option", "-v"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 7, argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->remote_option_count, 1); + EXPECT_EQ_STR(cfg->remote_options[0], "-v"); + config_delete(cfg); +} + +/* A missing argument bare --remote-option is rejected. */ +static void test_parse_args_remote_option_missing_value() { + Config* cfg = valid_client_config(); + EXPECT_NOT_NULL(cfg); + char* argv[] = {"fastsync", "--source-dir", "/src", "--dest-dir", "/dst", "--remote-option"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 6, argv, positional_args, &positional_count), -1); + config_delete(cfg); +} + +/* An empty --remote-option value and a value with control characters is + * rejected (the value would break the remote shell quoting). */ +static void test_parse_args_remote_option_rejects_bad_values() { + Config* cfg = valid_client_config(); + EXPECT_NOT_NULL(cfg); + char* argv[] = {"fastsync", "--source-dir", "/src", "--dest-dir", "/dst", "--remote-option="}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 6, argv, positional_args, &positional_count), -1); + EXPECT_EQ_INT(cfg->remote_option_count, 0); + + char* argv2[] = {"fastsync", "--source-dir", "/src", "--dest-dir", + "/dst", "--remote-option", "--bad\noption"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 7, argv2, positional_args, &positional_count), -1); + EXPECT_EQ_INT(cfg->remote_option_count, 0); + config_delete(cfg); +} + +/* Since the Phase-7 CLI-namespace pass, -M is rsync's --remote-option short + * form (FastSync metadata mode is long-only --preserve): it consumes the next + * argv as a remote-option value and must NOT set FastSync metadata mode. */ +static void test_parse_args_remote_option_short_M() { + Config* cfg = valid_client_config(); + EXPECT_NOT_NULL(cfg); + char* argv[] = {"fastsync", "-M", "--trust-sender", "--source-dir", "/src", "--dest-dir", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 7, argv, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->use_metadata); + EXPECT_EQ_INT(cfg->remote_option_count, 1); + config_delete(cfg); +} + +/* --no-motd is a real rsync option (client-side daemon MOTD display + * suppression), not a negation of a --motd flag: it sets config->no_motd. */ +static void test_parse_args_no_motd() { + Config* cfg = valid_client_config(); + EXPECT_NOT_NULL(cfg); + EXPECT_FALSE(cfg->no_motd); + char* argv[] = {"fastsync", "--source-dir", "/src", "--dest-dir", "/dst", "--no-motd"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 6, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->no_motd); + config_delete(cfg); + + cfg = valid_client_config(); + EXPECT_NOT_NULL(cfg); + EXPECT_FALSE(cfg->no_motd); + char* argv2[] = {"fastsync", "--source-dir", "/src", "--dest-dir", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv2, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->no_motd); + config_delete(cfg); +} + +/* --password-file stores its path on the config (the file is read later, once + * the destination form is known). */ +static void test_parse_args_password_file() { + Config* cfg = valid_client_config(); + EXPECT_NOT_NULL(cfg); + char* argv[] = {"fastsync", "--source-dir", "/src", + "--dest-dir", "/dst", "--password-file=/etc/fast.pw"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 6, argv, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->password_file, "/etc/fast.pw"); + + char* argv2[] = {"fastsync", "--source-dir", "/src", "--dest-dir", + "/dst", "--password-file", "/etc/other.pw"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 7, argv2, positional_args, &positional_count), 0); + EXPECT_EQ_STR(cfg->password_file, "/etc/other.pw"); + config_delete(cfg); +} + +/* --block-size (Delta block size): --block-size/--delta-block set + * config->delta_block_size, out-of-range values are rejected with the default + * kept, and the configured size genuinely reaches the delta engine (a larger + * block yields fewer signature blocks for identical data). */ +static void test_parse_args_block_size() { + Config* cfg = config_create(); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + int positional_args[2]; + int positional_count = 0; + + char* argv_long[] = {"fastsync", "--block-size", "4096", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv_long, positional_args, &positional_count), 0); + EXPECT_EQ_INT((int)cfg->delta_block_size, 4096); + + cfg->delta_block_size = DELTA_BLOCK_SIZE_DEFAULT; + char* argv_delta[] = {"fastsync", "--delta-block", "2048", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv_delta, positional_args, &positional_count), 0); + EXPECT_EQ_INT((int)cfg->delta_block_size, 2048); + + /* Inline =SIZE forms (the documented rsync spelling) are accepted too. */ + cfg->delta_block_size = DELTA_BLOCK_SIZE_DEFAULT; + char* argv_eq[] = {"fastsync", "--block-size=8192", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_eq, positional_args, &positional_count), 0); + EXPECT_EQ_INT((int)cfg->delta_block_size, 8192); + + cfg->delta_block_size = DELTA_BLOCK_SIZE_DEFAULT; + char* argv_delta_eq[] = {"fastsync", "--delta-block=1024", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_delta_eq, positional_args, &positional_count), 0); + EXPECT_EQ_INT((int)cfg->delta_block_size, 1024); + + /* Out of range: parsed, warned, and the default is kept (both spellings). */ + cfg->delta_block_size = DELTA_BLOCK_SIZE_DEFAULT; + char* argv_bad[] = {"fastsync", "--block-size", "1", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv_bad, positional_args, &positional_count), 0); + EXPECT_EQ_INT((int)cfg->delta_block_size, (int)DELTA_BLOCK_SIZE_DEFAULT); + + cfg->delta_block_size = DELTA_BLOCK_SIZE_DEFAULT; + char* argv_bad_inline[] = {"fastsync", "--delta-block=999999", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_bad_inline, positional_args, &positional_count), 0); + EXPECT_EQ_INT((int)cfg->delta_block_size, (int)DELTA_BLOCK_SIZE_DEFAULT); + + /* A non-numeric value is a hard error for both spellings. */ + char* argv_nan[] = {"fastsync", "--delta-block=abc", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_nan, positional_args, &positional_count), -1); + + /* A non-default block size changes the number of signature blocks for + identical data: block_count = ceil(size / block_size). */ + const char data[10000] = {0}; + DeltaSignature* small = delta_signature_create_seeded(data, sizeof(data), 1024, 0); + DeltaSignature* large = delta_signature_create_seeded(data, sizeof(data), 8192, 0); + EXPECT_NOT_NULL(small); + EXPECT_NOT_NULL(large); + /* cppcheck-suppress knownConditionTrueFalse -- EXPECT_NOT_NULL above asserts, + but cppcheck cannot see through the macro; the guard is defensive. */ + if (small && large) { + EXPECT_TRUE(large->block_size == 8192 && small->block_size == 1024); + EXPECT_TRUE(large->block_count < small->block_count); + EXPECT_EQ_INT((int)small->block_count, 10); /* ceil(10000/1024) */ + EXPECT_EQ_INT((int)large->block_count, 2); /* ceil(10000/8192) */ + } + delta_signature_destroy(small); + delta_signature_destroy(large); + + config_delete(cfg); +} + void test_client_cli() { + test_validate_config_required_paths(); + test_parse_args_numeric_ids(); + test_parse_args_usermap(); + test_parse_args_groupmap(); + test_parse_args_usermap_name_resolution(); + test_parse_args_chown(); + test_parse_args_copy_as(); + test_parse_args_rejects_malformed_identity(); + test_parse_args_preallocate(); + test_parse_args_metadata_times(); + test_parse_args_block_size(); + test_parse_args_devices_specials(); + test_parse_args_atimes_long_and_short(); + test_parse_args_omit_link_times_long(); + test_parse_args_address(); + test_parse_args_ipv4_ipv6(); + test_parse_args_sockopts(); + test_parse_args_append(); + test_parse_args_append_verify(); + test_parse_args_append_both(); + test_validate_config_append_rejects_chunk_serialization(); + test_validate_config_append_verify_rejects_chunk_serialization(); + test_validate_config_append_rejects_whole_file(); + test_validate_config_append_verify_rejects_whole_file(); + test_validate_config_incompatible_options(); + test_validate_config_tls_requirements(); + test_validate_config_credentials_require_tls_or_loopback(); + test_validate_config_delta_sendfile_constraints(); test_cli_help(); test_cli_archive_flags(); test_cli_dry_run(); + test_cli_remove_source_files(); + test_parse_args_remove_source_files(); test_cli_delete_flag(); test_cli_exclude_patterns(); + test_parse_args_help(); + test_parse_args_version(); + test_parse_args_protocol_accept_current(); + test_parse_args_protocol_rejects_other_versions(); + test_validate_config_protocol_version(); + test_parse_args_valid_port(); + test_parse_args_size_only(); + test_parse_args_ignore_existing(); + test_parse_args_executability(); + test_parse_args_chmod(); + test_parse_args_numeric_chmod(); + test_parse_args_rejects_invalid_chmod(); + test_parse_args_invalid_port(); + test_parse_args_non_numeric_port(); + test_parse_args_invalid_server_port(); + test_parse_args_invalid_compression_level(); + test_parse_args_valid_compression_level(); + test_parse_args_debug_flags(); + test_parse_args_debug_help(); + test_parse_args_debug_flags_validation(); + test_parse_args_modify_window(); + test_parse_args_rejects_invalid_modify_window(); + test_parse_args_skip_compress(); + test_parse_args_empty_skip_compress(); + test_parse_args_compression_threads(); + test_parse_args_max_alloc_sizes(); + test_parse_args_rejects_invalid_max_alloc(); + test_parse_args_unknown_option(); + test_parse_args_dirs_aliases(); + test_parse_args_relative_no_implied_mkpath(); + test_parse_args_delete_during_alias(); + test_parse_args_delete_timing_flags(); + test_parse_args_delete_timing_conflict_rejected(); + test_parse_args_delete_timing_without_delete_rejected(); + test_parse_args_rejects_unimplemented_options(); + test_parse_args_quiet(); + test_parse_args_human_readable(); + test_parse_args_hard_links(); + test_validate_config_hard_links_incompatible_modes(); + test_parse_args_update(); + test_parse_args_info_flags(); + test_parse_args_info_verbose_order(); + test_parse_args_rejects_invalid_info_flag(); + test_parse_args_archive(); + test_parse_args_negations(); + test_parse_args_negate_preserve_without_devices(); + test_parse_args_negation_order(); + test_parse_args_no_preserve_blocks_implicit_metadata(); + test_parse_args_rejects_unsafe_negation(); + test_parse_args_old_args(); + test_parse_args_rsh(); + test_parse_args_rsync_path_alias(); + test_parse_args_blocking_io(); + test_parse_args_outbuf(); + test_parse_args_fsync(); + test_parse_args_existing(); + test_parse_args_ignore_times(); + test_parse_args_8_bit_output(); + test_parse_args_stderr_modes(); + test_parse_args_rejects_unsupported_stderr_modes(); + test_parse_args_secluded_args(); + test_parse_args_chunk_serialization_long_form(); + test_parse_args_symlink_trust(); + test_parse_args_whole_file(); + test_parse_args_fuzzy_implies_delta(); + test_parse_args_fuzzy_negation(); + test_parse_args_fuzzy_with_whole_file(); + test_parse_args_fuzzy_respects_no_delta(); + test_parse_args_fuzzy_respects_no_incremental(); + test_validate_config_fuzzy_incompatible_modes(); + test_parse_args_one_file_system(); + test_parse_args_compression_aliases(); + test_parse_args_compression_equals_and_none(); + test_parse_args_compression_canonical_equals(); + test_parse_args_compression_alias_equals(); + test_parse_args_rejects_invalid_compression_level_equals(); + test_parse_args_rejects_invalid_compression_choice(); + test_parse_args_table_equals_size_options(); + test_parse_args_table_equals_string_and_int_options(); + test_parse_args_missing_argument_diagnostic(); + test_parse_args_xattrs_acls(); + test_parse_args_fake_super(); + test_parse_args_super(); + test_parse_args_partial_progress(); + test_parse_args_itemize_changes(); + test_parse_args_list_only(); + test_parse_args_out_format(); + test_parse_args_log_file_format(); + test_parse_args_checksum_choice_aliases(); + test_parse_args_checksum_choice_requires_value(); + test_parse_args_checksum_choice_equals_forms(); + test_parse_args_checksum_choice_rejects_unsupported(); + test_parse_args_checksum_seed(); + test_parse_args_temp_dir(); + test_parse_args_delay_updates(); + test_validate_config_delay_updates_rejects_inplace(); + test_validate_config_delay_updates_rejects_reserved_backup_dir(); + test_parse_args_files_from(); + test_parse_args_filter_rules(); + test_parse_args_from0_cvs_filter_file_flags(); + test_parse_args_basis_dirs(); + test_parse_args_basis_invalid_paths(); + test_validate_config_basis_rejects_chunk_serialization(); + test_parse_args_delete_policy_flags(); + test_parse_args_delete_policy_invalid_values(); + test_parse_args_max_delete_inert_without_delete(); + test_parse_args_missing_args_flags(); + test_parse_args_trust_sender_default_false(); + test_parse_args_trust_sender(); + test_parse_args_remote_option_multiple(); + test_parse_args_remote_option_space_form(); + test_parse_args_remote_option_missing_value(); + test_parse_args_remote_option_rejects_bad_values(); + test_parse_args_remote_option_short_M(); + test_parse_args_no_motd(); + test_parse_args_password_file(); } diff --git a/tests/test_compression.c b/tests/test_compression.c index 08d8615..8784046 100644 --- a/tests/test_compression.c +++ b/tests/test_compression.c @@ -13,6 +13,8 @@ static void test_data_compress_decompress_roundtrip() { size_t len = strlen(original); char* buf = malloc(len); + if (!buf) + return; memcpy(buf, original, len); Data* original_data = data_create(buf, len); EXPECT_NOT_NULL(original_data); @@ -53,6 +55,31 @@ static void test_data_compress_decompress_large() { data_destroy(decompressed); } +static void test_skip_compress_suffix_matching() { + char* suffixes[] = {".ZIP", ".GZ"}; + EXPECT_TRUE(compression_should_skip_with_suffixes("archive.zip", suffixes, 2)); + EXPECT_TRUE(compression_should_skip_with_suffixes("backup.TAR.GZ", suffixes, 2)); + EXPECT_FALSE(compression_should_skip_with_suffixes("notes.txt", suffixes, 2)); + EXPECT_FALSE(compression_should_skip_with_suffixes("archive.zip", suffixes, 0)); +} + +static void test_data_compress_with_threads_roundtrip() { + const size_t size = 8 * 1024 * 1024; + Data* input = data_create_empty(size); + EXPECT_NOT_NULL(input); + for (size_t i = 0; i < size; i++) + ((char*)input->data)[i] = (char)((i / 4096) % 7); + Data* compressed = data_compress_with_threads(input, 3, 2); + EXPECT_NOT_NULL(compressed); + Data* decompressed = data_decompress(compressed); + EXPECT_NOT_NULL(decompressed); + EXPECT_EQ_INT((int)decompressed->size, (int)size); + EXPECT_EQ_INT(memcmp(decompressed->data, input->data, size), 0); + data_destroy(input); + data_destroy(compressed); + data_destroy(decompressed); +} + static void test_chunk_compress_decompress_roundtrip() { char* path1 = "temp_comp_test_1.txt"; char* content1 = "chunk compression test file 1"; @@ -62,8 +89,8 @@ static void test_chunk_compress_decompress_roundtrip() { char* content2 = "chunk compression test file 2 with more data"; unsigned long long len2 = strlen(content2); - to_disk(path1, content1, len1); - to_disk(path2, content2, len2); + file_write_to_disk(path1, content1, len1, false, false); + file_write_to_disk(path2, content2, len2, false, false); struct stat st1, st2; EXPECT_EQ_INT(stat(path1, &st1), 0); @@ -113,5 +140,7 @@ static void test_chunk_compress_decompress_roundtrip() { void test_compression() { test_data_compress_decompress_roundtrip(); test_data_compress_decompress_large(); + test_skip_compress_suffix_matching(); + test_data_compress_with_threads_roundtrip(); test_chunk_compress_decompress_roundtrip(); } diff --git a/tests/test_config.c b/tests/test_config.c index 5fb3b60..f70d99d 100644 --- a/tests/test_config.c +++ b/tests/test_config.c @@ -1,5 +1,6 @@ #include "test_config.h" #include "config.h" +#include "identity.h" #include "multiprocessing.h" #include "protocol.h" #include "queue.h" @@ -83,6 +84,289 @@ static void test_config_ssh_dest_no_user() { config_delete(cfg); } +static void test_config_daemon_dest_parse() { + Config* cfg = make_config("1.0", "/src", "dahost::files/sub/dir", true, false, false, false, + false, 1, false, 0); + int ret = config_parse_daemon_dest(cfg); + EXPECT_EQ_INT(ret, 1); + EXPECT_EQ_INT(cfg->transport, TRANSPORT_TCP); + EXPECT_EQ_STR(cfg->server_host, "dahost"); + EXPECT_EQ_STR(cfg->module, "files"); + EXPECT_EQ_STR(cfg->receive_root_directory, "sub/dir"); + config_delete(cfg); +} + +static void test_config_daemon_dest_no_path() { + Config* cfg = + make_config("1.0", "/src", "dahost::files", true, false, false, false, false, 1, false, 0); + int ret = config_parse_daemon_dest(cfg); + EXPECT_EQ_INT(ret, 1); + EXPECT_EQ_STR(cfg->server_host, "dahost"); + EXPECT_EQ_STR(cfg->module, "files"); + EXPECT_EQ_STR(cfg->receive_root_directory, ""); + config_delete(cfg); +} + +static void test_config_daemon_dest_double_slash_normalized() { + Config* cfg = make_config("1.0", "/src", "dahost::files//sub", true, false, false, false, false, + 1, false, 0); + int ret = config_parse_daemon_dest(cfg); + EXPECT_EQ_INT(ret, 1); + EXPECT_EQ_STR(cfg->module, "files"); + EXPECT_EQ_STR(cfg->receive_root_directory, "sub"); + config_delete(cfg); +} + +static void test_config_daemon_dest_bad() { + /* Missing module name after "::". */ + Config* cfg = + make_config("1.0", "/src", "dahost::", true, false, false, false, false, 1, false, 0); + EXPECT_EQ_INT(config_parse_daemon_dest(cfg), -1); + config_delete(cfg); + + /* Invalid module name. */ + cfg = + make_config("1.0", "/src", "dahost::bad name", true, false, false, false, false, 1, false, 0); + EXPECT_EQ_INT(config_parse_daemon_dest(cfg), -1); + config_delete(cfg); + + /* Traversal path rejected. */ + cfg = make_config("1.0", "/src", "dahost::mod/../../etc", true, false, false, false, false, 1, + false, 0); + EXPECT_EQ_INT(config_parse_daemon_dest(cfg), -1); + config_delete(cfg); + + /* user@host::module is not yet supported. */ + cfg = + make_config("1.0", "/src", "user@dahost::mod", true, false, false, false, false, 1, false, 0); + EXPECT_EQ_INT(config_parse_daemon_dest(cfg), -1); + config_delete(cfg); + + /* A non-daemon destination is untouched (returns 0). */ + cfg = make_config("1.0", "/src", "plain:path", true, false, false, false, false, 1, false, 0); + EXPECT_EQ_INT(config_parse_daemon_dest(cfg), 0); + EXPECT_EQ_STR(cfg->receive_root_directory, "plain:path"); + config_delete(cfg); +} + +static void test_config_transport_dest_daemon_beats_ssh() { + /* host::module selects daemon TCP; host:path still selects SSH. */ + Config* cfg = make_config("1.0", "/src", "h::m/x", true, false, false, false, false, 1, false, 0); + int ret = config_parse_transport_dest(cfg); + EXPECT_EQ_INT(ret, 1); + EXPECT_EQ_INT(cfg->transport, TRANSPORT_TCP); + EXPECT_EQ_STR(cfg->module, "m"); + config_delete(cfg); + + cfg = make_config("1.0", "/src", "h:dst", true, false, false, false, false, 1, false, 0); + ret = config_parse_transport_dest(cfg); + EXPECT_EQ_INT(ret, 0); + EXPECT_EQ_INT(cfg->transport, TRANSPORT_SSH); + EXPECT_EQ_STR(cfg->receive_root_directory, "dst"); + config_delete(cfg); +} + +static void test_config_is_daemon_dest() { + EXPECT_TRUE(config_is_daemon_dest("host::mod")); + EXPECT_TRUE(config_is_daemon_dest("host::mod/path")); + EXPECT_FALSE(config_is_daemon_dest("host:path")); + EXPECT_FALSE(config_is_daemon_dest("/local/path")); + /* A colon inside the module-relative path does not change the detection. */ + EXPECT_TRUE(config_is_daemon_dest("host::mod/single:colon")); + EXPECT_FALSE(config_is_daemon_dest(NULL)); +} + +static void test_config_module_wire_roundtrip() { + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("rel/path"); + send_cfg->module = str_dup("backup"); + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv_cfg = config_receive(p[0]); + bool ok = recv_cfg != NULL && recv_cfg->module != NULL && + strcmp(recv_cfg->module, "backup") == 0 && + strcmp(recv_cfg->receive_root_directory, "rel/path") == 0; + config_delete(recv_cfg); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +static void test_config_module_wire_empty_canonicalizes_to_null() { + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("/dst"); + /* module left NULL -> serialized as "" -> received back as NULL. */ + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv_cfg = config_receive(p[0]); + bool ok = recv_cfg != NULL && recv_cfg->module == NULL; + config_delete(recv_cfg); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +/* Daemon auth credentials (A7, protocol 2.19.0) ride the config frame as the + * username ONLY; the literal password never crosses the wire. Round-trip a + * present username. */ +static void test_config_daemon_auth_wire_roundtrip() { + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("rel/path"); + send_cfg->module = str_dup("backup"); + send_cfg->auth_user = str_dup("alice"); + send_cfg->auth_password = str_dup("alice-s3cret"); + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv_cfg = config_receive(p[0]); + /* The plaintext password is client-only: it is never serialized. */ + bool ok = recv_cfg != NULL && recv_cfg->auth_user != NULL && + strcmp(recv_cfg->auth_user, "alice") == 0 && recv_cfg->auth_password == NULL; + config_delete(recv_cfg); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +/* The receive side validates the auth payload: a present-but-malformed username + * is refused (config_receive returns NULL), so a hostile peer cannot slip a + * garbage credential past the receive guard into the module gate. */ +static void test_config_daemon_auth_wire_rejects_malformed() { + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("/dst"); + send_cfg->module = str_dup("m"); + send_cfg->auth_user = str_dup("bad user"); + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv_cfg = config_receive(p[0]); + bool ok = recv_cfg == NULL; + config_delete(recv_cfg); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_FALSE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +/* A module gate that rejects any connection that names a module. */ +static const char* reject_named_module_gate(const Config* config, void* context) { + (void)context; + if (config && config->module && config->module[0] != '\0') + return "test rejection"; + return NULL; +} + +static void test_config_receive_with_validate_rejects() { + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("/dst"); + send_cfg->module = str_dup("any-module"); + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive_with_validate(p[0], reject_named_module_gate, NULL); + bool ok = recv == NULL; + config_delete(recv); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_FALSE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + static void test_pipeline_sender_lifecycle() { Config* cfg = make_config("2.0", "/src2", "/dst2", false, false, true, true, false, 1, false, 0); Queue* q1 = queue_create(5, NULL); @@ -95,6 +379,7 @@ static void test_pipeline_sender_lifecycle() { EXPECT_EQ_INT(pcs->queue_loader->capacity, 15); EXPECT_FALSE(pcs->scanner_done); EXPECT_FALSE(pcs->loader_done); + EXPECT_EQ_INT((int)pcs->allocation_session.max_alloc, (int)cfg->max_alloc); pipeline_context_sender_destroy(pcs); } @@ -121,11 +406,30 @@ static void test_config_send_receive() { send_cfg->receive_root_directory = str_dup("/send/dst"); send_cfg->save_to_disk = true; send_cfg->use_multithreading = true; - send_cfg->use_chunk_serialization = true; + send_cfg->use_chunk_serialization = false; send_cfg->use_compression = true; send_cfg->use_metadata = true; + send_cfg->use_executability = true; + send_cfg->preserve_hard_links = true; + send_cfg->use_delta = true; + send_cfg->whole_file = true; + send_cfg->fuzzy = true; + send_cfg->ignore_times = true; + send_cfg->size_only = true; send_cfg->compression_level = 5; send_cfg->chunk_size = 1024; + send_cfg->eight_bit_output = true; + send_cfg->modify_window = 4; + send_cfg->existing = true; + send_cfg->ignore_existing = true; + send_cfg->delay_updates = true; + send_cfg->relative = true; + send_cfg->mkpath = true; + send_cfg->skip_compress_set = true; + send_cfg->skip_compress_count = 1; + send_cfg->skip_compress_suffixes = calloc(1, sizeof(char*)); + send_cfg->skip_compress_suffixes[0] = str_dup(".zip"); + send_cfg->max_alloc = MAX_SERVER_ALLOC + 1; /* Use socketpair for bidirectional communication */ int p[2]; @@ -154,12 +458,43 @@ static void test_config_send_receive() { ok = false; if (!recv_cfg->use_multithreading) ok = false; - if (!recv_cfg->use_chunk_serialization) + if (recv_cfg->use_chunk_serialization) ok = false; if (recv_cfg->compression_level != 5) ok = false; if (recv_cfg->chunk_size != 1024) ok = false; + if (!recv_cfg->use_executability) + ok = false; + if (!recv_cfg->preserve_hard_links) + ok = false; + if (!recv_cfg->size_only) + ok = false; + if (!recv_cfg->ignore_times) + ok = false; + if (!recv_cfg->eight_bit_output) + ok = false; + if (recv_cfg->use_delta) + ok = false; + if (!recv_cfg->fuzzy) + ok = false; + if (recv_cfg->modify_window != 4) + ok = false; + if (!recv_cfg->existing) + ok = false; + if (!recv_cfg->ignore_existing) + ok = false; + if (!recv_cfg->delay_updates) + ok = false; + if (!recv_cfg->relative) + ok = false; + if (!recv_cfg->mkpath) + ok = false; + if (!recv_cfg->skip_compress_set || recv_cfg->skip_compress_count != 1 || + strcmp(recv_cfg->skip_compress_suffixes[0], ".zip") != 0) + ok = false; + if (recv_cfg->max_alloc != MAX_SERVER_ALLOC) + ok = false; } config_delete(recv_cfg); close(p[0]); @@ -185,11 +520,11 @@ static void test_config_send_receive() { } static void test_config_send_receive_version_mismatch() { - /* Create a config with a different protocol version */ + /* A peer using the previous wire format must be rejected. */ Config* cfg = config_create(); EXPECT_NOT_NULL(cfg); free(cfg->version); - cfg->version = str_dup("0.0"); + cfg->version = str_dup("2.3.0"); cfg->send_directory = str_dup("/src"); cfg->receive_root_directory = str_dup("/dst"); @@ -224,26 +559,1403 @@ static void test_config_send_receive_version_mismatch() { } } -static void test_is_remote_dest() { +static void test_config_receive_truncated() { + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[0]); + io_set_bwlimit(0); + + /* A valid prefix exercises cleanup after allocated wire strings and a + * partially received scalar field. */ + EXPECT_TRUE(send_str(p[1], PROTOCOL_VERSION)); + unsigned long long max_alloc = DEFAULT_MAX_ALLOC; + EXPECT_TRUE(send_n_data(p[1], &max_alloc, sizeof(max_alloc))); + EXPECT_TRUE(send_str(p[1], "/src")); + EXPECT_TRUE(send_str(p[1], "/dst")); + EXPECT_TRUE(send_int(p[1], 1)); + shutdown(p[1], SHUT_WR); + + const Config* cfg = config_receive(p[0]); + EXPECT_NULL(cfg); + close(p[0]); + close(p[1]); +} + +static bool config_string_roundtrip_matches(const Config* send_cfg, Config* recv) { + /* The sender serializes NULL strings as "" on the wire. Receivers must + canonicalize those empty values back to NULL for the options whose client + default is NULL (backup_dir, temp_dir, partial_dir, suffix), while a real + non-empty value round-trips unchanged. */ + const char* fields[4]; + char* const* recv_fields[4]; + fields[0] = send_cfg->backup_dir; + recv_fields[0] = &recv->backup_dir; + fields[1] = send_cfg->temp_dir; + recv_fields[1] = &recv->temp_dir; + fields[2] = send_cfg->partial_dir; + recv_fields[2] = &recv->partial_dir; + fields[3] = send_cfg->suffix; + recv_fields[3] = &recv->suffix; + for (int i = 0; i < 4; i++) { + const char* sent = fields[i]; + const char* got = *recv_fields[i]; + if (sent == NULL || sent[0] == '\0') { + if (got != NULL) + return false; + } else if (got == NULL || strcmp(sent, got) != 0) { + return false; + } + } + return true; +} + +static bool roundtrip_config_ok(const Config* send_cfg) { + int p[2]; + if (socketpair(AF_UNIX, SOCK_STREAM, 0, p) != 0) + return false; + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool ok = recv != NULL; + if (ok) { + ok = recv->version != NULL && strcmp(recv->version, PROTOCOL_VERSION) == 0; + ok = ok && recv->send_directory && recv->receive_root_directory; + ok = ok && config_string_roundtrip_matches(send_cfg, recv); + } + config_delete(recv); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + return sent && WIFEXITED(status) && WEXITSTATUS(status) == 0; + } +} + +/* Issue #252: NULL-vs-empty must survive the wire for backup_dir, temp_dir, + partial_dir, and suffix. NULL and explicitly-empty client values are both + serialized as "" and must be reconstructed as NULL so plain --backup (with + no --suffix/--backup-dir) works exactly like the client configured it. */ +static void test_config_string_null_vs_empty_roundtrip() { + if (is_running_under_valgrind()) + return; + + /* NULL values on the wire must come back as NULL. */ + Config* a = config_create(); + EXPECT_NOT_NULL(a); + a->send_directory = str_dup("/src"); + a->receive_root_directory = str_dup("/dst"); + EXPECT_TRUE(roundtrip_config_ok(a)); + config_delete(a); + + /* Explicitly empty strings (indistinguishable on the wire from NULL) must + be canonicalized to NULL by the receiver. */ + Config* b = config_create(); + EXPECT_NOT_NULL(b); + b->send_directory = str_dup("/src"); + b->receive_root_directory = str_dup("/dst"); + b->backup_dir = str_dup(""); + b->temp_dir = str_dup(""); + b->partial_dir = str_dup(""); + b->suffix = str_dup(""); + EXPECT_TRUE(roundtrip_config_ok(b)); + config_delete(b); + + /* Non-empty values must round-trip unchanged. */ + Config* c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->backup_dir = str_dup("backups"); + c->temp_dir = str_dup("/tmp/fast"); + c->partial_dir = str_dup(".partial"); + c->suffix = str_dup(".bak"); + EXPECT_TRUE(roundtrip_config_ok(c)); + config_delete(c); +} + +/* A --temp-dir value must survive config_send/config_receive unchanged on the + receive side (round-trips through the resume-options wire block). */ +static void test_config_temp_dir_roundtrip() { + if (is_running_under_valgrind()) + return; + Config* c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->temp_dir = str_dup("scratch"); + EXPECT_TRUE(roundtrip_config_ok(c)); + config_delete(c); + + /* An empty-STRING wire value is canonicalized back to NULL (never an empty + scratch-dir name). */ + c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->temp_dir = str_dup(""); + EXPECT_TRUE(roundtrip_config_ok(c)); + config_delete(c); +} + +static void test_config_delay_updates_reserved_backup_rejected() { + if (is_running_under_valgrind()) + return; + Config* c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->delay_updates = true; + c->backup_dir = str_dup(".fastsync-stage"); + /* The receiver-side wire validation must reject a --backup-dir that collides + with the internal delay-updates staging directory. */ + EXPECT_FALSE(roundtrip_config_ok(c)); + config_delete(c); + + c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->delay_updates = true; + c->backup_dir = str_dup("backups"); + EXPECT_TRUE(roundtrip_config_ok(c)); + config_delete(c); +} + +static void test_config_delete_timing_early_helper() { + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + EXPECT_FALSE(config_delete_timing_early(cfg)); + EXPECT_TRUE(config_has_valid_delete_timing(cfg)); + cfg->use_delete = true; + EXPECT_TRUE(config_has_valid_delete_timing(cfg)); + EXPECT_FALSE(config_delete_timing_early(cfg)); + config_delete(cfg); + + cfg = config_create(); + cfg->use_delete = true; + cfg->delete_before = true; + EXPECT_TRUE(config_delete_timing_early(cfg)); + EXPECT_TRUE(config_has_valid_delete_timing(cfg)); + config_delete(cfg); + + cfg = config_create(); + cfg->use_delete = true; + cfg->delete_during = true; + EXPECT_TRUE(config_delete_timing_early(cfg)); + EXPECT_TRUE(config_has_valid_delete_timing(cfg)); + config_delete(cfg); + + cfg = config_create(); + cfg->use_delete = true; + cfg->delete_delay = true; + EXPECT_FALSE(config_delete_timing_early(cfg)); + EXPECT_TRUE(config_has_valid_delete_timing(cfg)); + config_delete(cfg); + + cfg = config_create(); + cfg->use_delete = true; + cfg->delete_after = true; + EXPECT_FALSE(config_delete_timing_early(cfg)); + EXPECT_TRUE(config_has_valid_delete_timing(cfg)); + config_delete(cfg); + + /* Two simultaneous timings are invalid. */ + cfg = config_create(); + cfg->use_delete = true; + cfg->delete_before = true; + cfg->delete_after = true; + EXPECT_TRUE(config_delete_timing_early(cfg)); + EXPECT_FALSE(config_has_valid_delete_timing(cfg)); + config_delete(cfg); + + /* A timing flag without deletion is invalid. */ + cfg = config_create(); + cfg->delete_delay = true; + EXPECT_FALSE(config_has_valid_delete_timing(cfg)); + EXPECT_FALSE(config_delete_timing_early(cfg)); + config_delete(cfg); +} + +/* New delete-timing fields must survive config_send/config_receive unchanged, + and a config carrying two conflicting timings must be rejected. */ +static void test_config_delete_timing_wire_roundtrip() { + if (is_running_under_valgrind()) + return; + + struct { + bool before, during, delay, after; + } cases[] = { + {false, false, false, false}, {true, false, false, false}, {false, true, false, false}, + {false, false, true, false}, {false, false, false, true}, + }; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool ok = recv != NULL; + if (ok) { + ok = recv->use_delete && recv->delete_before == cases[i].before && + recv->delete_during == cases[i].during && recv->delete_delay == cases[i].delay && + recv->delete_after == cases[i].after; + } + config_delete(recv); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("/dst"); + send_cfg->use_delete = true; + send_cfg->delete_before = cases[i].before; + send_cfg->delete_during = cases[i].during; + send_cfg->delete_delay = cases[i].delay; + send_cfg->delete_after = cases[i].after; + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } + } +} + +/* The receiver-side wire validation rejects a keep-set config with two + conflicting delete-timing flags. */ +static void test_config_delete_timing_conflict_rejected() { + if (is_running_under_valgrind()) + return; + Config* c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->use_delete = true; + c->delete_before = true; + c->delete_delay = true; + EXPECT_FALSE(roundtrip_config_ok(c)); + config_delete(c); +} + +/* The deletion-policy fields that cross the wire survive a config round trip: + --force (force_delete), --delete-excluded, --prune-empty-dirs and the + --max-delete number (default -1 == no client limit). */ +static void test_config_delete_policy_wire_roundtrip() { + if (is_running_under_valgrind()) + return; + + struct { + bool force_delete, delete_excluded, prune_empty_dirs; + int max_delete; + } cases[] = { + {false, false, false, -1}, + {true, false, false, 0}, + {false, true, true, 7}, + }; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool ok = recv != NULL; + if (ok) { + ok = recv->force_delete == cases[i].force_delete && + recv->delete_excluded == cases[i].delete_excluded && + recv->prune_empty_dirs == cases[i].prune_empty_dirs && + recv->max_delete == cases[i].max_delete; + } + config_delete(recv); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("/dst"); + send_cfg->force_delete = cases[i].force_delete; + send_cfg->delete_excluded = cases[i].delete_excluded; + send_cfg->prune_empty_dirs = cases[i].prune_empty_dirs; + send_cfg->max_delete = cases[i].max_delete; + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } + } +} + +/* Phase 4 symlink-trust wire split: --munge-links and -K/--keep-dirlinks CROSS + the wire (the receiver unmunges targets and follows an in-root dir-link), + while -k/--copy-dirlinks is client/sender-only and must NOT reach the + receiver (it would observe it false). */ +static void test_config_symlink_trust_wire_roundtrip() { + if (is_running_under_valgrind()) + return; + + struct { + bool munge_links, keep_dirlinks, copy_dirlinks; + } cases[] = { + {false, false, false}, + {true, false, false}, + {false, true, false}, + {true, true, true}, + }; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool ok = recv != NULL; + if (ok) { + ok = recv->munge_links == cases[i].munge_links && + recv->keep_dirlinks == cases[i].keep_dirlinks && + /* copy_dirlinks never crosses the wire. */ + recv->copy_dirlinks == false; + } + config_delete(recv); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("/dst"); + send_cfg->munge_links = cases[i].munge_links; + send_cfg->keep_dirlinks = cases[i].keep_dirlinks; + send_cfg->copy_dirlinks = cases[i].copy_dirlinks; + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } + } +} +static void test_config_delete_missing_args_wire_roundtrip() { + if (is_running_under_valgrind()) + return; + + struct { + bool delete_missing_args, ignore_missing_args; + } cases[] = { + {false, false}, + {true, false}, + {true, true}, + }; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool ok = recv != NULL; + if (ok) { + ok = recv->delete_missing_args == cases[i].delete_missing_args && + /* ignore_missing_args never crosses the wire. */ + recv->ignore_missing_args == false; + } + config_delete(recv); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("/dst"); + send_cfg->delete_missing_args = cases[i].delete_missing_args; + send_cfg->ignore_missing_args = cases[i].ignore_missing_args; + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } + } +} + +/* Basis-dir lists survive the config wire: each entry's type and path must + round-trip unchanged. */ +static void test_config_basis_roundtrip() { + if (is_running_under_valgrind()) + return; + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/send/src"); + send_cfg->receive_root_directory = str_dup("/send/dst"); + EXPECT_EQ_INT(config_basis_append(send_cfg, BASIS_DEST_LINK, "prior"), 0); + EXPECT_EQ_INT(config_basis_append(send_cfg, BASIS_DEST_COMPARE, "snap/2026-01"), 0); + EXPECT_EQ_INT(config_basis_append(send_cfg, BASIS_DEST_COPY, "copy"), 0); + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool ok = recv != NULL && recv->basis_count == 3 && recv->basis_dirs != NULL; + if (ok) { + ok = recv->basis_dirs[0].type == BASIS_DEST_LINK && + strcmp(recv->basis_dirs[0].path, "prior") == 0; + ok = ok && recv->basis_dirs[1].type == BASIS_DEST_COMPARE && + strcmp(recv->basis_dirs[1].path, "snap/2026-01") == 0; + ok = ok && recv->basis_dirs[2].type == BASIS_DEST_COPY && + strcmp(recv->basis_dirs[2].path, "copy") == 0; + } + config_delete(recv); + close(p[0]); + close(p[1]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +/* The receiver must reject a basis-dir path that would escape the destination + root. The values are injected directly (bypassing the client-side append + validator) so the receiver-side wire validation is what is exercised. */ +static void test_config_basis_wire_rejects_escaping() { + if (is_running_under_valgrind()) + return; + Config* c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->basis_count = 1; + c->basis_dirs = calloc(1, sizeof(BasisDest)); + c->basis_dirs[0].type = BASIS_DEST_LINK; + c->basis_dirs[0].path = str_dup("../../etc"); + EXPECT_FALSE(roundtrip_config_ok(c)); + config_delete(c); + + c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->basis_count = 1; + c->basis_dirs = calloc(1, sizeof(BasisDest)); + c->basis_dirs[0].type = BASIS_DEST_LINK; + c->basis_dirs[0].path = str_dup("/abs"); + EXPECT_FALSE(roundtrip_config_ok(c)); + config_delete(c); + + /* A well-formed list still round-trips even with a manually built struct. */ + c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->basis_count = 1; + c->basis_dirs = calloc(1, sizeof(BasisDest)); + c->basis_dirs[0].type = BASIS_DEST_COPY; + c->basis_dirs[0].path = str_dup("safe"); + EXPECT_TRUE(roundtrip_config_ok(c)); + config_delete(c); +} + +/* Basis-dir paths are canonicalized on the way in: trailing slashes and + interior empty / "." components are dropped so validation, the delete-walker + prefix and the receiver lookup all agree on one stored form. */ +static void test_config_basis_normalization() { + Config* c = config_create(); + EXPECT_NOT_NULL(c); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "prior/"), 0); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "a//b"), 0); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "./x/./y/"), 0); + EXPECT_EQ_INT(c->basis_count, 3); + EXPECT_EQ_STR(c->basis_dirs[0].path, "prior"); + EXPECT_EQ_STR(c->basis_dirs[1].path, "a/b"); + EXPECT_EQ_STR(c->basis_dirs[2].path, "x/y"); + + /* Degenerate values that normalize away to nothing stay rejected. */ + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "."), -1); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, ".."), -1); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "/abs"), -1); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "a/../b"), -1); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, ""), -1); + config_delete(c); +} + +static void test_config_is_remote_dest() { /* Valid SSH-style destinations */ - EXPECT_TRUE(is_remote_dest("user@host:/path")); - EXPECT_TRUE(is_remote_dest("host:/path")); - EXPECT_TRUE(is_remote_dest("user@192.168.1.1:/remote/path")); + EXPECT_TRUE(config_is_remote_dest("user@host:/path")); + EXPECT_TRUE(config_is_remote_dest("host:/path")); + EXPECT_TRUE(config_is_remote_dest("user@192.168.1.1:/remote/path")); /* Invalid destinations */ - EXPECT_FALSE(is_remote_dest(NULL)); - EXPECT_FALSE(is_remote_dest("")); - EXPECT_FALSE(is_remote_dest(":")); - EXPECT_FALSE(is_remote_dest("/local/path")); - EXPECT_FALSE(is_remote_dest("relative/path")); + EXPECT_FALSE(config_is_remote_dest(NULL)); + EXPECT_FALSE(config_is_remote_dest("")); + EXPECT_FALSE(config_is_remote_dest(":")); + EXPECT_FALSE(config_is_remote_dest("/local/path")); + EXPECT_FALSE(config_is_remote_dest("relative/path")); /* C:/windows/path is treated as remote (colon with no preceding slash) */ - EXPECT_TRUE(is_remote_dest("C:/windows/path")); + EXPECT_TRUE(config_is_remote_dest("C:/windows/path")); /* Edge cases */ - EXPECT_FALSE(is_remote_dest("noslash")); - EXPECT_FALSE(is_remote_dest("/")); - EXPECT_TRUE(is_remote_dest("host:")); - EXPECT_TRUE(is_remote_dest("user@host:")); + EXPECT_FALSE(config_is_remote_dest("noslash")); + EXPECT_FALSE(config_is_remote_dest("/")); + EXPECT_TRUE(config_is_remote_dest("host:")); + EXPECT_TRUE(config_is_remote_dest("user@host:")); +} + +/* The append-mode fields cross the wire unchanged: --append and --append-verify + are negotiated to the receiver so it knows to reply STATUS_APPEND on a + shorter destination. */ +static void test_config_append_wire_roundtrip() { + struct { + bool append, append_verify; + } cases[] = {{true, false}, {false, true}, {true, true}, {false, false}}; + if (is_running_under_valgrind()) + return; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool ok = recv != NULL; + if (ok) + ok = recv->append == cases[i].append && recv->append_verify == cases[i].append_verify; + config_delete(recv); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("/dst"); + send_cfg->append = cases[i].append; + send_cfg->append_verify = cases[i].append_verify; + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } + } +} + +/* --checksum-choice/--cc and --checksum-seed cross the wire intact so the + receiver hashes the on-disk old file with the same algorithm and seed. */ +static void test_config_checksum_options_wire_roundtrip() { + struct { + int algo; + unsigned long long seed; + } cases[] = { + {CHECKSUM_ALGO_XXH64, 0}, + {CHECKSUM_ALGO_XXH64, 42}, + {CHECKSUM_ALGO_MD5, 7}, + {CHECKSUM_ALGO_MD5, 0}, + }; + if (is_running_under_valgrind()) + return; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool ok = recv != NULL && recv->checksum_algo == cases[i].algo && + recv->checksum_seed == cases[i].seed; + config_delete(recv); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("/dst"); + send_cfg->checksum_algo = cases[i].algo; + send_cfg->checksum_seed = cases[i].seed; + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } + } +} + +/* An out-of-range algorithm id on the wire must be rejected on receive, never + accepted as-is (prevents mixing unsupported digests on a path). */ +static void test_config_receive_rejects_invalid_checksum_algo() { + if (is_running_under_valgrind()) + return; + Config* c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->checksum_algo = 99; + EXPECT_FALSE(roundtrip_config_ok(c)); + config_delete(c); +} +/* The identity-mapping fields (--numeric-ids / --usermap / --groupmap / + --chown) cross the config wire unchanged: the receiver needs them to apply + ownership with the same policy the client requested. */ +static void test_config_metadata_times_wire_roundtrip() { + if (is_running_under_valgrind()) + return; + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/send/src"); + send_cfg->receive_root_directory = str_dup("/send/dst"); + send_cfg->preserve_atimes = true; + send_cfg->preserve_crtimes = true; + send_cfg->omit_dir_times = true; + send_cfg->omit_link_times = true; + /* --open-noatime is client-only and must NOT cross the wire. */ + send_cfg->open_noatime = true; + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool ok = recv != NULL; + if (ok) { + ok = recv->preserve_atimes && recv->preserve_crtimes && recv->omit_dir_times && + recv->omit_link_times && !recv->open_noatime; + } + config_delete(recv); + close(p[0]); + close(p[1]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +static void test_config_identity_wire_roundtrip() { + if (is_running_under_valgrind()) + return; + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/send/src"); + send_cfg->receive_root_directory = str_dup("/send/dst"); + send_cfg->numeric_ids = true; + send_cfg->chown_uid_set = true; + send_cfg->chown_uid = 1001; + send_cfg->chown_gid_set = true; + send_cfg->chown_gid = IDENTITY_CURRENT; + send_cfg->usermap_count = 2; + send_cfg->usermap = calloc(2, sizeof(IdentityMap)); + send_cfg->usermap[0].from = IDENTITY_MATCH_ANY; + send_cfg->usermap[0].to = 65534; + send_cfg->usermap[1].from = 1000; + send_cfg->usermap[1].to = 1000; + send_cfg->groupmap_count = 1; + send_cfg->groupmap = calloc(1, sizeof(IdentityMap)); + send_cfg->groupmap[0].from = 0; + send_cfg->groupmap[0].to = IDENTITY_CURRENT; + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool ok = recv != NULL; + if (ok) { + ok = recv->numeric_ids && recv->chown_uid_set && recv->chown_uid == 1001 && + recv->chown_gid_set && recv->chown_gid == IDENTITY_CURRENT && recv->usermap_count == 2 && + recv->groupmap_count == 1 && recv->usermap[0].from == IDENTITY_MATCH_ANY && + recv->usermap[0].to == 65534 && recv->usermap[1].from == 1000 && + recv->usermap[1].to == 1000 && recv->groupmap[0].from == 0 && + recv->groupmap[0].to == IDENTITY_CURRENT; + } + config_delete(recv); + close(p[0]); + close(p[1]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +/* The receiver must reject an out-of-range identity-map count or id on the + wire (defense against a malicious/oversized table). */ +static void test_config_receive_rejects_invalid_identity() { + if (is_running_under_valgrind()) + return; + Config* c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->usermap_count = 1; + c->usermap = calloc(1, sizeof(IdentityMap)); + c->usermap[0].from = -2; /* below IDENTITY_MATCH_ANY */ + c->usermap[0].to = 0; + EXPECT_FALSE(roundtrip_config_ok(c)); + config_delete(c); + + c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->chown_uid_set = true; + c->chown_uid = -5; + EXPECT_FALSE(roundtrip_config_ok(c)); + config_delete(c); + + /* A well-formed identity config still round-trips through the shared helper. */ + c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->numeric_ids = true; + EXPECT_TRUE(roundtrip_config_ok(c)); + config_delete(c); +} + +/* --preallocate crosses the wire unchanged (receiver-side flag): the receiver + must learn to allocate the destination file's space before data flows. */ +static void test_config_preallocate_wire_roundtrip() { + struct { + bool preallocate; + } cases[] = {{false}, {true}}; + if (is_running_under_valgrind()) + return; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool ok = recv != NULL && recv->preallocate == cases[i].preallocate; + config_delete(recv); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("/dst"); + send_cfg->preallocate = cases[i].preallocate; + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } + } +} +static void test_config_devices_wire_roundtrip() { + if (is_running_under_valgrind()) + return; + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/send/src"); + send_cfg->receive_root_directory = str_dup("/send/dst"); + send_cfg->preserve_devices = true; + send_cfg->preserve_specials = true; + send_cfg->copy_devices = true; + send_cfg->write_devices = true; + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool ok = recv != NULL; + if (ok) { + ok = recv->preserve_devices && recv->preserve_specials && recv->copy_devices && + recv->write_devices; + } + config_delete(recv); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +/* Phase-4: preserve_xattrs/--acls (in file options) and --fake-super (trailing) + * cross the config wire; the receiver recomputes the derived use_xattrs. */ +static void test_config_phase4_xattr_wire_roundtrip() { + if (is_running_under_valgrind()) + return; + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/send/src"); + send_cfg->receive_root_directory = str_dup("/send/dst"); + send_cfg->preserve_xattrs = true; + send_cfg->preserve_acls = true; + send_cfg->fake_super = true; + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool ok = recv != NULL; + if (ok) { + ok = recv->preserve_xattrs && recv->preserve_acls && recv->fake_super && recv->use_xattrs; + } + config_delete(recv); + close(p[0]); + close(p[1]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +/* --trust-sender defaults to OFF (a receiver-local policy). */ +static void test_config_trust_sender_default_false() { + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + EXPECT_FALSE(cfg->trust_sender); + EXPECT_NULL(cfg->remote_options); + EXPECT_EQ_INT(cfg->remote_option_count, 0); + config_delete(cfg); +} + +/* --trust-sender and --remote-option are LOCAL to the process that sets them: + * they must never cross the wire. After a round-trip the receiver observes the + * neutral defaults (trust_sender=false, no remote options), even when the + * sender had them set. */ +static void test_config_local_only_fields_not_serialized() { + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool ok = recv != NULL && !recv->trust_sender && recv->remote_options == NULL && + recv->remote_option_count == 0; + config_delete(recv); + close(p[0]); + _exit(ok ? 0 : 1); + } + + close(p[0]); + io_set_fds(p[1], p[1]); + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->trust_sender = true; + /* remote_options is client-side state; populate it like the CLI would. */ + send_cfg->remote_options = malloc(sizeof(char*)); + send_cfg->remote_options[0] = str_dup("--allow-delete"); + send_cfg->remote_option_count = 1; + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("/dst"); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); +} + +static void test_config_iconv_spec_wire_roundtrip() { + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("rel/path"); + send_cfg->iconv_spec = str_dup("utf-8,iso-8859-1"); + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv_cfg = config_receive(p[0]); + bool ok = recv_cfg != NULL && recv_cfg->iconv_spec != NULL && + strcmp(recv_cfg->iconv_spec, "utf-8,iso-8859-1") == 0; + config_delete(recv_cfg); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +static void test_config_iconv_spec_empty_canonicalizes_to_null() { + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("/dst"); + /* iconv_spec left NULL -> serialized as "" -> received back as NULL. */ + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv_cfg = config_receive(p[0]); + bool ok = recv_cfg != NULL && recv_cfg->iconv_spec == NULL; + config_delete(recv_cfg); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +static void test_config_receive_rejects_invalid_iconv_spec() { + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("/dst"); + send_cfg->iconv_spec = str_dup("no-such-charset,utf-8"); + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + /* A malformed/unsupported spec must be refused at the config handshake + (STATUS_ERROR makes config_send fail on the parent). */ + Config* recv_cfg = config_receive(p[0]); + config_delete(recv_cfg); + close(p[0]); + _exit(recv_cfg ? 1 : 0); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_FALSE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +/* P7 Wave E: the --super / --no-super tri-state crosses the config wire + unchanged (AUTO/ON/OFF), so the receiver can enforce the privilege policy. */ +static void test_config_super_mode_wire_roundtrip() { + if (is_running_under_valgrind()) + return; + int modes[] = {SUPER_MODE_AUTO, SUPER_MODE_ON, SUPER_MODE_OFF}; + for (size_t i = 0; i < sizeof(modes) / sizeof(modes[0]); i++) { + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool ok = recv != NULL && recv->super_mode == modes[i]; + config_delete(recv); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("/dst"); + send_cfg->super_mode = modes[i]; + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } + } +} + +/* --copy-as (P7 Wave E, protocol 2.18.0) travels as a trailing config-frame + block: a presence int, then the two int32 ids when set. */ +static void test_config_copy_as_wire_roundtrip() { + struct { + bool set; + int32_t uid; + int32_t gid; + } cases[] = {{false, 0, 0}, {true, 1000, 1001}}; + if (is_running_under_valgrind()) + return; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool ok = recv != NULL && recv->copy_as_set == cases[i].set && + (!cases[i].set || + (recv->copy_as_uid == cases[i].uid && recv->copy_as_gid == cases[i].gid)); + config_delete(recv); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("/dst"); + send_cfg->copy_as_set = cases[i].set; + send_cfg->copy_as_uid = cases[i].uid; + send_cfg->copy_as_gid = cases[i].gid; + /* --copy-as requires the metadata path (the receiver chowns from the + transmitted source ids); a raw frame with copy_as_set but no metadata + is now rejected by validate_received_config. */ + send_cfg->use_metadata = cases[i].set; + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } + } +} + +/* An out-of-range super_mode value on the wire must be refused on receive + (never silently clamped or accepted). */ +static void test_config_receive_rejects_invalid_super_mode() { + if (is_running_under_valgrind()) + return; + Config* c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->super_mode = 99; + EXPECT_FALSE(roundtrip_config_ok(c)); + config_delete(c); + + /* A negative value is equally invalid. */ + c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->super_mode = -1; + EXPECT_FALSE(roundtrip_config_ok(c)); + config_delete(c); +} + +/* A hostile peer must not smuggle a negative (sentinel) copy-as id into the + ownership path: the receive side rejects it and the run fails the handshake. */ +static void test_config_receive_rejects_negative_copy_as() { + if (is_running_under_valgrind()) + return; + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/src"); + send_cfg->receive_root_directory = str_dup("/dst"); + send_cfg->copy_as_set = true; + send_cfg->copy_as_uid = -1; + send_cfg->copy_as_gid = 0; + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv_cfg = config_receive(p[0]); + config_delete(recv_cfg); + close(p[0]); + _exit(recv_cfg ? 1 : 0); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_FALSE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +/* --copy-as forces ownership through the metadata path. A frame that sets + copy_as_set but not use_metadata would pass the receiver's privilege gate + while chowning nothing, so validate_received_config must reject it (and the + sender observes the rejection as a failed config_send). */ +static void test_config_receive_rejects_copy_as_without_metadata() { + if (is_running_under_valgrind()) + return; + Config* c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->copy_as_set = true; + c->copy_as_uid = 1000; + c->copy_as_gid = 1000; + c->use_metadata = false; + EXPECT_FALSE(roundtrip_config_ok(c)); + config_delete(c); + + /* With metadata enabled the same block is accepted. */ + c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->copy_as_set = true; + c->copy_as_uid = 1000; + c->copy_as_gid = 1000; + c->use_metadata = true; + EXPECT_TRUE(roundtrip_config_ok(c)); + config_delete(c); +} + +/* identity_copy_as_refused() is the pure, pre-snapshot refusal predicate: a + --copy-as is refused when the receiver is not root OR the effective super + mode is OFF (an operator veto), and never when --copy-as is unset. */ +static void test_identity_copy_as_refused() { + Config* c = config_create(); + EXPECT_NOT_NULL(c); + EXPECT_FALSE(identity_copy_as_refused(c)); + EXPECT_FALSE(identity_copy_as_refused(NULL)); + + c->copy_as_set = true; + c->super_mode = SUPER_MODE_AUTO; + if (geteuid() == 0) { + EXPECT_FALSE(identity_copy_as_refused(c)); /* AUTO permits as root */ + c->super_mode = SUPER_MODE_ON; + EXPECT_FALSE(identity_copy_as_refused(c)); + c->super_mode = SUPER_MODE_OFF; + EXPECT_TRUE(identity_copy_as_refused(c)); + } else { + /* Unprivileged: refused regardless of the mode. */ + EXPECT_TRUE(identity_copy_as_refused(c)); + c->super_mode = SUPER_MODE_OFF; + EXPECT_TRUE(identity_copy_as_refused(c)); + } + config_delete(c); +} + +/* P7 Wave E: privilege_super_permitted() maps the super_mode tri-state. OFF + forbids super-user activities even for root; ON and AUTO permit the confined + attempt (matching FastSync's historical best-effort behavior, where the kernel + refuses an unprivileged attempt and the caller skips it). */ +static void test_privilege_super_permitted_modes() { + Config* c = config_create(); + EXPECT_NOT_NULL(c); + c->super_mode = SUPER_MODE_OFF; + EXPECT_TRUE(identity_set_active(c)); + EXPECT_FALSE(privilege_super_permitted()); + c->super_mode = SUPER_MODE_ON; + EXPECT_TRUE(identity_set_active(c)); + EXPECT_TRUE(privilege_super_permitted()); + c->super_mode = SUPER_MODE_AUTO; + EXPECT_TRUE(identity_set_active(c)); + EXPECT_TRUE(privilege_super_permitted()); + config_delete(c); + + /* After clearing, the neutral default is AUTO (attempt), never a stale + snapshot from a previous connection. */ + identity_clear_active(); + EXPECT_TRUE(privilege_super_permitted()); +} + +/* P7 Wave E hardening (A1): identity_ownership_requested() is the pure, + config-only predicate the daemon module gate uses. It must fire for every + client-chosen ownership / super-user request and stay false for a plain + transfer and for SUPER_MODE_AUTO (the default) alone. */ +static void test_identity_ownership_requested() { + EXPECT_FALSE(identity_ownership_requested(NULL)); + + Config* c = config_create(); + EXPECT_NOT_NULL(c); + EXPECT_FALSE(identity_ownership_requested(c)); + c->super_mode = SUPER_MODE_AUTO; + EXPECT_FALSE(identity_ownership_requested(c)); /* AUTO alone is not ownership */ + c->super_mode = SUPER_MODE_ON; + EXPECT_TRUE(identity_ownership_requested(c)); /* explicit --super is */ + c->super_mode = SUPER_MODE_AUTO; + + c->numeric_ids = true; + EXPECT_TRUE(identity_ownership_requested(c)); + c->numeric_ids = false; + c->chown_uid_set = true; + EXPECT_TRUE(identity_ownership_requested(c)); + c->chown_uid_set = false; + c->chown_gid_set = true; + EXPECT_TRUE(identity_ownership_requested(c)); + c->chown_gid_set = false; + c->copy_as_set = true; + EXPECT_TRUE(identity_ownership_requested(c)); + c->copy_as_set = false; + c->fake_super = true; + EXPECT_TRUE(identity_ownership_requested(c)); + config_delete(c); + + Config* um = config_create(); + EXPECT_NOT_NULL(um); + EXPECT_EQ_INT(identity_parse_map(um, "@1:@2", false), 0); + EXPECT_TRUE(identity_ownership_requested(um)); + config_delete(um); + + Config* gm = config_create(); + EXPECT_NOT_NULL(gm); + EXPECT_EQ_INT(identity_parse_map(gm, "@1:@2", true), 0); + EXPECT_TRUE(identity_ownership_requested(gm)); + config_delete(gm); +} + +/* P7 Wave E hardening (A3): --super no longer implies raw numeric-id + preservation, so it must never enable ownership application on its own; an + explicit identity flag is required. */ +static void test_super_does_not_imply_numeric() { + Config* c = config_create(); + EXPECT_NOT_NULL(c); + c->super_mode = SUPER_MODE_ON; + c->use_metadata = true; + EXPECT_TRUE(identity_set_active(c)); + EXPECT_FALSE(identity_active_enabled()); + c->numeric_ids = true; + EXPECT_TRUE(identity_set_active(c)); + EXPECT_TRUE(identity_active_enabled()); + identity_clear_active(); + config_delete(c); } void test_config() { @@ -251,11 +1963,58 @@ void test_config() { test_config_ssh_dest(); test_config_ssh_dest_local_path(); test_config_ssh_dest_no_user(); + test_config_daemon_dest_parse(); + test_config_daemon_dest_no_path(); + test_config_daemon_dest_double_slash_normalized(); + test_config_daemon_dest_bad(); + test_config_transport_dest_daemon_beats_ssh(); + test_config_is_daemon_dest(); + test_config_trust_sender_default_false(); test_pipeline_sender_lifecycle(); test_pipeline_receiver_lifecycle(); if (!is_running_under_valgrind()) { test_config_send_receive(); + test_config_local_only_fields_not_serialized(); test_config_send_receive_version_mismatch(); + test_config_receive_truncated(); + test_config_string_null_vs_empty_roundtrip(); + test_config_temp_dir_roundtrip(); + test_config_delay_updates_reserved_backup_rejected(); + test_config_delete_timing_wire_roundtrip(); + test_config_delete_timing_conflict_rejected(); + test_config_delete_policy_wire_roundtrip(); + test_config_symlink_trust_wire_roundtrip(); + test_config_delete_missing_args_wire_roundtrip(); + test_config_append_wire_roundtrip(); + test_config_basis_roundtrip(); + test_config_basis_wire_rejects_escaping(); + test_config_basis_normalization(); + test_config_checksum_options_wire_roundtrip(); + test_config_receive_rejects_invalid_checksum_algo(); + test_config_identity_wire_roundtrip(); + test_config_receive_rejects_invalid_identity(); + test_config_metadata_times_wire_roundtrip(); + test_config_devices_wire_roundtrip(); + test_config_preallocate_wire_roundtrip(); + test_config_phase4_xattr_wire_roundtrip(); + test_config_module_wire_roundtrip(); + test_config_module_wire_empty_canonicalizes_to_null(); + test_config_daemon_auth_wire_roundtrip(); + test_config_daemon_auth_wire_rejects_malformed(); + test_config_iconv_spec_wire_roundtrip(); + test_config_iconv_spec_empty_canonicalizes_to_null(); + test_config_receive_rejects_invalid_iconv_spec(); + test_config_super_mode_wire_roundtrip(); + test_config_receive_rejects_invalid_super_mode(); + test_config_copy_as_wire_roundtrip(); + test_config_receive_rejects_negative_copy_as(); + test_config_receive_rejects_copy_as_without_metadata(); + test_config_receive_with_validate_rejects(); } - test_is_remote_dest(); + test_identity_copy_as_refused(); + test_identity_ownership_requested(); + test_super_does_not_imply_numeric(); + test_privilege_super_permitted_modes(); + test_config_delete_timing_early_helper(); + test_config_is_remote_dest(); } diff --git a/tests/test_credentials.c b/tests/test_credentials.c new file mode 100644 index 0000000..86d4dd6 --- /dev/null +++ b/tests/test_credentials.c @@ -0,0 +1,1086 @@ +#include "test_credentials.h" +#include "credentials.h" +#include "test_utils.h" +#include "utils.h" +#include +#include +#include +#include +#include +#include +#include +#include + +/* Known-answer vector, independently recomputed with Python + * (hashlib.pbkdf2_hmac / hmac / hashlib.sha256) at the default work factor. */ +#define KAT_PASSWORD "alice-s3cret" +#define KAT_USER "alice" +#define KAT_ITERS CREDENTIAL_DEFAULT_ITERS +#define KAT_CLIENT_KEY "845891d65ab3c9807f7ae5c123ab70714cc8b56173fccfce6a758993e858e17c" +#define KAT_STORED_KEY "d192f6da1c54bf73768f0a7c713995212d303c46f809de1b2e407fb3ad1c206b" +#define KAT_SERVER_KEY "508ad587574f59700c2d0bbec8417d6fe94bf8e39669adbabe7cbbd8700f86ce" +#define KAT_CLIENT_PROOF "e0cb4b894a7438d75cbb3066aa135d10200b76eea78137c5c04059895eed9242" +#define KAT_SERVER_SIG "f564e00fa6368e78d35b7116c7624d6cb047a950d87e3799e4e6e8c8954b618a" +#define KAT_SALT_B64 "AAECAwQFBgcICQoLDA0ODw==" +#define KAT_NAME_PREFIX "$fastsync$1$pbkdf2-sha256$" +/* The exact store line for the KAT user/password at the KAT salt/count. */ +#define KAT_STORE_LINE \ + "alice:$fastsync$1$pbkdf2-sha256$600000$AAECAwQFBgcICQoLDA0ODw==$" \ + "0ZL22hxUv3N2jwp8cTmVIS0wPEb4Cd4bLkB/s60cIGs=$UIrVh1dPWXAMLQu+yEF9b+lL+OOWaa26vny72HAPhs4=" + +static int g_file_counter = 0; + +static int hex_nibble(char c) { + if (c >= '0' && c <= '9') + return c - '0'; + if (c >= 'a' && c <= 'f') + return c - 'a' + 10; + if (c >= 'A' && c <= 'F') + return c - 'A' + 10; + return -1; +} + +static void unhex(const char* hex, uint8_t* out, size_t out_len) { + for (size_t i = 0; i < out_len; i++) + out[i] = (uint8_t)((hex_nibble(hex[2 * i]) << 4) | hex_nibble(hex[2 * i + 1])); +} + +/* Fill deterministic nonces: out[i] = first + i. */ +static void ramp(uint8_t* out, size_t len, uint8_t first) { + for (size_t i = 0; i < len; i++) + out[i] = (uint8_t)(first + i); +} + +static char* make_tmp_file(const char* contents) { + char path[256]; + snprintf(path, sizeof(path), "/tmp/fs_cred_test_%d_%d", (int)getpid(), g_file_counter++); + FILE* fp = fopen(path, "w"); + if (!fp) + return NULL; + size_t n = strlen(contents); + if (n > 0 && fwrite(contents, 1, n, fp) != n) { + fclose(fp); + unlink(path); + return NULL; + } + fclose(fp); + /* Credential/password files are owner-only; the reader rejects group/other + * permission bits, so create temp files 0600 like the real ones. */ + chmod(path, 0600); + return str_dup(path); +} + +static void rm_temp(const char* path) { + if (!path) + return; + unlink(path); + /* Every successfully loaded store auto-creates an exact-mode-0600 + * `.dummykey` sidecar; remove it too so tests leave no stray key. The + * atomic-publish temps carry a random suffix, so glob them all and remove any + * that a failing path may have left behind. */ + size_t n = strlen(path) + strlen(".dummykey") + 1; + char* sidecar = malloc(n); + if (sidecar) { + snprintf(sidecar, n, "%s.dummykey", path); + unlink(sidecar); + free(sidecar); + } + n = strlen(path) + strlen(".dummykey.tmp.*") + 1; + char* pattern = malloc(n); + if (pattern) { + snprintf(pattern, n, "%s.dummykey.tmp.*", path); + glob_t matches; + memset(&matches, 0, sizeof(matches)); + if (glob(pattern, 0, NULL, &matches) == 0) { + for (size_t i = 0; i < matches.gl_pathc; i++) + unlink(matches.gl_pathv[i]); + } + globfree(&matches); + free(pattern); + } +} + +/* `.dummykey` sidecar path (caller frees). */ +static char* dummy_sidecar_path(const char* store_path) { + size_t n = strlen(store_path) + strlen(".dummykey") + 1; + char* out = malloc(n); + if (!out) + return NULL; + snprintf(out, n, "%s.dummykey", store_path); + return out; +} + +/* Build a valid new-format line for user/password at iters. */ +static bool make_store_line(const char* user, const char* password, uint32_t iters, char* out, + size_t out_sz) { + char err[256]; + return credentials_hash_store_line(user, password, iters, out, out_sz, err, sizeof(err)); +} + +static void test_credentials_secure_equal() { + EXPECT_TRUE(credentials_secure_equal("abc", "abc", 3)); + EXPECT_TRUE(credentials_secure_equal("", "", 0)); + EXPECT_FALSE(credentials_secure_equal("abc", "abd", 3)); + EXPECT_TRUE(credentials_secure_equal("abc", "ab", 2)); + EXPECT_FALSE(credentials_secure_equal("ab", "ac", 2)); +} + +static void test_credentials_b64() { + uint8_t salt[CREDENTIAL_SALT_LEN]; + ramp(salt, sizeof(salt), 0x00); + char encoded[25]; + EXPECT_TRUE(credentials_b64_encode(salt, sizeof(salt), encoded, sizeof(encoded))); + EXPECT_EQ_STR(encoded, KAT_SALT_B64); + + uint8_t decoded[CREDENTIAL_SALT_LEN]; + size_t decoded_len = 0; + EXPECT_TRUE(credentials_b64_decode(encoded, decoded, sizeof(decoded), &decoded_len)); + EXPECT_EQ_INT((int)decoded_len, CREDENTIAL_SALT_LEN); + EXPECT_TRUE(memcmp(decoded, salt, sizeof(salt)) == 0); + + /* Malformed input is refused: bad length, bad alphabet, missing buffer. */ + EXPECT_FALSE(credentials_b64_decode("abc", decoded, sizeof(decoded), &decoded_len)); + EXPECT_FALSE(credentials_b64_decode("!!!!", decoded, sizeof(decoded), &decoded_len)); + EXPECT_FALSE(credentials_b64_decode("", decoded, sizeof(decoded), &decoded_len)); + EXPECT_FALSE(credentials_b64_decode(KAT_SALT_B64, decoded, 4, &decoded_len)); + EXPECT_FALSE(credentials_b64_decode(NULL, decoded, sizeof(decoded), &decoded_len)); + EXPECT_FALSE(credentials_b64_encode(NULL, 3, encoded, sizeof(encoded))); + EXPECT_FALSE(credentials_b64_encode(salt, sizeof(salt), encoded, 3)); +} + +static void test_credentials_random_bytes() { + uint8_t a[CREDENTIAL_NONCE_LEN]; + uint8_t b[CREDENTIAL_NONCE_LEN]; + EXPECT_TRUE(credentials_random_bytes(a, sizeof(a))); + EXPECT_TRUE(credentials_random_bytes(b, sizeof(b))); + EXPECT_TRUE(memcmp(a, b, sizeof(a)) != 0); + EXPECT_FALSE(credentials_random_bytes(NULL, 4)); +} + +static void test_credentials_compute_keys_kat() { + uint8_t salt[CREDENTIAL_SALT_LEN]; + ramp(salt, sizeof(salt), 0x00); + uint8_t client_key[CREDENTIAL_KEY_LEN]; + uint8_t stored_key[CREDENTIAL_KEY_LEN]; + uint8_t server_key[CREDENTIAL_KEY_LEN]; + EXPECT_TRUE( + credentials_compute_keys(KAT_PASSWORD, salt, KAT_ITERS, client_key, stored_key, server_key)); + uint8_t expect[CREDENTIAL_KEY_LEN]; + unhex(KAT_CLIENT_KEY, expect, sizeof(expect)); + EXPECT_TRUE(memcmp(client_key, expect, sizeof(expect)) == 0); + unhex(KAT_STORED_KEY, expect, sizeof(expect)); + EXPECT_TRUE(memcmp(stored_key, expect, sizeof(expect)) == 0); + unhex(KAT_SERVER_KEY, expect, sizeof(expect)); + EXPECT_TRUE(memcmp(server_key, expect, sizeof(expect)) == 0); + EXPECT_FALSE(credentials_compute_keys(KAT_PASSWORD, salt, 0, client_key, stored_key, server_key)); + EXPECT_FALSE(credentials_compute_keys(KAT_PASSWORD, salt, CREDENTIAL_MIN_ITERS - 1, client_key, + stored_key, server_key)); + EXPECT_FALSE(credentials_compute_keys(KAT_PASSWORD, salt, CREDENTIAL_MAX_ITERS + 1, client_key, + stored_key, server_key)); + EXPECT_FALSE(credentials_compute_keys(NULL, salt, KAT_ITERS, client_key, stored_key, server_key)); +} + +static void test_credentials_auth_message_and_proof_kat() { + uint8_t snonce[CREDENTIAL_NONCE_LEN]; + uint8_t cnonce[CREDENTIAL_NONCE_LEN]; + ramp(snonce, sizeof(snonce), 0xa0); + ramp(cnonce, sizeof(cnonce), 0x10); + + uint8_t auth_msg[CREDENTIAL_AUTH_MESSAGE_MAX]; + size_t msg_len = 0; + EXPECT_TRUE(credentials_build_auth_message(KAT_USER, snonce, cnonce, auth_msg, sizeof(auth_msg), + &msg_len)); + const char* expect_msg = + "4661737453796e632d417574682d763100000005616c69636500000020a0a1a2a3a4a5a6a7a8a9aaabac" + "adaeafb0b1b2b3b4b5b6b7b8b9babbbcbdbebf00000020101112131415161718191a1b1c1d1e1f2021" + "22232425262728292a2b2c2d2e2f"; + uint8_t expect[CREDENTIAL_AUTH_MESSAGE_MAX]; + size_t expect_len = strlen(expect_msg) / 2; + unhex(expect_msg, expect, expect_len); + EXPECT_EQ_INT((int)msg_len, (int)expect_len); + EXPECT_TRUE(memcmp(auth_msg, expect, expect_len) == 0); + + uint8_t client_key[CREDENTIAL_KEY_LEN]; + uint8_t stored_key[CREDENTIAL_KEY_LEN]; + uint8_t server_key[CREDENTIAL_KEY_LEN]; + unhex(KAT_CLIENT_KEY, client_key, sizeof(client_key)); + unhex(KAT_STORED_KEY, stored_key, sizeof(stored_key)); + unhex(KAT_SERVER_KEY, server_key, sizeof(server_key)); + uint8_t proof[CREDENTIAL_KEY_LEN]; + uint8_t server_sig[CREDENTIAL_KEY_LEN]; + EXPECT_TRUE(credentials_client_proof(client_key, stored_key, server_key, auth_msg, msg_len, proof, + server_sig)); + unhex(KAT_CLIENT_PROOF, expect, CREDENTIAL_KEY_LEN); + EXPECT_TRUE(memcmp(proof, expect, CREDENTIAL_KEY_LEN) == 0); + unhex(KAT_SERVER_SIG, expect, CREDENTIAL_KEY_LEN); + EXPECT_TRUE(memcmp(server_sig, expect, CREDENTIAL_KEY_LEN) == 0); +} + +static CredentialVerifier kat_verifier(void) { + CredentialVerifier v; + memset(&v, 0, sizeof(v)); + ramp(v.salt, sizeof(v.salt), 0x00); + v.iters = KAT_ITERS; + unhex(KAT_STORED_KEY, v.stored_key, sizeof(v.stored_key)); + unhex(KAT_SERVER_KEY, v.server_key, sizeof(v.server_key)); + v.found = true; + return v; +} + +static void test_credentials_verify_response_kat() { + CredentialVerifier v = kat_verifier(); + uint8_t snonce[CREDENTIAL_NONCE_LEN]; + uint8_t cnonce[CREDENTIAL_NONCE_LEN]; + ramp(snonce, sizeof(snonce), 0xa0); + ramp(cnonce, sizeof(cnonce), 0x10); + uint8_t proof[CREDENTIAL_KEY_LEN]; + uint8_t expect_sig[CREDENTIAL_KEY_LEN]; + unhex(KAT_CLIENT_PROOF, proof, sizeof(proof)); + unhex(KAT_SERVER_SIG, expect_sig, sizeof(expect_sig)); + + uint8_t server_sig[CREDENTIAL_KEY_LEN]; + EXPECT_TRUE(credentials_verify_response(&v, KAT_USER, snonce, cnonce, proof, server_sig)); + EXPECT_TRUE(memcmp(server_sig, expect_sig, CREDENTIAL_KEY_LEN) == 0); + + /* Tampered proof refused. */ + uint8_t bad[CREDENTIAL_KEY_LEN]; + memcpy(bad, proof, sizeof(bad)); + bad[0] ^= 0x01; + EXPECT_FALSE(credentials_verify_response(&v, KAT_USER, snonce, cnonce, bad, server_sig)); + + /* Unit replay: the same proof bound to a different client nonce is refused. */ + uint8_t other[CREDENTIAL_NONCE_LEN]; + memcpy(other, cnonce, sizeof(other)); + other[0] ^= 0x01; + EXPECT_FALSE(credentials_verify_response(&v, KAT_USER, snonce, other, proof, server_sig)); + /* Different server nonce too. */ + uint8_t other_server[CREDENTIAL_NONCE_LEN]; + memcpy(other_server, snonce, sizeof(other_server)); + other_server[0] ^= 0x01; + EXPECT_FALSE(credentials_verify_response(&v, KAT_USER, other_server, cnonce, proof, server_sig)); + + /* Wrong user changes the AuthMessage and fails. */ + EXPECT_FALSE(credentials_verify_response(&v, "bob", snonce, cnonce, proof, server_sig)); + + /* found=false never accepts. */ + v.found = false; + EXPECT_FALSE(credentials_verify_response(&v, KAT_USER, snonce, cnonce, proof, server_sig)); + + /* NULL arguments fail closed. */ + v.found = true; + EXPECT_FALSE(credentials_verify_response(NULL, KAT_USER, snonce, cnonce, proof, server_sig)); + EXPECT_FALSE(credentials_verify_response(&v, NULL, snonce, cnonce, proof, server_sig)); + EXPECT_FALSE(credentials_verify_response(&v, KAT_USER, NULL, cnonce, proof, server_sig)); + EXPECT_FALSE(credentials_verify_response(&v, KAT_USER, snonce, cnonce, NULL, server_sig)); +} + +static void test_credentials_username_valid() { + EXPECT_TRUE(credentials_username_valid("alice")); + EXPECT_TRUE(credentials_username_valid("a")); + EXPECT_FALSE(credentials_username_valid(NULL)); + EXPECT_FALSE(credentials_username_valid("")); + EXPECT_FALSE(credentials_username_valid("bad user")); + EXPECT_FALSE(credentials_username_valid("tab\there")); + EXPECT_FALSE(credentials_username_valid("nul\nhere")); +} + +/* A generated line round-trips through the store parser and verifies with the + * same password. */ +static void test_credentials_hash_store_line_roundtrip() { + char line[CREDENTIAL_MAX_LINE]; + EXPECT_TRUE(make_store_line("alice", KAT_PASSWORD, CREDENTIAL_MIN_ITERS, line, sizeof(line))); + EXPECT_TRUE(strncmp(line, "alice:", 6) == 0); + EXPECT_TRUE(strstr(line, KAT_NAME_PREFIX) != NULL); + + char* path = make_tmp_file(line); + EXPECT_NOT_NULL(path); + char err[512]; + CredentialStore* store = credentials_load(path, NULL, err, sizeof(err)); + EXPECT_NOT_NULL(store); + EXPECT_EQ_INT(credentials_store_size(store), 1); + EXPECT_TRUE(credentials_store_has(store, "alice")); + + CredentialVerifier v; + const char* module_users[] = {"alice"}; + EXPECT_TRUE(credentials_get_verifier(store, "alice", module_users, 1, &v)); + EXPECT_TRUE(v.found); + EXPECT_EQ_INT((int)v.iters, (int)CREDENTIAL_MIN_ITERS); + uint8_t client_key[CREDENTIAL_KEY_LEN]; + uint8_t stored_key[CREDENTIAL_KEY_LEN]; + uint8_t server_key[CREDENTIAL_KEY_LEN]; + EXPECT_TRUE( + credentials_compute_keys(KAT_PASSWORD, v.salt, v.iters, client_key, stored_key, server_key)); + EXPECT_TRUE(memcmp(stored_key, v.stored_key, CREDENTIAL_KEY_LEN) == 0); + EXPECT_TRUE(memcmp(server_key, v.server_key, CREDENTIAL_KEY_LEN) == 0); + + credentials_free(store); + rm_temp(path); + free(path); +} + +/* The golden store line (KAT user/password/salt/count) parses back to exactly + * the KAT verifier keys, pinning the on-disk encoding independently. */ +static void test_credentials_store_line_golden() { + char* path = make_tmp_file(KAT_STORE_LINE "\n"); + EXPECT_NOT_NULL(path); + char err[512]; + CredentialStore* store = credentials_load(path, NULL, err, sizeof(err)); + EXPECT_NOT_NULL(store); + EXPECT_TRUE(credentials_store_has(store, "alice")); + + CredentialVerifier v; + const char* module_users[] = {"alice"}; + EXPECT_TRUE(credentials_get_verifier(store, "alice", module_users, 1, &v)); + EXPECT_TRUE(v.found); + EXPECT_EQ_INT((int)v.iters, (int)KAT_ITERS); + uint8_t expect[CREDENTIAL_KEY_LEN]; + unhex(KAT_STORED_KEY, expect, CREDENTIAL_KEY_LEN); + EXPECT_TRUE(memcmp(v.stored_key, expect, CREDENTIAL_KEY_LEN) == 0); + unhex(KAT_SERVER_KEY, expect, CREDENTIAL_KEY_LEN); + EXPECT_TRUE(memcmp(v.server_key, expect, CREDENTIAL_KEY_LEN) == 0); + uint8_t salt[CREDENTIAL_SALT_LEN]; + ramp(salt, sizeof(salt), 0x00); + EXPECT_TRUE(memcmp(v.salt, salt, sizeof(salt)) == 0); + + credentials_free(store); + rm_temp(path); + free(path); +} + +static void test_credentials_store_parse_valid() { + char line_alice[CREDENTIAL_MAX_LINE]; + char line_bob[CREDENTIAL_MAX_LINE]; + EXPECT_TRUE( + make_store_line("alice", KAT_PASSWORD, CREDENTIAL_MIN_ITERS, line_alice, sizeof(line_alice))); + EXPECT_TRUE( + make_store_line("bob", "bob-s3cret", CREDENTIAL_MIN_ITERS, line_bob, sizeof(line_bob))); + char contents[2 * CREDENTIAL_MAX_LINE + 64]; + snprintf(contents, sizeof(contents), "# server credential store\n; comment\n\n%s\n%s\n", + line_alice, line_bob); + char* path = make_tmp_file(contents); + EXPECT_NOT_NULL(path); + char err[512]; + CredentialStore* store = credentials_load(path, NULL, err, sizeof(err)); + EXPECT_NOT_NULL(store); + EXPECT_EQ_INT(credentials_store_size(store), 2); + EXPECT_TRUE(credentials_store_has(store, "alice")); + EXPECT_TRUE(credentials_store_has(store, "bob")); + EXPECT_FALSE(credentials_store_has(store, "mallory")); + + /* Unknown user and off-list user both yield a not-found dummy. */ + const char* module_users[] = {"alice"}; + CredentialVerifier v; + EXPECT_TRUE(credentials_get_verifier(store, "mallory", module_users, 1, &v)); + EXPECT_FALSE(v.found); + EXPECT_TRUE(credentials_get_verifier(store, "bob", module_users, 1, &v)); + EXPECT_FALSE(v.found); + EXPECT_TRUE(credentials_get_verifier(store, "alice", module_users, 1, &v)); + EXPECT_TRUE(v.found); + /* The dummy keys are fixed (all zero) so they can never authenticate. */ + const uint8_t zero[CREDENTIAL_KEY_LEN] = {0}; + EXPECT_TRUE(credentials_get_verifier(store, "mallory", module_users, 1, &v)); + EXPECT_TRUE(memcmp(v.stored_key, zero, CREDENTIAL_KEY_LEN) == 0); + EXPECT_TRUE(memcmp(v.server_key, zero, CREDENTIAL_KEY_LEN) == 0); + /* Deterministic dummy challenge: the same unknown username always yields the + * same salt and iteration count, while different usernames differ, so probing + * the store twice cannot reveal membership. */ + uint8_t salt_a[CREDENTIAL_SALT_LEN]; + uint8_t salt_b[CREDENTIAL_SALT_LEN]; + uint8_t salt_c[CREDENTIAL_SALT_LEN]; + uint32_t miss_iters_a = 0; + uint32_t miss_iters_b = 0; + EXPECT_TRUE(credentials_get_verifier(store, "mallory", module_users, 1, &v)); + memcpy(salt_a, v.salt, sizeof(salt_a)); + miss_iters_a = v.iters; + EXPECT_TRUE(credentials_get_verifier(store, "mallory", module_users, 1, &v)); + memcpy(salt_b, v.salt, sizeof(salt_b)); + miss_iters_b = v.iters; + EXPECT_TRUE(memcmp(salt_a, salt_b, sizeof(salt_a)) == 0); + EXPECT_EQ_INT((int)miss_iters_a, (int)miss_iters_b); + /* A miss is answered with the store-wide uniform iteration count. */ + EXPECT_EQ_INT((int)miss_iters_a, (int)CREDENTIAL_MIN_ITERS); + EXPECT_TRUE(credentials_get_verifier(store, "trudy", module_users, 1, &v)); + memcpy(salt_c, v.salt, sizeof(salt_c)); + EXPECT_TRUE(memcmp(salt_a, salt_c, sizeof(salt_a)) != 0); + + credentials_free(store); + rm_temp(path); + free(path); +} + +static void test_credentials_store_parse_rejects_malformed() { + /* A valid salt (16 bytes -> 24 b64 chars) / keys (32 bytes -> 44 chars). */ + uint8_t sixteen[CREDENTIAL_SALT_LEN] = {0}; + uint8_t thirtytwo[CREDENTIAL_KEY_LEN] = {0}; + char salt_b64[25]; + char key_b64[45]; + credentials_b64_encode(sixteen, sizeof(sixteen), salt_b64, sizeof(salt_b64)); + credentials_b64_encode(thirtytwo, sizeof(thirtytwo), key_b64, sizeof(key_b64)); + + char below_min[CREDENTIAL_MAX_LINE]; + char above_max[CREDENTIAL_MAX_LINE]; + char short_salt[CREDENTIAL_MAX_LINE]; + char short_key[CREDENTIAL_MAX_LINE]; + char empty_field[CREDENTIAL_MAX_LINE]; + snprintf(below_min, sizeof(below_min), "alice:$fastsync$1$pbkdf2-sha256$99$%s$%s$%s\n", salt_b64, + key_b64, key_b64); + snprintf(above_max, sizeof(above_max), "alice:$fastsync$1$pbkdf2-sha256$99999999$%s$%s$%s\n", + salt_b64, key_b64, key_b64); + snprintf(short_salt, sizeof(short_salt), "alice:$fastsync$1$pbkdf2-sha256$600000$AAAA$%s$%s\n", + key_b64, key_b64); + snprintf(short_key, sizeof(short_key), "alice:$fastsync$1$pbkdf2-sha256$600000$%s$AAAA$%s\n", + salt_b64, key_b64); + snprintf(empty_field, sizeof(empty_field), "alice:$fastsync$1$pbkdf2-sha256$600000$%s$%s$\n", + salt_b64, key_b64); + + const char* cases[] = { + "alice\n", + ":anything\n", + "alice:not-a-verifier\n", + below_min, + above_max, + short_salt, + short_key, + empty_field, + "ali " + "ce:$fastsync$1$pbkdf2-sha256$600000$AAECAwQFBgcICQoLDA0ODw==$0ZL22hxUv3N2jwp8" + "cTmVIS0wPEb4Cd4bLkB/s60cIGs=$UIrVh1dPWXAMLQu+yEF9b+lL+OOWaa26vny72HAPhs4=\n", + }; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + char* path = make_tmp_file(cases[i]); + EXPECT_NOT_NULL(path); + char err[512]; + const CredentialStore* store = credentials_load(path, NULL, err, sizeof(err)); + EXPECT_NULL(store); + EXPECT_TRUE(err[0] != '\0'); + rm_temp(path); + free(path); + } +} + +static void test_credentials_store_rejects_legacy_hex() { + const char* secret = "9b90e524e94995ee4aeae2ee3c428a53405d1e8db147f44facc46797d0caf4c3"; + char contents[CREDENTIAL_MAX_LINE]; + snprintf(contents, sizeof(contents), "alice:%s\n", secret); + char* path = make_tmp_file(contents); + EXPECT_NOT_NULL(path); + char err[512]; + const CredentialStore* store = credentials_load(path, NULL, err, sizeof(err)); + EXPECT_NULL(store); + EXPECT_TRUE(strstr(err, "legacy") != NULL); + EXPECT_TRUE(strstr(err, "alice") != NULL); + rm_temp(path); + free(path); +} + +static void test_credentials_store_duplicate_rejected() { + char line[CREDENTIAL_MAX_LINE]; + EXPECT_TRUE(make_store_line("alice", KAT_PASSWORD, CREDENTIAL_MIN_ITERS, line, sizeof(line))); + char contents[2 * CREDENTIAL_MAX_LINE + 8]; + snprintf(contents, sizeof(contents), "%s\n%s\n", line, line); + char* path = make_tmp_file(contents); + EXPECT_NOT_NULL(path); + char err[512]; + EXPECT_NULL(credentials_load(path, NULL, err, sizeof(err))); + EXPECT_TRUE(strstr(err, "duplicate") != NULL); + rm_temp(path); + free(path); +} + +/* A store must be uniform in its iteration count so a miss can be challenged + * with the store-wide count without leaking membership. */ +static void test_credentials_store_rejects_nonuniform_iters() { + char line_a[CREDENTIAL_MAX_LINE]; + char line_b[CREDENTIAL_MAX_LINE]; + EXPECT_TRUE(make_store_line("alice", KAT_PASSWORD, CREDENTIAL_MIN_ITERS, line_a, sizeof(line_a))); + EXPECT_TRUE( + make_store_line("bob", "bob-s3cret", CREDENTIAL_MIN_ITERS * 2, line_b, sizeof(line_b))); + char contents[2 * CREDENTIAL_MAX_LINE + 8]; + snprintf(contents, sizeof(contents), "%s\n%s\n", line_a, line_b); + char* path = make_tmp_file(contents); + EXPECT_NOT_NULL(path); + char err[512]; + EXPECT_NULL(credentials_load(path, NULL, err, sizeof(err))); + EXPECT_TRUE(strstr(err, "uniform") != NULL); + rm_temp(path); + free(path); +} + +static void test_credentials_store_parse_missing_file() { + char err[512]; + const CredentialStore* store = + credentials_load("/nonexistent/cred-file-xyz", NULL, err, sizeof(err)); + EXPECT_NULL(store); + EXPECT_TRUE(strstr(err, "cannot open") != NULL); +} + +static void test_credentials_store_empty_and_null() { + char err[512]; + CredentialStore* store = credentials_load(NULL, NULL, err, sizeof(err)); + EXPECT_NOT_NULL(store); + EXPECT_EQ_INT(credentials_store_size(store), 0); + /* An empty store answers a miss with the default work factor. */ + CredentialVerifier v; + EXPECT_TRUE(credentials_get_verifier(store, "nobody", NULL, 0, &v)); + EXPECT_FALSE(v.found); + EXPECT_EQ_INT((int)v.iters, (int)CREDENTIAL_DEFAULT_ITERS); + credentials_free(store); + + char* path = make_tmp_file("# nothing here\n; nor here\n"); + EXPECT_NOT_NULL(path); + store = credentials_load(path, NULL, err, sizeof(err)); + EXPECT_NOT_NULL(store); + EXPECT_EQ_INT(credentials_store_size(store), 0); + credentials_free(store); + rm_temp(path); + free(path); +} + +static void test_credentials_store_overlong_line_rejected() { + char big[CREDENTIAL_MAX_LINE + 80]; + int n = snprintf(big, sizeof(big), "alice:%s", KAT_NAME_PREFIX); + memset(big + n, 'a', sizeof(big) - (size_t)n - 1); + big[sizeof(big) - 2] = '\n'; + big[sizeof(big) - 1] = '\0'; + char* path = make_tmp_file(big); + EXPECT_NOT_NULL(path); + char err[512]; + const CredentialStore* store = credentials_load(path, NULL, err, sizeof(err)); + EXPECT_NULL(store); + rm_temp(path); + free(path); +} + +static void test_credentials_early_input_merge() { + char alice[CREDENTIAL_MAX_LINE]; + char bob[CREDENTIAL_MAX_LINE]; + EXPECT_TRUE(make_store_line("alice", KAT_PASSWORD, CREDENTIAL_MIN_ITERS, alice, sizeof(alice))); + EXPECT_TRUE(make_store_line("bob", "bob-s3cret", CREDENTIAL_MIN_ITERS, bob, sizeof(bob))); + char alice_file[CREDENTIAL_MAX_LINE + 2]; + char bob_file[CREDENTIAL_MAX_LINE + 2]; + char alice_other[CREDENTIAL_MAX_LINE]; + char bob_other_iters[CREDENTIAL_MAX_LINE]; + snprintf(alice_file, sizeof(alice_file), "%s\n", alice); + snprintf(bob_file, sizeof(bob_file), "%s\n", bob); + EXPECT_TRUE(make_store_line("alice", "different-s3cret", CREDENTIAL_MIN_ITERS, alice_other, + sizeof(alice_other))); + EXPECT_TRUE(make_store_line("carol", "carol-s3cret", CREDENTIAL_MIN_ITERS * 2, bob_other_iters, + sizeof(bob_other_iters))); + + char* pw = make_tmp_file(alice_file); + char* early = make_tmp_file(bob_file); + char* early_same = make_tmp_file(alice_file); /* byte-identical verifier dedupes */ + char* early_diff = make_tmp_file(alice_other); + char* early_iters = make_tmp_file(bob_other_iters); + EXPECT_NOT_NULL(pw); + EXPECT_NOT_NULL(early); + EXPECT_NOT_NULL(early_same); + EXPECT_NOT_NULL(early_diff); + EXPECT_NOT_NULL(early_iters); + char err[512]; + + /* A second file adds a new user. */ + CredentialStore* store = credentials_load(pw, early, err, sizeof(err)); + EXPECT_NOT_NULL(store); + EXPECT_EQ_INT(credentials_store_size(store), 2); + EXPECT_TRUE(credentials_store_has(store, "alice")); + EXPECT_TRUE(credentials_store_has(store, "bob")); + credentials_free(store); + + /* The same user with the SAME verifier dedupes. */ + store = credentials_load(pw, early_same, err, sizeof(err)); + EXPECT_NOT_NULL(store); + EXPECT_EQ_INT(credentials_store_size(store), 1); + credentials_free(store); + + /* The same user with a DIFFERENT verifier fails closed (ambiguous). */ + store = credentials_load(pw, early_diff, err, sizeof(err)); + EXPECT_NULL(store); + EXPECT_TRUE(err[0] != '\0'); + + /* A layered store must stay uniform in its iteration count. */ + store = credentials_load(pw, early_iters, err, sizeof(err)); + EXPECT_NULL(store); + EXPECT_TRUE(strstr(err, "uniform") != NULL); + + rm_temp(pw); + rm_temp(early); + rm_temp(early_same); + rm_temp(early_diff); + rm_temp(early_iters); + free(pw); + free(early); + free(early_same); + free(early_diff); + free(early_iters); +} + +static void test_credentials_read_secret_file() { + char err[512]; + char* user = NULL; + char* password = NULL; + + char* path = make_tmp_file("# password file\n" + "\n" + "alice:correct horse battery staple\n" + "ignored:second line\n"); + EXPECT_NOT_NULL(path); + EXPECT_EQ_INT(credentials_read_secret_file(path, &user, &password, err, sizeof(err)), 0); + EXPECT_EQ_STR(user, "alice"); + EXPECT_EQ_STR(password, "correct horse battery staple"); + free(user); + free(password); + user = password = NULL; + rm_temp(path); + free(path); + + path = make_tmp_file(" bob : s3cret \r\n"); + EXPECT_NOT_NULL(path); + EXPECT_EQ_INT(credentials_read_secret_file(path, &user, &password, err, sizeof(err)), 0); + EXPECT_EQ_STR(user, "bob"); + EXPECT_EQ_STR(password, " s3cret "); + free(user); + free(password); + user = password = NULL; + rm_temp(path); + free(path); + + path = make_tmp_file("carol: \n"); + EXPECT_NOT_NULL(path); + EXPECT_EQ_INT(credentials_read_secret_file(path, &user, &password, err, sizeof(err)), 0); + EXPECT_EQ_STR(user, "carol"); + EXPECT_EQ_STR(password, " "); + free(user); + free(password); + user = password = NULL; + rm_temp(path); + free(path); + + path = make_tmp_file(""); + EXPECT_NOT_NULL(path); + EXPECT_EQ_INT(credentials_read_secret_file(path, &user, &password, err, sizeof(err)), -1); + EXPECT_NULL(user); + EXPECT_NULL(password); + EXPECT_TRUE(strstr(err, "no 'user:password'") != NULL); + rm_temp(path); + free(path); +} + +static void test_credentials_read_secret_file_bad() { + char err[512]; + const char* cases[] = { + "alicepassword\n", + ":password\n", + "alice:\n", + "alice:\r\n", + }; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + char* path = make_tmp_file(cases[i]); + EXPECT_NOT_NULL(path); + char* user = (char*)1; + char* password = (char*)1; + EXPECT_EQ_INT(credentials_read_secret_file(path, &user, &password, err, sizeof(err)), -1); + EXPECT_NULL(user); + EXPECT_NULL(password); + EXPECT_TRUE(err[0] != '\0'); + rm_temp(path); + free(path); + } + + char* missing = "/nonexistent/password-file-xyz"; + EXPECT_EQ_INT(credentials_read_secret_file(missing, NULL, NULL, err, sizeof(err)), -1); +} + +static void test_credentials_hash_file() { + char* plaintext = make_tmp_file("# comment\n\n alice :" KAT_PASSWORD "\nbob:bob-s3cret\n"); + EXPECT_NOT_NULL(plaintext); + FILE* out = tmpfile(); + EXPECT_NOT_NULL(out); + char err[512]; + EXPECT_EQ_INT(credentials_hash_file(plaintext, CREDENTIAL_MIN_ITERS, out, err, sizeof(err)), 0); + rewind(out); + + char line1[CREDENTIAL_MAX_LINE]; + char line2[CREDENTIAL_MAX_LINE]; + EXPECT_NOT_NULL(fgets(line1, sizeof(line1), out)); + EXPECT_NOT_NULL(fgets(line2, sizeof(line2), out)); + EXPECT_NULL(fgets(err, sizeof(err), out)); /* exactly two entries */ + size_t n1 = strlen(line1); + if (n1 > 0 && line1[n1 - 1] == '\n') + line1[--n1] = '\0'; + size_t n2 = strlen(line2); + if (n2 > 0 && line2[n2 - 1] == '\n') + line2[--n2] = '\0'; + EXPECT_TRUE(strncmp(line1, "alice:", 6) == 0); + EXPECT_TRUE(strncmp(line2, "bob:", 4) == 0); + EXPECT_TRUE(strstr(line1, KAT_NAME_PREFIX) != NULL); + fclose(out); + + /* The generated lines load as a valid store. */ + char contents[2 * CREDENTIAL_MAX_LINE + 8]; + snprintf(contents, sizeof(contents), "%s\n%s\n", line1, line2); + char* store_path = make_tmp_file(contents); + EXPECT_NOT_NULL(store_path); + CredentialStore* store = credentials_load(store_path, NULL, err, sizeof(err)); + EXPECT_NOT_NULL(store); + EXPECT_EQ_INT(credentials_store_size(store), 2); + credentials_free(store); + rm_temp(store_path); + free(store_path); + rm_temp(plaintext); + free(plaintext); + + /* An invalid iteration count is refused up front. */ + char* p2 = make_tmp_file("alice:pw\n"); + EXPECT_NOT_NULL(p2); + FILE* out2 = tmpfile(); + EXPECT_NOT_NULL(out2); + EXPECT_EQ_INT(credentials_hash_file(p2, 10, out2, err, sizeof(err)), -1); + EXPECT_TRUE(err[0] != '\0'); + fclose(out2); + rm_temp(p2); + free(p2); +} + +static void test_credentials_rejects_group_or_other_accessible() { + char err[512]; + char line[CREDENTIAL_MAX_LINE]; + EXPECT_TRUE(make_store_line("alice", KAT_PASSWORD, CREDENTIAL_MIN_ITERS, line, sizeof(line))); + char contents[CREDENTIAL_MAX_LINE + 2]; + snprintf(contents, sizeof(contents), "%s\n", line); + char* path = make_tmp_file(contents); + EXPECT_NOT_NULL(path); + + EXPECT_EQ_INT(chmod(path, 0600), 0); + CredentialStore* store = credentials_load(path, NULL, err, sizeof(err)); + EXPECT_NOT_NULL(store); + credentials_free(store); + + EXPECT_EQ_INT(chmod(path, 0640), 0); + EXPECT_NULL(credentials_load(path, NULL, err, sizeof(err))); + EXPECT_TRUE(strstr(err, "owner-only") != NULL); + EXPECT_EQ_INT(chmod(path, 0604), 0); + EXPECT_NULL(credentials_load(path, NULL, err, sizeof(err))); + + EXPECT_EQ_INT(chmod(path, 0644), 0); + char* user = NULL; + char* password = NULL; + EXPECT_EQ_INT(credentials_read_secret_file(path, &user, &password, err, sizeof(err)), -1); + EXPECT_NULL(user); + EXPECT_NULL(password); + EXPECT_TRUE(strstr(err, "owner-only") != NULL); + + char* pw = make_tmp_file("bob:bob-s3cret\n"); + EXPECT_NOT_NULL(pw); + EXPECT_EQ_INT(chmod(path, 0644), 0); + EXPECT_NULL(credentials_load(pw, path, err, sizeof(err))); + + rm_temp(pw); + rm_temp(path); + free(pw); + free(path); +} + +/* Loading a store auto-creates an owner-only `.dummykey` sidecar whose + * key is stable across reloads, so an unknown-user dummy salt is identical + * across two loads (the anti-restart enumeration property). */ +static void test_credentials_dummy_key_persisted() { + char line[CREDENTIAL_MAX_LINE]; + EXPECT_TRUE(make_store_line("alice", KAT_PASSWORD, CREDENTIAL_MIN_ITERS, line, sizeof(line))); + char contents[CREDENTIAL_MAX_LINE + 2]; + snprintf(contents, sizeof(contents), "%s\n", line); + char* path = make_tmp_file(contents); + EXPECT_NOT_NULL(path); + char* sidecar = dummy_sidecar_path(path); + EXPECT_NOT_NULL(sidecar); + + char err[512]; + CredentialStore* store = credentials_load(path, NULL, err, sizeof(err)); + EXPECT_NOT_NULL(store); + + struct stat st; + EXPECT_EQ_INT(stat(sidecar, &st), 0); + EXPECT_TRUE(S_ISREG(st.st_mode)); + EXPECT_TRUE((st.st_mode & (S_IRWXG | S_IRWXO)) == 0); + EXPECT_EQ_INT((int)(st.st_mode & 07777), 0600); + EXPECT_EQ_INT((int)st.st_size, CREDENTIAL_KEY_LEN); + + CredentialVerifier v1; + EXPECT_TRUE(credentials_get_verifier(store, "unknown-user", NULL, 0, &v1)); + EXPECT_FALSE(v1.found); + credentials_free(store); + + store = credentials_load(path, NULL, err, sizeof(err)); + EXPECT_NOT_NULL(store); + CredentialVerifier v2; + EXPECT_TRUE(credentials_get_verifier(store, "unknown-user", NULL, 0, &v2)); + EXPECT_FALSE(v2.found); + EXPECT_TRUE(memcmp(v1.salt, v2.salt, sizeof(v1.salt)) == 0); + credentials_free(store); + + rm_temp(path); + free(path); + free(sidecar); +} + +/* A restrictive umask must not leave the freshly published sidecar with owner + * bits cleared: creation forces exact 0600 with fchmod (the reader requires an + * exact 0600), so the daemon cannot lock itself out on the next restart. */ +static void test_credentials_dummy_key_exact_mode_under_umask() { + char line[CREDENTIAL_MAX_LINE]; + EXPECT_TRUE(make_store_line("alice", KAT_PASSWORD, CREDENTIAL_MIN_ITERS, line, sizeof(line))); + char contents[CREDENTIAL_MAX_LINE + 2]; + snprintf(contents, sizeof(contents), "%s\n", line); + char* path = make_tmp_file(contents); + EXPECT_NOT_NULL(path); + char* sidecar = dummy_sidecar_path(path); + EXPECT_NOT_NULL(sidecar); + + /* Clear every permission bit the O_CREAT mode would otherwise provide; only + * the explicit fchmod can restore the exact 0600 the reader demands. */ + mode_t old_umask = umask(0777); + char err[512]; + CredentialStore* store = credentials_load(path, NULL, err, sizeof(err)); + umask(old_umask); + EXPECT_NOT_NULL(store); + + struct stat st; + EXPECT_EQ_INT(stat(sidecar, &st), 0); + EXPECT_EQ_INT((int)(st.st_mode & 07777), 0600); + EXPECT_EQ_INT((int)st.st_size, CREDENTIAL_KEY_LEN); + credentials_free(store); + + rm_temp(path); + free(path); + free(sidecar); +} + +/* A sidecar that is group/other accessible, the wrong size, or not a regular + * file must fail the load closed. */ +static void test_credentials_dummy_key_rejects_bad_sidecar() { + char err[512]; + char line[CREDENTIAL_MAX_LINE]; + EXPECT_TRUE(make_store_line("alice", KAT_PASSWORD, CREDENTIAL_MIN_ITERS, line, sizeof(line))); + char contents[CREDENTIAL_MAX_LINE + 2]; + snprintf(contents, sizeof(contents), "%s\n", line); + + /* Group/other permission bits on the sidecar. */ + char* path = make_tmp_file(contents); + EXPECT_NOT_NULL(path); + char* sidecar = dummy_sidecar_path(path); + EXPECT_NOT_NULL(sidecar); + uint8_t key[CREDENTIAL_KEY_LEN]; + memset(key, 0x5a, sizeof(key)); + FILE* fp = fopen(sidecar, "wb"); + EXPECT_NOT_NULL(fp); + EXPECT_TRUE(fwrite(key, 1, sizeof(key), fp) == sizeof(key)); + fclose(fp); + EXPECT_EQ_INT(chmod(sidecar, 0640), 0); + EXPECT_NULL(credentials_load(path, NULL, err, sizeof(err))); + EXPECT_TRUE(err[0] != '\0'); + rm_temp(path); + free(path); + free(sidecar); + + /* Wrong size (not exactly 32 bytes). */ + path = make_tmp_file(contents); + EXPECT_NOT_NULL(path); + sidecar = dummy_sidecar_path(path); + EXPECT_NOT_NULL(sidecar); + fp = fopen(sidecar, "wb"); + EXPECT_NOT_NULL(fp); + EXPECT_TRUE(fwrite(key, 1, CREDENTIAL_SALT_LEN, fp) == CREDENTIAL_SALT_LEN); + fclose(fp); + EXPECT_EQ_INT(chmod(sidecar, 0600), 0); + EXPECT_NULL(credentials_load(path, NULL, err, sizeof(err))); + EXPECT_TRUE(err[0] != '\0'); + rm_temp(path); + free(path); + free(sidecar); + + /* Non-regular file (a directory at the sidecar path). */ + path = make_tmp_file(contents); + EXPECT_NOT_NULL(path); + sidecar = dummy_sidecar_path(path); + EXPECT_NOT_NULL(sidecar); + EXPECT_EQ_INT(mkdir(sidecar, 0700), 0); + EXPECT_NULL(credentials_load(path, NULL, err, sizeof(err))); + EXPECT_TRUE(err[0] != '\0'); + rmdir(sidecar); + rm_temp(path); + free(path); + free(sidecar); + + /* Exact-mode rule: 0400 has no group/other bits but is not 0600, so it is + * rejected now (the mode must be exactly owner read+write). */ + path = make_tmp_file(contents); + EXPECT_NOT_NULL(path); + sidecar = dummy_sidecar_path(path); + EXPECT_NOT_NULL(sidecar); + fp = fopen(sidecar, "wb"); + EXPECT_NOT_NULL(fp); + EXPECT_TRUE(fwrite(key, 1, sizeof(key), fp) == sizeof(key)); + fclose(fp); + EXPECT_EQ_INT(chmod(sidecar, 0400), 0); + EXPECT_NULL(credentials_load(path, NULL, err, sizeof(err))); + EXPECT_TRUE(err[0] != '\0'); + rm_temp(path); + free(path); + free(sidecar); +} + +/* A pre-existing valid sidecar is adopted verbatim (no regeneration): the + * unknown-user dummy salt must equal HMAC-SHA256(known key, username), and a + * reload must yield the same salt. */ +static void test_credentials_dummy_key_existing_sidecar_adopted() { + char line[CREDENTIAL_MAX_LINE]; + EXPECT_TRUE(make_store_line("alice", KAT_PASSWORD, CREDENTIAL_MIN_ITERS, line, sizeof(line))); + char contents[CREDENTIAL_MAX_LINE + 2]; + snprintf(contents, sizeof(contents), "%s\n", line); + char* path = make_tmp_file(contents); + EXPECT_NOT_NULL(path); + char* sidecar = dummy_sidecar_path(path); + EXPECT_NOT_NULL(sidecar); + + /* Pre-create a valid owner-only sidecar with a known key. */ + uint8_t key[CREDENTIAL_KEY_LEN]; + memset(key, 0x5a, sizeof(key)); + FILE* fp = fopen(sidecar, "wb"); + EXPECT_NOT_NULL(fp); + EXPECT_TRUE(fwrite(key, 1, sizeof(key), fp) == sizeof(key)); + fclose(fp); + EXPECT_EQ_INT(chmod(sidecar, 0600), 0); + + char err[512]; + CredentialStore* store = credentials_load(path, NULL, err, sizeof(err)); + EXPECT_NOT_NULL(store); + CredentialVerifier v1; + EXPECT_TRUE(credentials_get_verifier(store, "unknown-user", NULL, 0, &v1)); + EXPECT_FALSE(v1.found); + /* HMAC-SHA256(0x5a * 32, "unknown-user")[:16], computed independently. */ + uint8_t expect[CREDENTIAL_SALT_LEN]; + unhex("4b0d2e6b73025cc2fcb41d0a710ff469", expect, sizeof(expect)); + EXPECT_TRUE(memcmp(v1.salt, expect, sizeof(expect)) == 0); + credentials_free(store); + + /* The adopted sidecar still holds exactly the pre-created key (not a fresh + * random one). */ + uint8_t readback[CREDENTIAL_KEY_LEN]; + fp = fopen(sidecar, "rb"); + EXPECT_NOT_NULL(fp); + EXPECT_TRUE(fread(readback, 1, sizeof(readback), fp) == sizeof(readback)); + fclose(fp); + EXPECT_TRUE(memcmp(readback, key, sizeof(key)) == 0); + + /* Persisted across a reload. */ + CredentialVerifier v2; + store = credentials_load(path, NULL, err, sizeof(err)); + EXPECT_NOT_NULL(store); + EXPECT_TRUE(credentials_get_verifier(store, "unknown-user", NULL, 0, &v2)); + EXPECT_TRUE(memcmp(v1.salt, v2.salt, sizeof(v1.salt)) == 0); + credentials_free(store); + + rm_temp(path); + free(path); + free(sidecar); +} + +/* A symlink planted at the sidecar path must fail the load closed (O_NOFOLLOW), + * even when it resolves to a valid owner-only key file. */ +static void test_credentials_dummy_key_symlink_rejected() { + char line[CREDENTIAL_MAX_LINE]; + EXPECT_TRUE(make_store_line("alice", KAT_PASSWORD, CREDENTIAL_MIN_ITERS, line, sizeof(line))); + char contents[CREDENTIAL_MAX_LINE + 2]; + snprintf(contents, sizeof(contents), "%s\n", line); + char* path = make_tmp_file(contents); + EXPECT_NOT_NULL(path); + char* sidecar = dummy_sidecar_path(path); + EXPECT_NOT_NULL(sidecar); + + char target[256]; + snprintf(target, sizeof(target), "/tmp/fs_cred_key_%d_%d", (int)getpid(), g_file_counter++); + uint8_t key[CREDENTIAL_KEY_LEN]; + memset(key, 0x5a, sizeof(key)); + FILE* fp = fopen(target, "wb"); + EXPECT_NOT_NULL(fp); + EXPECT_TRUE(fwrite(key, 1, sizeof(key), fp) == sizeof(key)); + fclose(fp); + EXPECT_EQ_INT(chmod(target, 0600), 0); + EXPECT_EQ_INT(symlink(target, sidecar), 0); + + char err[512]; + EXPECT_NULL(credentials_load(path, NULL, err, sizeof(err))); + EXPECT_TRUE(err[0] != '\0'); + + unlink(sidecar); /* remove the symlink itself, not its target */ + unlink(target); + rm_temp(path); + free(path); + free(sidecar); +} + +/* A NULL store path has nowhere to persist a key, so each load gets a fresh + * ephemeral key (and creates no sidecar). */ +static void test_credentials_dummy_key_null_store_ephemeral() { + char err[512]; + CredentialStore* store = credentials_load(NULL, NULL, err, sizeof(err)); + EXPECT_NOT_NULL(store); + CredentialVerifier v1; + CredentialVerifier v1b; + EXPECT_TRUE(credentials_get_verifier(store, "nobody", NULL, 0, &v1)); + EXPECT_FALSE(v1.found); + /* Within one store the dummy challenge is still deterministic. */ + EXPECT_TRUE(credentials_get_verifier(store, "nobody", NULL, 0, &v1b)); + EXPECT_TRUE(memcmp(v1.salt, v1b.salt, sizeof(v1.salt)) == 0); + credentials_free(store); + + store = credentials_load(NULL, NULL, err, sizeof(err)); + EXPECT_NOT_NULL(store); + CredentialVerifier v2; + EXPECT_TRUE(credentials_get_verifier(store, "nobody", NULL, 0, &v2)); + /* No persistence path, so the second load's random key differs (and with it + * the dummy salt). */ + EXPECT_TRUE(memcmp(v1.salt, v2.salt, sizeof(v1.salt)) != 0); + credentials_free(store); +} + +static void test_credentials_burn() { + char secret[32]; + memcpy(secret, "supersecretvalue", 17); + credentials_burn(secret, 16); + for (int i = 0; i < 16; i++) + EXPECT_EQ_INT(secret[i], 0); + credentials_burn(NULL, 0); +} + +void test_credentials(void) { + test_credentials_secure_equal(); + test_credentials_b64(); + test_credentials_random_bytes(); + test_credentials_compute_keys_kat(); + test_credentials_auth_message_and_proof_kat(); + test_credentials_verify_response_kat(); + test_credentials_username_valid(); + test_credentials_hash_store_line_roundtrip(); + test_credentials_store_line_golden(); + test_credentials_store_parse_valid(); + test_credentials_store_parse_rejects_malformed(); + test_credentials_store_rejects_legacy_hex(); + test_credentials_store_duplicate_rejected(); + test_credentials_store_rejects_nonuniform_iters(); + test_credentials_store_parse_missing_file(); + test_credentials_store_empty_and_null(); + test_credentials_store_overlong_line_rejected(); + test_credentials_early_input_merge(); + test_credentials_read_secret_file(); + test_credentials_read_secret_file_bad(); + test_credentials_hash_file(); + test_credentials_rejects_group_or_other_accessible(); + test_credentials_dummy_key_persisted(); + test_credentials_dummy_key_exact_mode_under_umask(); + test_credentials_dummy_key_rejects_bad_sidecar(); + test_credentials_dummy_key_existing_sidecar_adopted(); + test_credentials_dummy_key_symlink_rejected(); + test_credentials_dummy_key_null_store_ephemeral(); + test_credentials_burn(); +} diff --git a/tests/test_credentials.h b/tests/test_credentials.h new file mode 100644 index 0000000..cc9ed6a --- /dev/null +++ b/tests/test_credentials.h @@ -0,0 +1,6 @@ +#ifndef TEST_CREDENTIALS_H +#define TEST_CREDENTIALS_H + +void test_credentials(); + +#endif diff --git a/tests/test_daemon_conf.c b/tests/test_daemon_conf.c new file mode 100644 index 0000000..9580f8b --- /dev/null +++ b/tests/test_daemon_conf.c @@ -0,0 +1,355 @@ +#include "test_daemon_conf.h" +#include "daemon_conf.h" +#include "test_utils.h" +#include +#include +#include +#include + +/* Write a config body into a fresh temp file and return its path in out_path + * (heap-allocated; caller frees). Returns 0 on success. */ +static int write_conf(const char* body, char** out_path) { + char tmpl[] = "/tmp/fastsync_daemon_conf_XXXXXX"; + int fd = mkstemp(tmpl); + if (fd < 0) + return -1; + size_t len = strlen(body); + if (write(fd, body, len) != (ssize_t)len) { + close(fd); + unlink(tmpl); + return -1; + } + close(fd); + *out_path = strdup(tmpl); + return *out_path ? 0 : -1; +} + +static void test_daemon_conf_create_defaults() { + DaemonConf* conf = daemon_conf_create(); + EXPECT_NOT_NULL(conf); + EXPECT_EQ_INT(conf->global.port, DAEMON_CONF_DEFAULT_PORT); + EXPECT_NULL(conf->global.motd_file); + EXPECT_NULL(conf->global.address); + EXPECT_EQ_INT(conf->module_count, 0); + daemon_conf_free(conf); +} + +static void test_daemon_conf_full_parse() { + char* path; + EXPECT_EQ_INT(write_conf("port = 8734\n" + "motd file = /etc/fastsync/motd\n" + "address = 127.0.0.1\n" + "\n" + "[backup]\n" + "path = /srv/backup\n" + "read only = yes\n" + "client owner = yes\n" + "auth users = alice, bob\n", + &path), + 0); + char err[256]; + DaemonConf* conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_EQ_INT(conf->global.port, 8734); + EXPECT_EQ_STR(conf->global.motd_file, "/etc/fastsync/motd"); + EXPECT_EQ_STR(conf->global.address, "127.0.0.1"); + EXPECT_EQ_INT(conf->module_count, 1); + EXPECT_EQ_STR(conf->modules[0].name, "backup"); + EXPECT_EQ_STR(conf->modules[0].path, "/srv/backup"); + EXPECT_TRUE(conf->modules[0].read_only); + EXPECT_TRUE(conf->modules[0].client_owner); + EXPECT_EQ_INT(conf->modules[0].auth_user_count, 2); + EXPECT_EQ_STR(conf->modules[0].auth_users[0], "alice"); + EXPECT_EQ_STR(conf->modules[0].auth_users[1], "bob"); + daemon_conf_free(conf); +} + +static void test_daemon_conf_comments_and_blank_lines() { + char* path; + EXPECT_EQ_INT(write_conf("# a full-line comment\n" + "; a semicolon comment\n" + " # indented comment\n" + " \n" + "\t; another\n" + "[alpha]\n" + "path = /a\n" + "\n" + "[beta]\n" + "path = /b\n", + &path), + 0); + char err[256]; + DaemonConf* conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_EQ_INT(conf->module_count, 2); + EXPECT_EQ_STR(conf->modules[0].name, "alpha"); + EXPECT_EQ_STR(conf->modules[1].name, "beta"); + /* `client owner` defaults to off: a module must opt in to client-chosen + ownership. */ + EXPECT_FALSE(conf->modules[0].client_owner); + EXPECT_FALSE(conf->modules[1].client_owner); + daemon_conf_free(conf); +} + +static void test_daemon_conf_case_insensitive_and_bool_variants() { + char* path; + EXPECT_EQ_INT(write_conf("PORT = 9001\n" + "[CaseMod]\n" + "PATH = /cm\n" + "READ ONLY = True\n", + &path), + 0); + char err[256]; + DaemonConf* conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_EQ_INT(conf->global.port, 9001); + EXPECT_EQ_INT(conf->module_count, 1); + EXPECT_EQ_STR(conf->modules[0].name, "CaseMod"); + EXPECT_TRUE(conf->modules[0].read_only); + daemon_conf_free(conf); + + EXPECT_EQ_INT(write_conf("[m]\npath = /x\nread only = 0\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_FALSE(conf->modules[0].read_only); + daemon_conf_free(conf); + + EXPECT_EQ_INT(write_conf("[m]\npath = /x\nread only = false\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_FALSE(conf->modules[0].read_only); + daemon_conf_free(conf); +} + +static void test_daemon_conf_quoted_value() { + char* path; + EXPECT_EQ_INT(write_conf("[m]\npath = \"/srv/my dir/mod\"\n", &path), 0); + char err[256]; + DaemonConf* conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_EQ_STR(conf->modules[0].path, "/srv/my dir/mod"); + daemon_conf_free(conf); +} + +static void test_daemon_conf_global_defaults_when_absent() { + char* path; + EXPECT_EQ_INT(write_conf("[m]\npath = /x\n", &path), 0); + char err[256]; + DaemonConf* conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_EQ_INT(conf->global.port, DAEMON_CONF_DEFAULT_PORT); /* 873 */ + EXPECT_NULL(conf->global.motd_file); + EXPECT_NULL(conf->global.address); + daemon_conf_free(conf); +} + +static void test_daemon_conf_unknown_key_rejected() { + char* path; + char err[256]; + EXPECT_EQ_INT(write_conf("bogus_key = 1\n", &path), 0); + const DaemonConf* conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "unknown global key") != NULL); + + EXPECT_EQ_INT(write_conf("[m]\npath = /x\nflavor = van\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "unknown key 'flavor'") != NULL); +} + +static void test_daemon_conf_malformed_rejected() { + char* path; + char err[256]; + const DaemonConf* conf; + + EXPECT_EQ_INT(write_conf("port 8734\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "expected 'key = value'") != NULL); + + EXPECT_EQ_INT(write_conf("[m]\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "no 'path'") != NULL); + + EXPECT_EQ_INT(write_conf("[m\npath = /x\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "unterminated module header") != NULL); + + EXPECT_EQ_INT(write_conf("[m] trailing\npath = /x\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "after module header") != NULL); + + EXPECT_EQ_INT(write_conf("port = notanumber\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "invalid port") != NULL); + + EXPECT_EQ_INT(write_conf("port = 70000\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "invalid port") != NULL); + + EXPECT_EQ_INT(write_conf("[m]\npath = /x\nread only = maybe\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "read only") != NULL); + + EXPECT_EQ_INT(write_conf("[m]\npath = /x\nclient owner = maybe\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "client owner") != NULL); + + EXPECT_EQ_INT(write_conf("= value\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "empty key") != NULL); + + EXPECT_EQ_INT(write_conf("[]\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "invalid module name") != NULL); + + EXPECT_EQ_INT(write_conf("[bad/name]\npath = /x\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + + EXPECT_EQ_INT(write_conf("[m]\npath = \"/unterminated\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "unterminated quoted value") != NULL); +} + +static void test_daemon_conf_duplicate_module_rejected() { + char* path; + char err[256]; + EXPECT_EQ_INT(write_conf("[m]\npath = /a\n[m]\npath = /b\n", &path), 0); + const DaemonConf* conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "duplicate module") != NULL); +} + +static void test_daemon_conf_long_line_rejected() { + char* path; + char err[256]; + char body[4600]; + memset(body, 'a', sizeof(body) - 1); + memcpy(body, "[m]\npath = /x\nport = ", 21); + body[sizeof(body) - 1] = '\0'; + EXPECT_EQ_INT(write_conf(body, &path), 0); + const DaemonConf* conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "exceeds the") != NULL); +} + +static void test_daemon_conf_missing_file_rejected() { + char err[256]; + const DaemonConf* conf = + daemon_conf_load("/nonexistent/fastsync_daemon_conf_zzz", err, sizeof(err)); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "cannot open") != NULL); +} + +static void test_daemon_conf_find_module() { + char* path; + EXPECT_EQ_INT(write_conf("[known]\npath = /rooted\n[m]\npath = /other\n", &path), 0); + char err[256]; + DaemonConf* conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + const DaemonModule* found = daemon_conf_find_module(conf, "known"); + EXPECT_NOT_NULL(found); + EXPECT_EQ_STR(found->path, "/rooted"); + EXPECT_NULL(daemon_conf_find_module(conf, "nope")); + /* Case-sensitive like rsync module names. */ + EXPECT_NULL(daemon_conf_find_module(conf, "Known")); + daemon_conf_free(conf); +} + +static void test_daemon_conf_dparam_override() { + DaemonConf* conf = daemon_conf_create(); + EXPECT_NOT_NULL(conf); + char err[256]; + + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "port=8734", err, sizeof(err)), 0); + EXPECT_EQ_INT(conf->global.port, 8734); + + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "motd file=/tmp/motd", err, sizeof(err)), 0); + EXPECT_EQ_STR(conf->global.motd_file, "/tmp/motd"); + + /* Keys are case-insensitive. */ + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "ADDRESS=127.0.0.1", err, sizeof(err)), 0); + EXPECT_EQ_STR(conf->global.address, "127.0.0.1"); + + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "port = 9000", err, sizeof(err)), 0); + EXPECT_EQ_INT(conf->global.port, 9000); + + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "port=notaport", err, sizeof(err)), -1); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "bogus=1", err, sizeof(err)), -1); + EXPECT_TRUE(strstr(err, "unknown global key") != NULL); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "port", err, sizeof(err)), -1); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "=1", err, sizeof(err)), -1); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "port=", err, sizeof(err)), -1); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "", err, sizeof(err)), -1); + + daemon_conf_free(conf); +} + +static void test_daemon_module_name_valid() { + EXPECT_TRUE(daemon_module_name_valid("backup")); + EXPECT_TRUE(daemon_module_name_valid("Backup_2")); + EXPECT_TRUE(daemon_module_name_valid("a.b-c")); + EXPECT_FALSE(daemon_module_name_valid("")); + EXPECT_FALSE(daemon_module_name_valid("with space")); + EXPECT_FALSE(daemon_module_name_valid("with/slash")); + EXPECT_FALSE(daemon_module_name_valid("with\t\ttab")); + EXPECT_FALSE(daemon_module_name_valid("bracket]")); + { + char long_name[DAEMON_MAX_MODULE_NAME + 2]; + memset(long_name, 'a', sizeof(long_name) - 1); + long_name[sizeof(long_name) - 1] = '\0'; + EXPECT_FALSE(daemon_module_name_valid(long_name)); + } +} + +void test_daemon_conf() { + test_daemon_conf_create_defaults(); + test_daemon_conf_full_parse(); + test_daemon_conf_comments_and_blank_lines(); + test_daemon_conf_case_insensitive_and_bool_variants(); + test_daemon_conf_quoted_value(); + test_daemon_conf_global_defaults_when_absent(); + test_daemon_conf_unknown_key_rejected(); + test_daemon_conf_malformed_rejected(); + test_daemon_conf_duplicate_module_rejected(); + test_daemon_conf_long_line_rejected(); + test_daemon_conf_missing_file_rejected(); + test_daemon_conf_find_module(); + test_daemon_conf_dparam_override(); + test_daemon_module_name_valid(); +} \ No newline at end of file diff --git a/tests/test_daemon_conf.h b/tests/test_daemon_conf.h new file mode 100644 index 0000000..d2dab46 --- /dev/null +++ b/tests/test_daemon_conf.h @@ -0,0 +1,6 @@ +#ifndef TEST_DAEMON_CONF_H +#define TEST_DAEMON_CONF_H + +void test_daemon_conf(); + +#endif \ No newline at end of file diff --git a/tests/test_delay_updates.c b/tests/test_delay_updates.c new file mode 100644 index 0000000..4ff9d81 --- /dev/null +++ b/tests/test_delay_updates.c @@ -0,0 +1,296 @@ +#include "test_delay_updates.h" +#include "config.h" +#include "delay_updates.h" +#include "file.h" +#include "file_receive.h" +#include "test_utils.h" +#include "utils.h" +#include +#include +#include +#include +#include +#include + +/* Recursively remove a test tree (never follows symlinks). */ +static void remove_tree(const char* path) { + struct stat st; + if (lstat(path, &st) != 0) + return; + if (S_ISDIR(st.st_mode)) { + DIR* dir = opendir(path); + if (!dir) + return; + const struct dirent* entry; + while ((entry = readdir(dir)) != NULL) { + if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) + continue; + char* child = path_cat(path, entry->d_name); + if (child) { + remove_tree(child); + free(child); + } + } + closedir(dir); + rmdir(path); + } else { + unlink(path); + } +} + +/* Build a File that carries `content`. */ +static File* make_file(const char* path, const char* content) { + File* f = file_create(path); + if (!f) + return NULL; + f->data->data = malloc(strlen(content)); + if (!f->data->data) { + file_destroy(f); + return NULL; + } + memcpy(f->data->data, content, strlen(content)); + f->data->size = strlen(content); + return f; +} + +static char* read_all(const char* path) { + FILE* fp = fopen(path, "rb"); + if (!fp) + return NULL; + char buf[256] = {0}; + size_t n = fread(buf, 1, sizeof(buf) - 1, fp); + fclose(fp); + char* out = malloc(n + 1); + if (!out) + return NULL; + memcpy(out, buf, n); + out[n] = '\0'; + return out; +} + +static void test_delay_updates_no_final_before_publish() { + const char* root = "test_delay_tmp"; + remove_tree(root); + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + cfg->delay_updates = true; + + File* f = make_file("sub/file.txt", "staged payload"); + EXPECT_NOT_NULL(f); + // cppcheck-suppress knownConditionTrueFalse + if (!cfg || !f) + goto out; + + EXPECT_EQ_INT(file_save_to_disk_full(root, f, cfg), FILE_SAVE_WRITTEN); + EXPECT_NOT_NULL(cfg->delay_context); + + const char* final_path = "test_delay_tmp/sub/file.txt"; + /* Before publication the final destination must not contain the file. */ + EXPECT_FALSE(file_path_exists_secure(final_path)); + /* The complete staged copy must live inside the staging tree. */ + char* staged = path_cat("test_delay_tmp/.fastsync-stage", "/sub/file.txt"); + EXPECT_NOT_NULL(staged); + // cppcheck-suppress knownConditionTrueFalse + if (staged) { + char* content = read_all(staged); + EXPECT_NOT_NULL(content); + // cppcheck-suppress knownConditionTrueFalse + if (content) { + EXPECT_EQ_STR(content, "staged payload"); + free(content); + } + free(staged); + } + +out: + file_destroy(f); + config_delete(cfg); + remove_tree(root); +} + +static void test_delay_updates_publish_installs_files() { + const char* root = "test_delay_pub_tmp"; + remove_tree(root); + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + cfg->delay_updates = true; + + File* f = make_file("sub/file.txt", "published payload"); + EXPECT_NOT_NULL(f); + // cppcheck-suppress knownConditionTrueFalse + if (!cfg || !f) + goto out; + + EXPECT_EQ_INT(file_save_to_disk_full(root, f, cfg), FILE_SAVE_WRITTEN); + const char* final_path = "test_delay_pub_tmp/sub/file.txt"; + EXPECT_FALSE(file_path_exists_secure(final_path)); + + EXPECT_TRUE(delay_updates_publish(cfg->delay_context, cfg)); + /* After a successful publish the file is installed and staging is gone. */ + char* content = read_all(final_path); + EXPECT_NOT_NULL(content); + // cppcheck-suppress knownConditionTrueFalse + if (content) { + EXPECT_EQ_STR(content, "published payload"); + free(content); + } + EXPECT_FALSE(file_path_exists_secure("test_delay_pub_tmp/.fastsync-stage")); + +out: + file_destroy(f); + config_delete(cfg); + remove_tree(root); +} + +/* The staged tree is cleaned on the error/abort path and final files that were + never published do not appear at the destination. */ +static void test_delay_updates_cleanup_removes_staged() { + const char* root = "test_delay_clean_tmp"; + remove_tree(root); + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + cfg->delay_updates = true; + + File* f = make_file("sub/file.txt", "never installed"); + EXPECT_NOT_NULL(f); + // cppcheck-suppress knownConditionTrueFalse + if (!cfg || !f) + goto out; + + EXPECT_EQ_INT(file_save_to_disk_full(root, f, cfg), FILE_SAVE_WRITTEN); + EXPECT_TRUE(file_path_exists_secure("test_delay_clean_tmp/.fastsync-stage/sub/file.txt")); + + delay_updates_cleanup(cfg->delay_context); + EXPECT_FALSE(file_path_exists_secure("test_delay_clean_tmp/.fastsync-stage")); + EXPECT_FALSE(file_path_exists_secure("test_delay_clean_tmp/sub/file.txt")); + +out: + file_destroy(f); + config_delete(cfg); + remove_tree(root); +} + +/* With --backup the previous version is only moved aside at publication. */ +static void test_delay_updates_backup_deferred_to_publish() { + const char* root = "test_delay_bak_tmp"; + remove_tree(root); + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + cfg->delay_updates = true; + cfg->backup = true; + + EXPECT_TRUE(file_write_to_disk("test_delay_bak_tmp/file.txt", "AAAA", 4, false, false)); + + File* f = make_file("file.txt", "BBBB"); + EXPECT_NOT_NULL(f); + // cppcheck-suppress knownConditionTrueFalse + if (!cfg || !f) + goto out; + + EXPECT_EQ_INT(file_save_to_disk_full(root, f, cfg), FILE_SAVE_WRITTEN); + /* Stage time must not touch the final file or create the backup yet. */ + char* before = read_all("test_delay_bak_tmp/file.txt"); + EXPECT_NOT_NULL(before); + // cppcheck-suppress knownConditionTrueFalse + if (before) { + EXPECT_EQ_STR(before, "AAAA"); + free(before); + } + EXPECT_FALSE(file_path_exists_secure("test_delay_bak_tmp/file.txt~")); + + EXPECT_TRUE(delay_updates_publish(cfg->delay_context, cfg)); + char* after = read_all("test_delay_bak_tmp/file.txt"); + char* backup = read_all("test_delay_bak_tmp/file.txt~"); + EXPECT_NOT_NULL(after); + EXPECT_NOT_NULL(backup); + // cppcheck-suppress knownConditionTrueFalse + if (after) { + EXPECT_EQ_STR(after, "BBBB"); + free(after); + } + // cppcheck-suppress knownConditionTrueFalse + if (backup) { + EXPECT_EQ_STR(backup, "AAAA"); + free(backup); + } + +out: + file_destroy(f); + config_delete(cfg); + remove_tree(root); +} + +/* Skip/update policy checks run against the final path at stage time, matching + what an immediate run would decide. */ +static void test_delay_updates_skip_semantics() { + const char* root = "test_delay_skip_tmp"; + remove_tree(root); + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + cfg->delay_updates = true; + + /* --existing: final destination missing -> skipped, nothing staged. */ + File* missing = make_file("missing.txt", "new"); + EXPECT_NOT_NULL(missing); + // cppcheck-suppress knownConditionTrueFalse + if (!cfg || !missing) + goto out; + cfg->existing = true; + EXPECT_EQ_INT(file_save_to_disk_full(root, missing, cfg), FILE_SAVE_SKIPPED); + cfg->existing = false; + + /* --ignore-existing: final destination present -> skipped. */ + EXPECT_TRUE(file_write_to_disk("test_delay_skip_tmp/existing.txt", "old", 3, false, false)); + File* present = make_file("existing.txt", "new"); + EXPECT_NOT_NULL(present); + // cppcheck-suppress knownConditionTrueFalse + if (!present) + goto out; + cfg->ignore_existing = true; + EXPECT_EQ_INT(file_save_to_disk_full(root, present, cfg), FILE_SAVE_SKIPPED); + cfg->ignore_existing = false; + + /* Without a skip flag the file is staged and later published. */ + File* fresh = make_file("fresh.txt", "content"); + EXPECT_NOT_NULL(fresh); + // cppcheck-suppress knownConditionTrueFalse + if (!fresh) + goto out; + EXPECT_EQ_INT(file_save_to_disk_full(root, fresh, cfg), FILE_SAVE_WRITTEN); + EXPECT_TRUE(delay_updates_publish(cfg->delay_context, cfg)); + char* content = read_all("test_delay_skip_tmp/fresh.txt"); + EXPECT_NOT_NULL(content); + // cppcheck-suppress knownConditionTrueFalse + if (content) { + EXPECT_EQ_STR(content, "content"); + free(content); + } + +out: + file_destroy(missing); + file_destroy(present); + file_destroy(fresh); + config_delete(cfg); + remove_tree(root); +} + +/* The reserved staging name must be recognizable for validation, including + with a trailing slash. */ +static void test_delay_updates_reserved_name_helper() { + EXPECT_TRUE(delay_updates_staging_name_conflict(".fastsync-stage")); + EXPECT_TRUE(delay_updates_staging_name_conflict(".fastsync-stage/")); + EXPECT_TRUE(delay_updates_staging_name_conflict(".fastsync-stage///")); + EXPECT_FALSE(delay_updates_staging_name_conflict(NULL)); + EXPECT_FALSE(delay_updates_staging_name_conflict("")); + EXPECT_FALSE(delay_updates_staging_name_conflict("backups")); + EXPECT_FALSE(delay_updates_staging_name_conflict(".fastsync-stage.bak")); +} + +void test_delay_updates() { + test_delay_updates_reserved_name_helper(); + test_delay_updates_no_final_before_publish(); + test_delay_updates_publish_installs_files(); + test_delay_updates_cleanup_removes_staged(); + test_delay_updates_backup_deferred_to_publish(); + test_delay_updates_skip_semantics(); +} diff --git a/tests/test_delay_updates.h b/tests/test_delay_updates.h new file mode 100644 index 0000000..798b16c --- /dev/null +++ b/tests/test_delay_updates.h @@ -0,0 +1,6 @@ +#ifndef TEST_DELAY_UPDATES_H +#define TEST_DELAY_UPDATES_H + +void test_delay_updates(void); + +#endif diff --git a/tests/test_delta.c b/tests/test_delta.c index 09a6467..cfc8846 100644 --- a/tests/test_delta.c +++ b/tests/test_delta.c @@ -35,6 +35,10 @@ static void test_xxhash32_different_data() { EXPECT_TRUE(ha != hb); } +static void test_xxhash64_different_data() { + EXPECT_TRUE(delta_xxhash64("AAAA", 4) != delta_xxhash64("BBBB", 4)); +} + static void test_signature_roundtrip() { char old_data[4096]; for (int i = 0; i < 4096; i++) @@ -323,11 +327,372 @@ static void test_large_file_delta() { free(new_data); } +static void test_delta_apply_rejects_output_overflow() { + uint8_t old_data[8] = {0}; + uint8_t literal_data[2] = {'x', 'y'}; + DeltaInstruction instruction = { + .type = DELTA_INSTR_LITERAL, + .literal = {.data = literal_data, .length = sizeof(literal_data)}, + }; + Delta delta = { + .new_file_size = 1, + .instruction_count = 1, + .instructions = &instruction, + }; + + EXPECT_TRUE(delta_apply(old_data, sizeof(old_data), &delta, 1) == NULL); +} + +/* --------------------------------------------------------------------------- + * Hash-index lookup differential tests. + * + * delta_compute buckets signature blocks by their weak checksum. These tests + * prove the bucket-indexed candidate lookup is behaviour-identical to the + * original per-window linear scan: the emitted instruction stream (types, + * lengths, literal bytes and chosen block indices) must match a naive linear + * reference exactly, and the delta must reconstruct the new buffer. + * ------------------------------------------------------------------------- */ + +#define REF_NO_MATCH UINT32_MAX + +typedef struct { + DeltaInstruction* items; + uint32_t count; + uint32_t cap; +} RefDelta; + +static void ref_delta_free(RefDelta* ref) { + if (!ref->items) + return; + for (uint32_t i = 0; i < ref->count; i++) + if (ref->items[i].type == DELTA_INSTR_LITERAL) + free(ref->items[i].literal.data); + free(ref->items); + ref->items = NULL; + ref->count = 0; + ref->cap = 0; +} + +static bool ref_delta_push(RefDelta* ref, DeltaInstruction instr) { + if (ref->count == ref->cap) { + uint32_t new_cap = ref->cap ? ref->cap * 2 : 16; + DeltaInstruction* tmp = realloc(ref->items, (size_t)new_cap * sizeof(DeltaInstruction)); + if (!tmp) + return false; + ref->items = tmp; + ref->cap = new_cap; + } + ref->items[ref->count++] = instr; + return true; +} + +static bool ref_delta_flush_literal(RefDelta* ref, const uint8_t* data, uint64_t start, + uint64_t end) { + if (start >= end) + return true; + uint8_t* lit = malloc((size_t)(end - start)); + if (!lit) + return false; + memcpy(lit, data + start, (size_t)(end - start)); + DeltaInstruction instr = { + .type = DELTA_INSTR_LITERAL, + .literal = {.data = lit, .length = (uint32_t)(end - start)}, + }; + return ref_delta_push(ref, instr); +} + +/* Naive O(windows x blocks) re-implementation of the historical delta_compute + * candidate scan: only full windows may match, a candidate needs both the weak + * (Adler-32) and strong (xxHash32) checksums to agree, and the lowest matching + * block index is selected. */ +static bool ref_delta_build(RefDelta* ref, const uint8_t* new_data, uint64_t new_size, + const DeltaSignature* sig) { + uint32_t block_size = sig->block_size; + uint64_t i = 0; + uint64_t literal_start = 0; + bool has_literal = false; + + while (i < new_size) { + uint32_t window_len = (uint32_t)((new_size - i < block_size) ? (new_size - i) : block_size); + bool full_window = (window_len == block_size); + + uint32_t matched = REF_NO_MATCH; + if (full_window) { + uint32_t adler = delta_adler32(new_data + i, window_len); + for (uint32_t j = 0; j < sig->block_count; j++) { + if (sig->blocks[j].adler32 == adler && + delta_xxhash32(new_data + i, window_len) == sig->blocks[j].xxhash) { + matched = j; + break; + } + } + } + + if (matched != REF_NO_MATCH) { + if (has_literal) { + if (!ref_delta_flush_literal(ref, new_data, literal_start, i)) + return false; + has_literal = false; + } + DeltaInstruction instr = { + .type = DELTA_INSTR_BLOCK_MATCH, + .match = {.block_index = matched, .block_offset = 0, .length = window_len}, + }; + if (!ref_delta_push(ref, instr)) + return false; + i += window_len; + } else { + if (!has_literal) { + literal_start = i; + has_literal = true; + } + i++; + } + } + + if (has_literal && !ref_delta_flush_literal(ref, new_data, literal_start, new_size)) + return false; + return true; +} + +static bool ref_delta_matches(const RefDelta* ref, const Delta* delta) { + if (ref->count != delta->instruction_count) + return false; + for (uint32_t i = 0; i < ref->count; i++) { + const DeltaInstruction* a = &ref->items[i]; + const DeltaInstruction* b = &delta->instructions[i]; + if (a->type != b->type) + return false; + if (a->type == DELTA_INSTR_BLOCK_MATCH) { + if (a->match.block_index != b->match.block_index || + a->match.block_offset != b->match.block_offset || a->match.length != b->match.length) + return false; + } else { + if (a->literal.length != b->literal.length || + memcmp(a->literal.data, b->literal.data, a->literal.length) != 0) + return false; + } + } + return true; +} + +static void expect_linear_reference_match(const uint8_t* old_data, uint64_t old_size, + const uint8_t* new_data, uint64_t new_size, + uint32_t block_size, const char* label) { + DeltaSignature* sig = delta_signature_create(old_data, old_size, block_size); + if (!sig) { + printf(" [FAIL] %s: signature creation failed\n", label); + EXPECT_NOT_NULL(sig); + return; + } + Delta* delta = delta_compute(new_data, new_size, sig, block_size); + if (!delta) { + printf(" [FAIL] %s: delta_compute returned NULL\n", label); + delta_signature_destroy(sig); + EXPECT_NOT_NULL(delta); + return; + } + RefDelta ref = {0}; + bool ok = ref_delta_build(&ref, new_data, new_size, sig); + if (ok) + ok = ref_delta_matches(&ref, delta); + if (!ok) { + printf(" [FAIL] %s: instruction stream differs from linear reference " + "(linear=%u indexed=%u)\n", + label, ref.count, delta->instruction_count); + } + ref_delta_free(&ref); + delta_destroy(delta); + delta_signature_destroy(sig); + EXPECT_TRUE(ok); +} + +static void fill_delta_pattern(uint8_t* buf, uint64_t size, uint32_t seed) { + uint32_t x = seed ? seed : 1; + for (uint64_t i = 0; i < size; i++) { + x ^= x << 13; + x ^= x >> 17; + x ^= x << 5; + buf[i] = (uint8_t)(x >> 24); + } +} + +static void test_delta_hash_index_matches_linear_reference() { + /* Identical file (full block alignment). */ + uint8_t old_a[32768]; + uint8_t new_a[32768]; + fill_delta_pattern(old_a, sizeof(old_a), 42); + memcpy(new_a, old_a, sizeof(old_a)); + expect_linear_reference_match(old_a, sizeof(old_a), new_a, sizeof(new_a), 2048, + "identical 32KiB @ 2KiB"); + + /* Scattered single-byte edits in the middle of each block. */ + uint8_t new_b[32768]; + memcpy(new_b, old_a, sizeof(old_a)); + for (size_t p = 100; p < sizeof(new_b); p += 4096) + new_b[p] ^= 0x5A; + expect_linear_reference_match(old_a, sizeof(old_a), new_b, sizeof(new_b), 2048, + "32KiB scattered single-byte edits @ 2KiB"); + + /* Non-aligned old file (partial final block) with a single edit. */ + uint8_t old_c[30000]; + uint8_t new_c[30000]; + fill_delta_pattern(old_c, sizeof(old_c), 7); + memcpy(new_c, old_c, sizeof(old_c)); + new_c[15000] ^= 0x3C; + expect_linear_reference_match(old_c, sizeof(old_c), new_c, sizeof(new_c), 2048, + "30KiB partial-tail single edit @ 2KiB"); + + /* Growth: appended data after an identical prefix. */ + uint8_t old_d[24576]; + uint8_t new_d[34576]; + fill_delta_pattern(old_d, sizeof(old_d), 11); + memcpy(new_d, old_d, sizeof(old_d)); + fill_delta_pattern(new_d + sizeof(old_d), sizeof(new_d) - sizeof(old_d), 23); + expect_linear_reference_match(old_d, sizeof(old_d), new_d, sizeof(new_d), 2048, + "24KiB -> 34KiB appended @ 2KiB"); + + /* Insertion shifting everything after the edit point (rsync re-sync). */ + uint8_t old_e[65536]; + uint8_t new_e[65536 + 3000]; + fill_delta_pattern(old_e, sizeof(old_e), 99); + memcpy(new_e, old_e, 20000); + fill_delta_pattern(new_e + 20000, 3000, 101); + memcpy(new_e + 23000, old_e + 20000, sizeof(old_e) - 20000); + expect_linear_reference_match(old_e, sizeof(old_e), new_e, sizeof(new_e), 2048, + "64KiB + 3KiB insertion @ 2KiB"); + + /* Deletion shrinking the file. */ + uint8_t new_f[sizeof(old_e) - 5000]; + memcpy(new_f, old_e, 30000); + memcpy(new_f + 30000, old_e + 35000, sizeof(old_e) - 35000); + expect_linear_reference_match(old_e, sizeof(old_e), new_f, sizeof(new_f), 2048, + "64KiB - 5KiB deletion @ 2KiB"); + + /* Repeated identical blocks must resolve to the lowest block index. */ + uint8_t old_g[4 * 4096]; + uint8_t new_g[4 * 4096]; + for (uint32_t b = 0; b < 4; b++) + fill_delta_pattern(old_g + b * 4096, 4096, b % 2 == 0 ? 500 : 501); /* block0==block2 */ + memcpy(new_g, old_g, sizeof(old_g)); + new_g[4096 + 5] ^= 0x11; /* edit inside the second (duplicated) chunk */ + expect_linear_reference_match(old_g, sizeof(old_g), new_g, sizeof(new_g), 4096, + "duplicated chunks @ 4KiB"); + + /* Block larger than the file: nothing can match, all literal. */ + uint8_t old_h[1000]; + uint8_t new_h[1000]; + fill_delta_pattern(old_h, sizeof(old_h), 3); + memcpy(new_h, old_h, sizeof(old_h)); + expect_linear_reference_match(old_h, sizeof(old_h), new_h, sizeof(new_h), 4096, + "1KiB file @ 4KiB block"); +} + +static void test_delta_hash_index_large_mostly_matching() { + const uint64_t size = 4ULL * 1024 * 1024; + const uint32_t block_size = 8192; + + uint8_t* old_data = malloc((size_t)size); + uint8_t* new_data = malloc((size_t)size); + EXPECT_TRUE(old_data != NULL && new_data != NULL); + + fill_delta_pattern(old_data, size, 1234); + memcpy(new_data, old_data, (size_t)size); + + /* Scattered single-byte changes across the whole buffer. Each change forces + * the diff to re-synchronise by walking one byte at a time through the + * affected block, which is exactly the case that used to cost O(bytes x + * blocks) with the linear scan. */ + const uint64_t nchanges = 64; + for (uint64_t c = 0; c < nchanges; c++) { + uint64_t pos = (c * (size / nchanges)) + (c % 17); + new_data[pos] ^= (uint8_t)(0xA0 + (c % 16)); + } + + DeltaSignature* sig = delta_signature_create(old_data, size, block_size); + EXPECT_NOT_NULL(sig); + EXPECT_EQ_INT((int)sig->block_count, (int)(size / block_size)); + + Delta* delta = delta_compute(new_data, size, sig, block_size); + EXPECT_NOT_NULL(delta); + EXPECT_EQ_INT((int)delta->new_file_size, (int)size); + + uint32_t match_count = 0; + for (uint32_t i = 0; i < delta->instruction_count; i++) + if (delta->instructions[i].type == DELTA_INSTR_BLOCK_MATCH) + match_count++; + EXPECT_TRUE(match_count > 0); + + void* result = delta_apply(old_data, size, delta, block_size); + EXPECT_NOT_NULL(result); + EXPECT_EQ_INT(memcmp(result, new_data, (size_t)size), 0); + + free(result); + delta_destroy(delta); + delta_signature_destroy(sig); + free(old_data); + free(new_data); +} + +/* --checksum-seed: the delta strong (block) hash is genuinely seed-aware. A + * nonzero seed changes the per-block xxHash32, and a signature + delta computed + * with the same seed still reconstruct the file exactly (symmetric), while a + * mismatched seed produces a delta that does not match the signature blocks. */ +static void test_delta_xxhash32_seeded() { + const char* data = "seedme"; + uint32_t a = delta_xxhash32(data, 6); + uint32_t b = delta_xxhash32_seeded(data, 6, 42); + uint32_t c = delta_xxhash32_seeded(data, 6, 42); + EXPECT_TRUE(a != b); + EXPECT_EQ_INT((int)b, (int)c); + /* Unseeded == seeded with 0 (default reproduces today's behavior). */ + EXPECT_EQ_INT((int)delta_xxhash32(data, 6), (int)delta_xxhash32_seeded(data, 6, 0)); +} + +static void test_delta_seeded_signature_compute_matches() { + uint32_t block_size = 1024; + /* Identical old/new data with a non-zero seed: the receiver builds a seeded + signature and the sender computes a seeded delta over the same bytes, so + every block matches and applying the delta rebuilds the file exactly. */ + char data[4096]; + for (int i = 0; i < 4096; i++) + data[i] = (char)(i % 256); + + DeltaSignature* sig = delta_signature_create_seeded(data, 4096, block_size, 99); + EXPECT_NOT_NULL(sig); + Delta* delta = delta_compute_seeded(data, 4096, sig, block_size, 99); + EXPECT_NOT_NULL(delta); + void* rebuilt = delta_apply(data, 4096, delta, block_size); + EXPECT_NOT_NULL(rebuilt); + EXPECT_TRUE(memcmp(rebuilt, data, 4096) == 0); + free(rebuilt); + delta_destroy(delta); + delta_signature_destroy(sig); + + /* A MISMATCHED seed means the sender's window xxHash32 never equals the + receiver's signature-block xxHash32: no block can match, so the delta is + not worthwhile / has no block matches. This proves the seed really gates + the block comparison rather than being an inert parameter. */ + sig = delta_signature_create_seeded(data, 4096, block_size, 99); + EXPECT_NOT_NULL(sig); + delta = delta_compute_seeded(data, 4096, sig, block_size, 7); + EXPECT_NOT_NULL(delta); + bool any_match = false; + for (uint32_t i = 0; i < delta->instruction_count; i++) + if (delta->instructions[i].type == DELTA_INSTR_BLOCK_MATCH) + any_match = true; + EXPECT_FALSE(any_match); + delta_destroy(delta); + delta_signature_destroy(sig); +} + void test_delta() { test_adler32_basic(); test_adler32_different_data(); test_xxhash32_basic(); test_xxhash32_different_data(); + test_xxhash64_different_data(); + test_delta_xxhash32_seeded(); test_signature_roundtrip(); test_delta_identical_files(); test_delta_small_edit(); @@ -338,4 +703,8 @@ void test_delta() { test_should_attempt(); test_is_worthwhile(); test_large_file_delta(); + test_delta_apply_rejects_output_overflow(); + test_delta_hash_index_matches_linear_reference(); + test_delta_hash_index_large_mostly_matching(); + test_delta_seeded_signature_compute_matches(); } diff --git a/tests/test_file.c b/tests/test_file.c index 7e74232..fd715fc 100644 --- a/tests/test_file.c +++ b/tests/test_file.c @@ -1,13 +1,22 @@ +#ifndef _GNU_SOURCE +#define _GNU_SOURCE /* SEEK_HOLE/SEEK_DATA for the sparse-hole sparseness check */ +#endif #include "test_file.h" #include "file.h" +#include "file_store.h" +#include "file_receive.h" #include "data.h" +#include "config.h" #include "utils.h" #include "protocol.h" #include "test_utils.h" +#include +#include #include #include #include #include +#include #include static void test_file_create() { @@ -22,6 +31,29 @@ static void test_file_create() { file_destroy(f); } +/* rdev/type validation shared by the wire path and the secure recreation site: + * a legal char/block major/minor pair is accepted, out-of-range / negative + * values and non-device entries carrying an rdev are rejected. */ +static void test_file_special_rdev_valid() { + mode_t fake_char = S_IFCHR | 0600; + mode_t fake_blk = S_IFBLK | 0600; + mode_t fake_fifo = S_IFIFO | 0600; + mode_t fake_sock = S_IFSOCK | 0600; + /* char/block devices: accept a legal pair, reject negative / oversized. */ + EXPECT_TRUE(file_special_rdev_valid(1, 3, fake_char)); + EXPECT_TRUE(file_special_rdev_valid(0xffff, 0x00ffffff, fake_blk)); + EXPECT_FALSE(file_special_rdev_valid(-1, 3, fake_char)); + EXPECT_FALSE(file_special_rdev_valid(1, -1, fake_char)); + EXPECT_FALSE(file_special_rdev_valid(0x10000, 3, fake_char)); + EXPECT_FALSE(file_special_rdev_valid(1, 0x1000000, fake_char)); + /* FIFOs/sockets must carry an empty rdev. */ + EXPECT_TRUE(file_special_rdev_valid(0, 0, fake_fifo)); + EXPECT_FALSE(file_special_rdev_valid(1, 0, fake_fifo)); + EXPECT_TRUE(file_special_rdev_valid(0, 0, fake_sock)); + EXPECT_FALSE(file_special_rdev_valid(0, 1, fake_sock)); + EXPECT_FALSE(file_special_rdev_valid(0, 0, (mode_t)(S_IFREG | 0600))); +} + static void test_file_destroy_null() { file_destroy(NULL); } @@ -34,7 +66,8 @@ static void test_file_destroy_normal() { static void test_file_load_data() { const char* content = "Hello Load Test"; - EXPECT_TRUE(to_disk("test_file_load_data.txt", content, strlen(content))); + EXPECT_TRUE( + file_write_to_disk("test_file_load_data.txt", content, strlen(content), false, false)); struct stat st; EXPECT_EQ_INT(stat("test_file_load_data.txt", &st), 0); @@ -87,15 +120,282 @@ static void test_file_save_to_disk() { rmdir("test_save_tmp"); } -static void test_to_disk_basic() { - const char* content = "Basic to_disk test"; - EXPECT_TRUE(to_disk("test_to_disk_basic.txt", content, strlen(content))); +static void test_file_save_to_disk_with_fsync_config() { + File* f = file_create("saved_file_fsync.txt"); + EXPECT_NOT_NULL(f); + const char* content = "Save to disk with fsync"; + f->data->data = malloc(strlen(content)); + EXPECT_NOT_NULL(f->data->data); + memcpy(f->data->data, content, strlen(content)); + f->data->size = strlen(content); + + Config* config = config_create(); + EXPECT_NOT_NULL(config); + config->use_fsync = true; + EXPECT_TRUE(file_save_to_disk("test_save_fsync_tmp", f, config)); struct stat st; - EXPECT_EQ_INT(stat("test_to_disk_basic.txt", &st), 0); + EXPECT_EQ_INT(stat("test_save_fsync_tmp/saved_file_fsync.txt", &st), 0); EXPECT_EQ_INT((int)st.st_size, (int)strlen(content)); - FILE* fp = fopen("test_to_disk_basic.txt", "rb"); + file_destroy(f); + config_delete(config); + unlink("test_save_fsync_tmp/saved_file_fsync.txt"); + rmdir("test_save_fsync_tmp"); +} + +static void test_file_save_to_disk_existing() { + const char* root = "test_existing_tmp"; + const char* existing_path = "test_existing_tmp/existing.txt"; + const char* missing_path = "test_existing_tmp/missing.txt"; + EXPECT_TRUE(file_write_to_disk(existing_path, "old", 3, false, false)); + + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + cfg->existing = true; + + File* existing = file_create("existing.txt"); + EXPECT_NOT_NULL(existing); + existing->data->data = malloc(3); + EXPECT_NOT_NULL(existing->data->data); + memcpy(existing->data->data, "new", 3); + existing->data->size = 3; + EXPECT_TRUE(file_save_to_disk(root, existing, cfg)); + file_destroy(existing); + + File* missing = file_create("missing.txt"); + EXPECT_NOT_NULL(missing); + missing->data->data = malloc(7); + EXPECT_NOT_NULL(missing->data->data); + memcpy(missing->data->data, "skipped", 7); + missing->data->size = 7; + EXPECT_TRUE(file_save_to_disk(root, missing, cfg)); + file_destroy(missing); + + FILE* fp = fopen(existing_path, "rb"); + char content[4] = {0}; + EXPECT_NOT_NULL(fp); + // cppcheck-suppress knownConditionTrueFalse + if (fp) { + EXPECT_EQ_INT((int)fread(content, 1, 3, fp), 3); + fclose(fp); + } + EXPECT_EQ_STR(content, "new"); + EXPECT_EQ_INT(access(missing_path, F_OK), -1); + + config_delete(cfg); + unlink(existing_path); + rmdir("test_existing_tmp"); +} + +static void test_file_save_to_disk_ignore_existing() { + const char* path = "test_ignore_existing_tmp/existing.txt"; + EXPECT_TRUE(file_write_to_disk(path, "old", 3, false, false)); + + File* file = file_create("existing.txt"); + EXPECT_NOT_NULL(file); + file->data->data = malloc(3); + EXPECT_NOT_NULL(file->data->data); + memcpy(file->data->data, "new", 3); + file->data->size = 3; + + Config* config = config_create(); + EXPECT_NOT_NULL(config); + config->ignore_existing = true; + EXPECT_TRUE(file_save_to_disk("test_ignore_existing_tmp", file, config)); + + FILE* stream = fopen(path, "rb"); + char content[4] = {0}; + EXPECT_NOT_NULL(stream); + // cppcheck-suppress knownConditionTrueFalse + if (stream) { + EXPECT_EQ_INT((int)fread(content, 1, 3, stream), 3); + fclose(stream); + } + EXPECT_EQ_STR(content, "old"); + + file_destroy(file); + config_delete(config); + unlink(path); + rmdir("test_ignore_existing_tmp"); +} + +static void test_file_save_to_disk_ignore_existing_entry_types() { + const char* root = "test_ignore_existing_entries_tmp"; + const char* directory = "test_ignore_existing_entries_tmp/directory"; + const char* link = "test_ignore_existing_entries_tmp/link"; + const char* target = "test_ignore_existing_entries_tmp/target"; + const char* backup = "test_ignore_existing_entries_tmp/backup.txt~"; + const char* backup_file = "test_ignore_existing_entries_tmp/backup.txt"; + Config* config = config_create(); + File* file = file_create("unused"); + + unlink(link); + unlink(target); + unlink(backup); + unlink(backup_file); + rmdir(directory); + rmdir(root); + EXPECT_NOT_NULL(config); + EXPECT_NOT_NULL(file); + // cppcheck-suppress knownConditionTrueFalse + if (!config || !file) + return; + config->ignore_existing = true; + config->backup = true; + file->data->data = malloc(3); + EXPECT_NOT_NULL(file->data->data); + // cppcheck-suppress knownConditionTrueFalse + if (!file->data->data) { + file_destroy(file); + config_delete(config); + return; + } + memcpy(file->data->data, "new", 3); + file->data->size = 3; + + EXPECT_EQ_INT(mkdir(root, 0755), 0); + EXPECT_EQ_INT(mkdir(directory, 0755), 0); + EXPECT_TRUE(file_write_to_disk(target, "old", 3, false, false)); + EXPECT_EQ_INT(symlink("target", link), 0); + free(file->path); + file->path = str_dup("directory"); + EXPECT_TRUE(file_save_to_disk(root, file, config)); + free(file->path); + file->path = str_dup("link"); + EXPECT_TRUE(file_save_to_disk(root, file, config)); + + free(file->path); + file->path = str_dup("backup.txt"); + EXPECT_TRUE(file_write_to_disk(backup_file, "old", 3, false, false)); + EXPECT_TRUE(file_save_to_disk(root, file, config)); + EXPECT_TRUE(file_path_exists_secure(backup_file)); + EXPECT_FALSE(file_path_exists_secure(backup)); + + file_destroy(file); + config_delete(config); + unlink(link); + unlink(target); + unlink(backup_file); + rmdir(directory); + rmdir(root); +} + +/* Issue #253: with --partial --partial-dir a completed write must be installed + at the real destination rather than left under the partial directory. */ +static void test_file_save_to_disk_partial_install() { + const char* root = "test_partial_install_tmp"; + const char* dest_file = "test_partial_install_tmp/file.txt"; + const char* partial_file = "test_partial_install_tmp/.partial/file.txt"; + unlink(dest_file); + unlink(partial_file); + rmdir("test_partial_install_tmp/.partial"); + rmdir(root); + + File* f = file_create("file.txt"); + EXPECT_NOT_NULL(f); + const char* content = "partial-dir content"; + f->data->data = malloc(strlen(content)); + EXPECT_NOT_NULL(f->data->data); + memcpy(f->data->data, content, strlen(content)); + f->data->size = strlen(content); + + Config* config = config_create(); + EXPECT_NOT_NULL(config); + config->partial = true; + config->partial_dir = str_dup(".partial"); + + EXPECT_EQ_INT(file_save_to_disk_full(root, f, config), FILE_SAVE_WRITTEN); + + FILE* fp = fopen(dest_file, "rb"); + EXPECT_NOT_NULL(fp); + // cppcheck-suppress knownConditionTrueFalse + if (fp) { + char buf[64] = {0}; + size_t nread = fread(buf, 1, sizeof(buf) - 1, fp); + fclose(fp); + EXPECT_EQ_INT((int)nread, (int)strlen(content)); + EXPECT_EQ_INT(memcmp(buf, content, strlen(content)), 0); + } + /* A completed transfer must not linger under the partial dir. */ + EXPECT_EQ_INT(access(partial_file, F_OK), -1); + + file_destroy(f); + config_delete(config); + unlink(dest_file); + rmdir(root); +} + +/* Issue #251: file_save_to_disk_full must distinguish receiver-side skips + (--existing/--ignore-existing/--update) from real writes so the sender can + decide whether --remove-source-files may unlink its source. */ +static void test_file_save_to_disk_reports_skips() { + const char* root = "test_save_skip_tmp"; + const char* existing_path = "test_save_skip_tmp/existing.txt"; + unlink(existing_path); + rmdir(root); + EXPECT_TRUE(file_write_to_disk(existing_path, "old", 3, false, false)); + + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + + File* new_file = file_create("missing.txt"); + EXPECT_NOT_NULL(new_file); + new_file->data->data = malloc(7); + EXPECT_NOT_NULL(new_file->data->data); + memcpy(new_file->data->data, "skipped", 7); + new_file->data->size = 7; + + /* --existing: destination is missing -> skipped, not an error. */ + cfg->existing = true; + EXPECT_EQ_INT(file_save_to_disk_full(root, new_file, cfg), FILE_SAVE_SKIPPED); + cfg->existing = false; + + /* --ignore-existing: destination present -> skipped. */ + File* present = file_create("existing.txt"); + EXPECT_NOT_NULL(present); + present->data->data = malloc(3); + EXPECT_NOT_NULL(present->data->data); + memcpy(present->data->data, "new", 3); + present->data->size = 3; + cfg->ignore_existing = true; + EXPECT_EQ_INT(file_save_to_disk_full(root, present, cfg), FILE_SAVE_SKIPPED); + cfg->ignore_existing = false; + + /* A normal overwrite of an existing file is a real write. */ + EXPECT_EQ_INT(file_save_to_disk_full(root, present, cfg), FILE_SAVE_WRITTEN); + + /* --update: a newer destination is skipped. */ + struct stat st; + EXPECT_EQ_INT(stat(existing_path, &st), 0); + time_t now = time(NULL); + FileMetadata metadata = {.mode = st.st_mode, + .uid = st.st_uid, + .gid = st.st_gid, + .mtime_sec = now - 100, + .mtime_nsec = 0}; + present->metadata = &metadata; + cfg->update = true; + EXPECT_EQ_INT(file_save_to_disk_full(root, present, cfg), FILE_SAVE_SKIPPED); + present->metadata = NULL; + + file_destroy(new_file); + file_destroy(present); + config_delete(cfg); + unlink(existing_path); + rmdir(root); +} + +static void test_file_write_to_disk_basic() { + const char* content = "Basic file_write_to_disk test"; + EXPECT_TRUE(file_write_to_disk("test_file_write_to_disk_basic.txt", content, strlen(content), + false, false)); + + struct stat st; + EXPECT_EQ_INT(stat("test_file_write_to_disk_basic.txt", &st), 0); + EXPECT_EQ_INT((int)st.st_size, (int)strlen(content)); + + FILE* fp = fopen("test_file_write_to_disk_basic.txt", "rb"); EXPECT_NOT_NULL(fp); char buf[100]; size_t nread = fread(buf, 1, sizeof(buf), fp); @@ -103,12 +403,60 @@ static void test_to_disk_basic() { EXPECT_EQ_INT((int)nread, (int)strlen(content)); EXPECT_EQ_INT(memcmp(buf, content, strlen(content)), 0); - unlink("test_to_disk_basic.txt"); + unlink("test_file_write_to_disk_basic.txt"); } -static void test_to_disk_creates_dirs() { +static void test_file_write_to_disk_with_fsync() { + const char* path = "test_file_write_to_disk_fsync.txt"; + const char* content = "fsync file content"; + EXPECT_TRUE(file_to_disk_secure_with_fsync(path, content, strlen(content), false, false, false, + NULL, false, true, NULL)); + struct stat st; + EXPECT_EQ_INT(stat(path, &st), 0); + EXPECT_EQ_INT((int)st.st_size, (int)strlen(content)); + unlink(path); +} + +static void test_file_write_to_disk_preallocate_atomic() { + const char* path = "test_file_write_prealloc_atomic.txt"; + const char* content = "prealloc atomic content"; + EXPECT_TRUE( + file_to_disk_secure(path, content, strlen(content), false, false, true, NULL, false, NULL)); + struct stat st; + EXPECT_EQ_INT(stat(path, &st), 0); + EXPECT_EQ_INT((int)st.st_size, (int)strlen(content)); + FILE* fp = fopen(path, "rb"); + EXPECT_NOT_NULL(fp); + char buf[100]; + size_t nread = fread(buf, 1, sizeof(buf), fp); + fclose(fp); + EXPECT_EQ_INT((int)nread, (int)strlen(content)); + EXPECT_EQ_INT(memcmp(buf, content, strlen(content)), 0); + unlink(path); +} + +static void test_file_write_to_disk_preallocate_inplace() { + const char* path = "test_file_write_prealloc_inplace.txt"; + const char* content = "prealloc inplace content"; + EXPECT_TRUE( + file_to_disk_secure(path, content, strlen(content), true, false, true, NULL, false, NULL)); + struct stat st; + EXPECT_EQ_INT(stat(path, &st), 0); + EXPECT_EQ_INT((int)st.st_size, (int)strlen(content)); + FILE* fp = fopen(path, "rb"); + EXPECT_NOT_NULL(fp); + char buf[100]; + size_t nread = fread(buf, 1, sizeof(buf), fp); + fclose(fp); + EXPECT_EQ_INT((int)nread, (int)strlen(content)); + EXPECT_EQ_INT(memcmp(buf, content, strlen(content)), 0); + unlink(path); +} + +static void test_file_write_to_disk_creates_dirs() { const char* content = "Nested dir test"; - EXPECT_TRUE(to_disk("test_nested_tmp/nested/file.txt", content, strlen(content))); + EXPECT_TRUE(file_write_to_disk("test_nested_tmp/nested/file.txt", content, strlen(content), false, + false)); struct stat st; EXPECT_EQ_INT(stat("test_nested_tmp/nested/file.txt", &st), 0); @@ -126,9 +474,76 @@ static void test_to_disk_creates_dirs() { rmdir("test_nested_tmp"); } +static void test_file_write_to_disk_does_not_follow_symlink() { + const char* outside = "test_file_write_to_disk_outside.txt"; + const char* link = "test_file_write_to_disk_link.txt"; + const char* content = "confined"; + unlink(outside); + unlink(link); + EXPECT_TRUE(file_write_to_disk(outside, "outside", 7, false, false)); + EXPECT_EQ_INT(symlink(outside, link), 0); + EXPECT_TRUE(file_write_to_disk(link, content, strlen(content), false, false)); + FILE* fp = fopen(outside, "rb"); + char buf[16] = {0}; + EXPECT_NOT_NULL(fp); + // cppcheck-suppress knownConditionTrueFalse + if (!fp) + return; + size_t read_count = fread(buf, 1, sizeof(buf) - 1, fp); + EXPECT_TRUE(read_count <= sizeof(buf) - 1); + fclose(fp); + EXPECT_EQ_STR(buf, "outside"); + unlink(outside); + unlink(link); +} + +static void test_file_symlink_helpers() { + /* Munge/unmunge round-trip restores the original target. */ + char* munged = file_symlink_munge("target.txt"); + EXPECT_NOT_NULL(munged); + EXPECT_EQ_INT(memcmp(munged, SYMLINK_MUNGE_PREFIX, strlen(SYMLINK_MUNGE_PREFIX)), 0); + EXPECT_TRUE(file_symlink_unmunge(munged)); + EXPECT_EQ_STR(munged, "target.txt"); + free(munged); + + char noop[] = "plain-target"; + EXPECT_FALSE(file_symlink_unmunge(noop)); + EXPECT_EQ_STR(noop, "plain-target"); + + /* Containment: relative targets without ".." are safe; absolute or + ".."-escaping targets are not. */ + EXPECT_TRUE(file_symlink_target_contained("a.txt")); + EXPECT_TRUE(file_symlink_target_contained("sub/dir/file")); + EXPECT_FALSE(file_symlink_target_contained("/etc/passwd")); + EXPECT_FALSE(file_symlink_target_contained("../escape")); + EXPECT_FALSE(file_symlink_target_contained("a/../b")); + EXPECT_FALSE(file_symlink_target_contained("")); +} + +static void test_file_symlink_at_secure() { + const char* link = "test_symlink_at_secure_link"; + const char* outside = "test_symlink_at_secure_outside.txt"; + unlink(link); + unlink(outside); + EXPECT_TRUE(file_write_to_disk(outside, "out", 3, false, false)); + + EXPECT_TRUE(file_symlink_at_secure(link, "outside.text")); + struct stat st; + EXPECT_EQ_INT(lstat(link, &st), 0); + EXPECT_TRUE(S_ISLNK(st.st_mode)); + + /* Replacing an existing non-directory entry is fine. */ + EXPECT_TRUE(file_symlink_at_secure(link, "other.txt")); + EXPECT_EQ_INT(lstat(link, &st), 0); + EXPECT_TRUE(S_ISLNK(st.st_mode)); + + unlink(link); + unlink(outside); +} + static void test_file_content_to_buffer() { const char* content = "Buffer content test"; - EXPECT_TRUE(to_disk("test_buffer_file.txt", content, strlen(content))); + EXPECT_TRUE(file_write_to_disk("test_buffer_file.txt", content, strlen(content), false, false)); File* f = file_create("test_buffer_file.txt"); EXPECT_NOT_NULL(f); @@ -250,11 +665,11 @@ static void test_file_send_no_path() { } static void test_file_metadata_create() { - EXPECT_TRUE(to_disk("test_meta_file.txt", "metadata test", 13)); + EXPECT_TRUE(file_write_to_disk("test_meta_file.txt", "metadata test", 13, false, false)); struct stat st; EXPECT_EQ_INT(stat("test_meta_file.txt", &st), 0); - FileMetadata* m = file_metadata_create(&st); + FileMetadata* m = file_metadata_create("test_meta_file.txt", &st, false, false); EXPECT_NOT_NULL(m); EXPECT_EQ_INT(m->mode, st.st_mode); EXPECT_EQ_INT(m->uid, st.st_uid); @@ -359,7 +774,7 @@ static void test_file_send_single_calls_metadata_and_path() { /* Create a real file on disk so we can have metadata */ const char* content = "File with metadata"; size_t len = strlen(content); - EXPECT_TRUE(to_disk("test_meta_send.txt", content, len)); + EXPECT_TRUE(file_write_to_disk("test_meta_send.txt", content, len, false, false)); struct stat st; EXPECT_EQ_INT(stat("test_meta_send.txt", &st), 0); @@ -370,7 +785,7 @@ static void test_file_send_single_calls_metadata_and_path() { file->data->data = malloc(len); EXPECT_NOT_NULL(file->data->data); memcpy(file->data->data, content, len); - file->metadata = file_metadata_create(&st); + file->metadata = file_metadata_create("test_meta_send.txt", &st, false, false); EXPECT_NOT_NULL(file->metadata); Config* cfg = config_create(); @@ -425,18 +840,684 @@ static void test_file_send_single_calls_metadata_and_path() { } } +static void test_inplace_overwrite_clears_special_mode_bits() { + const char* root = "test_inplace_tmp"; + const char* path = "test_inplace_tmp/priv.txt"; + const char* content = "olddata"; + unlink(path); + rmdir(root); + EXPECT_EQ_INT(mkdir(root, 0700), 0); + + /* Create a destination carrying setuid + sticky bits. */ + int fd = open(path, O_WRONLY | O_CREAT | O_TRUNC | O_CLOEXEC, 0644); + EXPECT_TRUE(fd >= 0); + // cppcheck-suppress knownConditionTrueFalse + if (fd < 0) { + rmdir(root); + return; + } + EXPECT_EQ_INT((int)write(fd, content, strlen(content)), (int)strlen(content)); + EXPECT_EQ_INT(fchmod(fd, S_ISUID | S_ISVTX | 0755), 0); + EXPECT_EQ_INT(close(fd), 0); + + /* Overwrite in place without metadata: the mode must be normalized to a + safe default (0644) and the setuid/sticky bits must be gone. */ + File* f = file_create("priv.txt"); + EXPECT_NOT_NULL(f); + const char* new_content = "newdata"; + f->data->data = malloc(strlen(new_content)); + EXPECT_NOT_NULL(f->data->data); + memcpy(f->data->data, new_content, strlen(new_content)); + f->data->size = strlen(new_content); + + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + cfg->inplace = true; + EXPECT_TRUE(file_save_to_disk(root, f, cfg)); + file_destroy(f); + config_delete(cfg); + + struct stat st; + EXPECT_EQ_INT(stat(path, &st), 0); + EXPECT_EQ_INT((int)(st.st_mode & (S_ISUID | S_ISGID | S_ISVTX)), 0); + EXPECT_EQ_INT((int)(st.st_mode & 0777), 0644); + FILE* stream = fopen(path, "rb"); + char buf[16] = {0}; + EXPECT_NOT_NULL(stream); + // cppcheck-suppress knownConditionTrueFalse + if (stream) { + size_t nread = fread(buf, 1, sizeof(buf) - 1, stream); + fclose(stream); + EXPECT_EQ_INT((int)nread, (int)strlen(new_content)); + } + EXPECT_EQ_STR(buf, new_content); + + unlink(path); + rmdir(root); +} + +static void test_inplace_overwrite_metadata_strips_special_bits() { + const char* root = "test_inplace_meta_tmp"; + const char* path = "test_inplace_meta_tmp/meta.txt"; + const char* source = "test_inplace_meta_source.txt"; + unlink(path); + unlink(source); + rmdir(root); + EXPECT_EQ_INT(mkdir(root, 0700), 0); + + /* Existing destination with setuid+sticky set. */ + int fd = open(path, O_WRONLY | O_CREAT | O_TRUNC | O_CLOEXEC, 0644); + EXPECT_TRUE(fd >= 0); + // cppcheck-suppress knownConditionTrueFalse + if (fd < 0) { + rmdir(root); + return; + } + EXPECT_EQ_INT((int)write(fd, "olddata", 7), 7); + EXPECT_EQ_INT(fchmod(fd, S_ISUID | S_ISVTX | 0755), 0); + EXPECT_EQ_INT(close(fd), 0); + + /* Build source metadata carrying a plain executable mode (no specials). */ + EXPECT_TRUE(file_write_to_disk(source, "source", 6, false, false)); + EXPECT_EQ_INT(chmod(source, 0755), 0); + struct stat source_st; + EXPECT_EQ_INT(stat(source, &source_st), 0); + + File* f = file_create("meta.txt"); + EXPECT_NOT_NULL(f); + const char* new_content = "meta"; + f->data->data = malloc(strlen(new_content)); + EXPECT_NOT_NULL(f->data->data); + memcpy(f->data->data, new_content, strlen(new_content)); + f->data->size = strlen(new_content); + f->metadata = file_metadata_create(source, &source_st, false, false); + EXPECT_NOT_NULL(f->metadata); + + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + cfg->inplace = true; + EXPECT_TRUE(file_save_to_disk(root, f, cfg)); + file_destroy(f); + config_delete(cfg); + unlink(source); + + struct stat st; + EXPECT_EQ_INT(stat(path, &st), 0); + /* Metadata-derived mode is applied and never includes setuid/setgid/sticky. */ + EXPECT_EQ_INT((int)(st.st_mode & (S_ISUID | S_ISGID | S_ISVTX)), 0); + EXPECT_EQ_INT((int)(st.st_mode & 0777), 0755); + + unlink(path); + rmdir(root); +} + +static void test_inplace_overwrite_truncates_shorter_payload() { + const char* root = "test_inplace_trunc_tmp"; + const char* path = "test_inplace_trunc_tmp/big.txt"; + unlink(path); + rmdir(root); + EXPECT_EQ_INT(mkdir(root, 0700), 0); + + const char* old_content = "0123456789abcdef"; /* 16 bytes */ + EXPECT_TRUE(file_write_to_disk(path, old_content, strlen(old_content), false, false)); + + File* f = file_create("big.txt"); + EXPECT_NOT_NULL(f); + const char* new_content = "hi"; + f->data->data = malloc(strlen(new_content)); + EXPECT_NOT_NULL(f->data->data); + memcpy(f->data->data, new_content, strlen(new_content)); + f->data->size = strlen(new_content); + + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + cfg->inplace = true; + EXPECT_TRUE(file_save_to_disk(root, f, cfg)); + file_destroy(f); + config_delete(cfg); + + /* A shorter payload must truncate the file: no stale trailing bytes. */ + struct stat st; + EXPECT_EQ_INT(stat(path, &st), 0); + EXPECT_EQ_INT((int)st.st_size, (int)strlen(new_content)); + FILE* stream = fopen(path, "rb"); + char buf[32] = {0}; + EXPECT_NOT_NULL(stream); + // cppcheck-suppress knownConditionTrueFalse + if (stream) { + size_t nread = fread(buf, 1, sizeof(buf) - 1, stream); + fclose(stream); + EXPECT_EQ_INT((int)nread, (int)strlen(new_content)); + } + EXPECT_EQ_STR(buf, new_content); + + unlink(path); + rmdir(root); +} + +/* Explicit directory entries (--dirs) create the directory under the receive + root through the same save funnel, creating parents as needed, and reject + traversal the same way a file path does. */ +static void test_dir_entry_save_to_disk() { + const char* root = "test_dir_entry_root"; + EXPECT_EQ_INT(mkdir(root, 0755), 0); + + Config* config = config_create(); + EXPECT_NOT_NULL(config); + + File* dir = file_create("alpha/beta/gamma"); + EXPECT_NOT_NULL(dir); + dir->is_dir = true; + EXPECT_EQ_INT(file_save_to_disk_full(root, dir, config), FILE_SAVE_WRITTEN); + EXPECT_EQ_INT(file_save_to_disk_full(root, dir, config), FILE_SAVE_WRITTEN); + file_destroy(dir); + + struct stat st; + EXPECT_EQ_INT(stat("test_dir_entry_root/alpha/beta/gamma", &st), 0); + EXPECT_TRUE(S_ISDIR(st.st_mode)); + + /* The directory-entry save path never follows or escapes. */ + File* evil = file_create("../dir_entry_escape"); + EXPECT_NOT_NULL(evil); + evil->is_dir = true; + EXPECT_EQ_INT(file_save_to_disk_full(root, evil, config), FILE_SAVE_ERROR); + file_destroy(evil); + EXPECT_EQ_INT(lstat("../dir_entry_escape", &st), -1); + + config_delete(config); + rmdir("test_dir_entry_root/alpha/beta/gamma"); + rmdir("test_dir_entry_root/alpha/beta"); + rmdir("test_dir_entry_root/alpha"); + rmdir(root); +} + +/* ---- Phase 5 (--trust-sender) safety-floor tests ---- + * + * --trust-sender is a receiver-local policy that never crosses the wire: a real + * receiver enables it from its own process (the standalone server's --trust- + * sender CLI switch, which a client forwards as --remote-option=--trust-sender), + * so these tests force file_set_trust_sender(true) directly. Trust must RELAX + * only the redundant list-level re-validation (an escaping symlink TARGET is + * copied verbatim, rsync -l parity) and must NEVER disable the low-level + * fd-relative confinement floor: file_open_secure_parent's ".." rejection, the + * O_NOFOLLOW parent walk, leaf/destination confinement, and the ungated + * has_path_traversal on the link's own placement path in file_symlink_at_secure + * stay hard. A hostile sender therefore still cannot place a file, directory + * or symlink outside the receive root even with trust on. */ + +static void test_trust_sender_relaxes_symlink_target() { + const char* root = "test_trust_sender_root"; + const char* link = "test_trust_sender_root/escape_link"; + unlink(link); + rmdir(root); + EXPECT_EQ_INT(mkdir(root, 0755), 0); + + /* Control: without trust an absolute (escaping) target is refused and the + link is never placed. */ + file_set_trust_sender(false); + EXPECT_FALSE(file_symlink_at_secure(link, "/etc/passwd")); + struct stat st; + EXPECT_EQ_INT(lstat(link, &st), -1); + + /* Trust ON: the escaping target is copied verbatim (rsync -l parity) ... */ + file_set_trust_sender(true); + EXPECT_TRUE(file_symlink_at_secure(link, "/etc/passwd")); + EXPECT_EQ_INT(lstat(link, &st), 0); + EXPECT_TRUE(S_ISLNK(st.st_mode)); + /* ...but the link itself still lands beneath the receive root. */ + char target[128]; + ssize_t target_len = readlink(link, target, sizeof(target) - 1); + EXPECT_TRUE(target_len > 0); + // cppcheck-suppress knownConditionTrueFalse + if (target_len > 0) { + target[target_len] = '\0'; + EXPECT_EQ_STR(target, "/etc/passwd"); + } + unlink(link); + + /* Same relaxation through the real save funnel (file_save_to_disk_full). */ + Config* config = config_create(); + EXPECT_NOT_NULL(config); + const char* save_link = "test_trust_sender_root/save_link"; + unlink(save_link); + + File* sym = file_create("save_link"); + EXPECT_NOT_NULL(sym); + sym->is_symlink = true; + sym->symlink_target = str_dup("/etc/passwd"); + EXPECT_NOT_NULL(sym->symlink_target); + + file_set_trust_sender(false); + EXPECT_EQ_INT(file_save_to_disk_full(root, sym, config), FILE_SAVE_SKIPPED); + EXPECT_EQ_INT(lstat(save_link, &st), -1); + + file_set_trust_sender(true); + EXPECT_EQ_INT(file_save_to_disk_full(root, sym, config), FILE_SAVE_WRITTEN); + EXPECT_EQ_INT(lstat(save_link, &st), 0); + EXPECT_TRUE(S_ISLNK(st.st_mode)); + + file_destroy(sym); + config_delete(config); + unlink(save_link); + rmdir(root); +} + +static void test_trust_sender_confines_hostile_paths() { + const char* root = "test_trust_sender_root"; + const char* escaped_file = "../test_trust_sender_escaped_file.txt"; + const char* escaped_dir = "../test_trust_sender_escaped_dir"; + const char* escaped_link = "../test_trust_sender_escaped_link"; + unlink(escaped_file); + rmdir(escaped_dir); + unlink(escaped_link); + unlink(root); + rmdir(root); + EXPECT_EQ_INT(mkdir(root, 0755), 0); + + Config* config = config_create(); + EXPECT_NOT_NULL(config); + file_set_trust_sender(true); + struct stat st; + + /* A hostile regular-file path that would escape the root is contained: the + save-layer ".." re-check is relaxed under trust, so the attempt reaches the + secure floor, which refuses the walk -- nothing appears outside. */ + File* file = file_create(escaped_file); + EXPECT_NOT_NULL(file); + file->data->data = malloc(5); + EXPECT_NOT_NULL(file->data->data); + memcpy(file->data->data, "evil", 4); + file->data->size = 4; + EXPECT_EQ_INT(file_save_to_disk_full(root, file, config), FILE_SAVE_ERROR); + file_destroy(file); + EXPECT_EQ_INT(lstat(escaped_file, &st), -1); + + /* A hostile directory entry is contained the same way. */ + File* dir = file_create(escaped_dir); + EXPECT_NOT_NULL(dir); + dir->is_dir = true; + EXPECT_EQ_INT(file_save_to_disk_full(root, dir, config), FILE_SAVE_ERROR); + file_destroy(dir); + EXPECT_EQ_INT(lstat(escaped_dir, &st), -1); + + /* A hostile symlink whose OWN placement path escapes the root is refused even + under trust: the ungated has_path_traversal in file_symlink_at_secure never + turns off. */ + EXPECT_FALSE(file_symlink_at_secure("test_trust_sender_root/../escaped_link", "/etc/passwd")); + EXPECT_EQ_INT(lstat(escaped_link, &st), -1); + + /* file_open_secure_parent still refuses a ".." component outright. */ + char* leaf = NULL; + EXPECT_EQ_INT(file_open_secure_parent("test_trust_sender_root/../../etc/passwd", &leaf, true), + -1); + free(leaf); + + config_delete(config); + rmdir(root); +} + +/* The same guarantees under a configured authorized root: a within-root link + with an escaping target is created (relaxed), while a placement path that is + a clean absolute path OUTSIDE the authorized root (no ".." anywhere) is + refused by the leaf/destination confinement. */ +static void test_trust_sender_authorized_root_confinement() { + const char* root = "test_trust_sender_root"; + const char* sibling = "test_trust_sender_sibling"; + unlink(root); + rmdir(root); + rmdir(sibling); + EXPECT_EQ_INT(mkdir(root, 0755), 0); + EXPECT_EQ_INT(mkdir(sibling, 0755), 0); + + char root_abs[PATH_MAX]; + char sibling_abs[PATH_MAX]; + EXPECT_NOT_NULL(realpath(root, root_abs)); + EXPECT_NOT_NULL(realpath(sibling, sibling_abs)); + int root_fd = open(root_abs, O_RDONLY | O_DIRECTORY | O_CLOEXEC); + EXPECT_TRUE(root_fd >= 0); + // cppcheck-suppress knownConditionTrueFalse + if (root_fd < 0) { + rmdir(root); + rmdir(sibling); + return; + } + EXPECT_TRUE(file_set_authorized_root(root_fd, root_abs)); + + file_set_trust_sender(true); + struct stat st; + + /* Within the authorized root, an escaping symlink TARGET is copied verbatim. */ + char* inside_link = path_cat(root_abs, "authorized_escape_link"); + EXPECT_NOT_NULL(inside_link); + unlink(inside_link); + EXPECT_TRUE(file_symlink_at_secure(inside_link, "/etc/passwd")); + EXPECT_EQ_INT(lstat(inside_link, &st), 0); + EXPECT_TRUE(S_ISLNK(st.st_mode)); + unlink(inside_link); + + /* A clean absolute path in a sibling directory (outside the authorized root) + is still refused even under trust. */ + char* outside_link = path_cat(sibling_abs, "test_trust_sender_outside_link"); + EXPECT_NOT_NULL(outside_link); + unlink(outside_link); + EXPECT_FALSE(file_symlink_at_secure(outside_link, "/etc/passwd")); + EXPECT_EQ_INT(lstat(outside_link, &st), -1); + + free(outside_link); + free(inside_link); + file_set_authorized_root(-1, NULL); + close(root_fd); + unlink("test_trust_sender_outside_link"); + rmdir(sibling); + rmdir(root); +} + +void test_trust_sender() { + /* The final reset lines always run (a failing EXPECT only returns from the + helper), so a later group never inherits a stray trust/authorized-root + policy. */ + file_set_trust_sender(false); + test_trust_sender_relaxes_symlink_target(); + test_trust_sender_confines_hostile_paths(); + test_trust_sender_authorized_root_confinement(); + file_set_trust_sender(false); + file_set_authorized_root(-1, NULL); +} + +/* --sparse/-S hole preservation: a buffer with a long zero run written via + * file_store_write_secure(sparse=true) must round-trip its content exactly and + * have the right logical size, and should additionally be genuinely sparse on + * filesystems that support holes. The sparseness assertion is tolerant: if the + * filesystem reports no holes (SEEK_HOLE/SEEK_DATA -> ENXIO) we skip the strict + * block-count check, but content and size always hold. */ +static void test_file_write_to_disk_sparse_preserves_holes() { + const char* path = "test_sparse_file.bin"; + unlink(path); + /* 256 KiB with a 128 KiB zero run in the middle, bracketed by headers/tails. */ + const unsigned long long size = 256u * 1024u; + unsigned char* buf = malloc(size); + EXPECT_NOT_NULL(buf); + /* cppcheck-suppress knownConditionTrueFalse -- EXPECT_NOT_NULL above asserts, + but cppcheck cannot see through the macro; the guard is defensive. */ + if (!buf) + return; + memset(buf, 0, size); + for (unsigned long long i = 0; i < 4096; i++) { + buf[i] = (unsigned char)(i % 251); + buf[size - 1 - i] = (unsigned char)((i * 7) % 253); + } + + EXPECT_TRUE(file_store_write_secure(path, buf, size, false, true, NULL, false)); + + /* Logical size must equal data_size exactly. */ + struct stat st; + EXPECT_EQ_INT(stat(path, &st), 0); + EXPECT_EQ_INT((int)st.st_size, (int)size); + + /* Content must round-trip exactly: the full readback must equal the original + buffer byte-for-byte (header, the hole region staying zero, and tail) — + a writer bug in the lseek-offset bookkeeping would show up here. */ + int fd = open(path, O_RDONLY); + EXPECT_TRUE(fd >= 0); + /* cppcheck-suppress knownConditionTrueFalse -- EXPECT_TRUE above asserts, + but cppcheck cannot see through the macro; the guard is defensive. */ + if (fd >= 0) { + unsigned char* readback = malloc(size); + if (readback) { + unsigned long long got = 0; + while (got < size) { + ssize_t n = read(fd, readback + got, (size_t)(size - got)); + if (n <= 0) + break; + got += (unsigned long long)n; + } + EXPECT_EQ_INT((int)got, (int)size); + if (got == size) + EXPECT_EQ_INT(memcmp(readback, buf, size), 0); + free(readback); + } + /* Tolerant sparseness check: seek for holes; skip if unsupported. */ + off_t hole_off = lseek(fd, (off_t)4096, SEEK_HOLE); + if (hole_off >= 0 && hole_off < (off_t)size) { + off_t next_data = lseek(fd, hole_off, SEEK_DATA); + fstat(fd, &st); + int blocks = (int)(st.st_blocks * 512); + if (next_data > hole_off) + EXPECT_TRUE(blocks < (int)size); + } + close(fd); + } + free(buf); + unlink(path); +} + +/* --partial retention is hard to provoke end-to-end mid-transfer (the whole + * image is in one in-memory write), so this drives the failure path directly: + * a metadata whose mtime_nsec is out of the legal [0,999999999] range makes + * futimens (in file_restore_metadata_fd) fail with EINVAL AFTER the temp has + * been fully written. With keep_partial=true the written temp must be renamed + * to the destination path (a resumable partial); with keep_partial=false the + * same failure must leave NOTHING behind. The retention is always best-effort + * (never a corrupt blend), and this asserts the both-on/off behavior. */ +static void test_file_write_to_disk_partial_retention() { + const char* path = "test_partial_retention.bin"; + unlink(path); + const char content[] = "partial-retention payload"; + FileMetadata m; + memset(&m, 0, sizeof(m)); + m.mode = 0644; + m.uid = (uid_t)geteuid(); + m.gid = (gid_t)getegid(); + m.mtime_sec = 1700000000; + m.mtime_nsec = 2000000000; /* invalid: forces futimens EINVAL after the write */ + m.atime_valid = false; + m.crtime_valid = false; + bool ok = file_to_disk_secure_attrs(path, content, strlen(content), false, false, true, &m, false, + false, false, false, NULL, false, true, NULL); + EXPECT_FALSE(ok); /* the write itself succeeded, but metadata restore failed */ + /* Retained: the already-written temp now sits at the destination path. */ + int fd = open(path, O_RDONLY); + EXPECT_TRUE(fd >= 0); + /* cppcheck-suppress knownConditionTrueFalse -- EXPECT_TRUE above asserts, + but cppcheck cannot see through the macro; the guard is defensive. */ + if (fd >= 0) { + char buf[64]; + ssize_t n = read(fd, buf, sizeof(buf)); + close(fd); + EXPECT_EQ_INT((int)strlen(content), (int)n); + if (n == (ssize_t)strlen(content)) + EXPECT_TRUE(memcmp(buf, content, strlen(content)) == 0); + } + unlink(path); + + /* Same failure with keep_partial=false: temp is unlinked, nothing retained. */ + ok = file_to_disk_secure_attrs(path, content, strlen(content), false, false, true, &m, false, + false, false, false, NULL, false, false, NULL); + EXPECT_FALSE(ok); + EXPECT_TRUE(access(path, F_OK) == -1); +} + +/* P7 Wave D: the deferred directory-time list deep-copies entries and applies + * them (fd-relative, no-follow) to an existing directory, then frees cleanly. */ +static void test_dir_time_list() { + const char* root = "test_dir_time_root"; + const char* sub = "test_dir_time_root/sub"; + file_set_authorized_root(-1, NULL); + rmdir(sub); + rmdir(root); + EXPECT_EQ_INT(mkdir(root, 0755), 0); + EXPECT_EQ_INT(mkdir(sub, 0755), 0); + + DirTimeList list; + dir_time_list_init(&list); + EXPECT_EQ_INT((int)list.count, 0); + FileMetadata metadata = {.mtime_sec = 1000000000, .mtime_nsec = 0}; + EXPECT_TRUE(dir_time_list_add(&list, "sub", &metadata)); + EXPECT_TRUE(dir_time_list_add(&list, "sub", &metadata)); + EXPECT_EQ_INT((int)list.count, 2); + + dir_time_list_apply(&list, root); + struct stat st; + EXPECT_EQ_INT(stat(sub, &st), 0); + EXPECT_EQ_INT((int)st.st_mtime, 1000000000); + + dir_time_list_free(&list); + EXPECT_EQ_INT((int)list.count, 0); + EXPECT_NULL(list.paths); + EXPECT_NULL(list.entries); + + rmdir(sub); + rmdir(root); +} + +/* -K/--keep-dirlinks secure open: with an authorized root, a destination path + * component that is a symlink to an IN-ROOT directory is used as that directory + * (its referent is opened through a relative O_NOFOLLOW walk from the root fd, + * not by re-opening an absolute realpath() result), while a symlink resolving + * OUTSIDE the root is rejected. With -K off, even the in-root link is not + * followed. */ +static void test_keep_dirlinks_secure_open_impl() { + const char* root = "test_keep_dirlinks_root"; + const char* real = "test_keep_dirlinks_root/realdir"; + const char* link = "test_keep_dirlinks_root/linkdir"; + const char* abslink = "test_keep_dirlinks_root/abslink"; + const char* escape = "test_keep_dirlinks_root/escape"; + const char* outside = "test_keep_dirlinks_outside"; + unlink(link); + unlink(abslink); + unlink(escape); + rmdir(real); + rmdir(root); + rmdir(outside); + EXPECT_EQ_INT(mkdir(root, 0755), 0); + EXPECT_EQ_INT(mkdir(real, 0755), 0); + EXPECT_EQ_INT(mkdir(outside, 0755), 0); + + char root_abs[PATH_MAX]; + char real_abs[PATH_MAX]; + char outside_abs[PATH_MAX]; + EXPECT_NOT_NULL(realpath(root, root_abs)); + EXPECT_NOT_NULL(realpath(real, real_abs)); + EXPECT_NOT_NULL(realpath(outside, outside_abs)); + EXPECT_EQ_INT(symlink("realdir", link), 0); /* relative, in-root */ + EXPECT_EQ_INT(symlink(real_abs, abslink), 0); /* absolute, in-root */ + /* cppcheck-suppress knownConditionTrueFalse */ + EXPECT_EQ_INT(symlink(outside_abs, escape), 0); /* absolute, outside root */ + + int root_fd = open(root_abs, O_RDONLY | O_DIRECTORY | O_CLOEXEC); + EXPECT_TRUE(root_fd >= 0); + // cppcheck-suppress knownConditionTrueFalse + if (root_fd < 0) { + unlink(link); + unlink(abslink); + unlink(escape); + rmdir(real); + rmdir(root); + rmdir(outside); + return; + } + EXPECT_TRUE(file_set_authorized_root(root_fd, root_abs)); + file_set_keep_dirlinks(true); + + struct stat real_st; + EXPECT_EQ_INT(fstatat(root_fd, "realdir", &real_st, 0), 0); + + /* Relative in-root symlink-to-directory: followed to the referent dir. */ + char path[PATH_MAX + 64]; + snprintf(path, sizeof(path), "%s/linkdir/file.txt", root_abs); + char* leaf = NULL; + int parent_fd = file_open_secure_parent(path, &leaf, false); + EXPECT_TRUE(parent_fd >= 0); + EXPECT_NOT_NULL(leaf); + // cppcheck-suppress knownConditionTrueFalse + if (leaf) + EXPECT_EQ_STR(leaf, "file.txt"); + // cppcheck-suppress knownConditionTrueFalse + if (parent_fd >= 0) { + struct stat st; + EXPECT_EQ_INT(fstat(parent_fd, &st), 0); + EXPECT_TRUE(st.st_dev == real_st.st_dev && st.st_ino == real_st.st_ino); + close(parent_fd); + } + free(leaf); + + /* Absolute-but-in-root symlink-to-directory is followed the same way. */ + snprintf(path, sizeof(path), "%s/abslink/file.txt", root_abs); + leaf = NULL; + parent_fd = file_open_secure_parent(path, &leaf, false); + EXPECT_TRUE(parent_fd >= 0); + // cppcheck-suppress knownConditionTrueFalse + if (parent_fd >= 0) { + struct stat st; + EXPECT_EQ_INT(fstat(parent_fd, &st), 0); + EXPECT_TRUE(st.st_dev == real_st.st_dev && st.st_ino == real_st.st_ino); + close(parent_fd); + } + free(leaf); + + /* A symlink resolving outside the authorized root is rejected. */ + snprintf(path, sizeof(path), "%s/escape/file.txt", root_abs); + leaf = NULL; + EXPECT_EQ_INT(file_open_secure_parent(path, &leaf, false), -1); + free(leaf); + + /* With -K off the in-root symlink is not followed either. */ + file_set_keep_dirlinks(false); + snprintf(path, sizeof(path), "%s/linkdir/file.txt", root_abs); + leaf = NULL; + EXPECT_EQ_INT(file_open_secure_parent(path, &leaf, false), -1); + free(leaf); + + file_set_keep_dirlinks(false); + file_set_authorized_root(-1, NULL); + close(root_fd); + unlink(link); + unlink(abslink); + unlink(escape); + rmdir(real); + rmdir(root); + rmdir(outside); +} + +/* Wrapper guarantees the process-wide keep-dirlinks/authorized-root policy is + * cleared even when an EXPECT inside the body returns early (a failing EXPECT + * returns from its own function, so the body's trailing resets may be skipped). */ +static void test_keep_dirlinks_secure_open() { + file_set_authorized_root(-1, NULL); + file_set_keep_dirlinks(false); + test_keep_dirlinks_secure_open_impl(); + file_set_authorized_root(-1, NULL); + file_set_keep_dirlinks(false); +} + void test_file() { test_file_create(); + test_file_special_rdev_valid(); test_file_destroy_null(); test_file_destroy_normal(); test_file_load_data(); test_file_load_data_missing_file(); test_file_save_to_disk(); - test_to_disk_basic(); - test_to_disk_creates_dirs(); + test_file_save_to_disk_with_fsync_config(); + test_file_save_to_disk_existing(); + test_file_save_to_disk_ignore_existing(); + test_file_save_to_disk_ignore_existing_entry_types(); + test_file_save_to_disk_partial_install(); + test_file_save_to_disk_reports_skips(); + test_file_write_to_disk_sparse_preserves_holes(); + test_file_write_to_disk_partial_retention(); + test_file_write_to_disk_basic(); + test_file_write_to_disk_with_fsync(); + test_file_write_to_disk_preallocate_atomic(); + test_file_write_to_disk_preallocate_inplace(); + test_file_write_to_disk_creates_dirs(); + test_file_write_to_disk_does_not_follow_symlink(); test_file_content_to_buffer(); + test_file_symlink_helpers(); + test_file_symlink_at_secure(); test_file_save_to_disk_path_traversal(); test_file_save_to_disk_deep_traversal(); + test_dir_entry_save_to_disk(); if (!is_running_under_valgrind()) { // Fork tests are skipped under valgrind because the parent process runs // orders of magnitude slower than the child (parent is instrumented, child @@ -449,4 +1530,9 @@ void test_file() { test_file_send_single_calls_metadata_and_path(); } test_file_metadata_create(); + test_dir_time_list(); + test_keep_dirlinks_secure_open(); + test_inplace_overwrite_clears_special_mode_bits(); + test_inplace_overwrite_metadata_strips_special_bits(); + test_inplace_overwrite_truncates_shorter_payload(); } diff --git a/tests/test_file.h b/tests/test_file.h index 1b07e48..7d6fafc 100644 --- a/tests/test_file.h +++ b/tests/test_file.h @@ -2,5 +2,6 @@ #define TEST_FILE_H void test_file(); +void test_trust_sender(); #endif diff --git a/tests/test_file_sendfile.c b/tests/test_file_sendfile.c index 430ed8e..61663e3 100644 --- a/tests/test_file_sendfile.c +++ b/tests/test_file_sendfile.c @@ -15,7 +15,7 @@ static void test_sendfile_basic() { const char* content = "Hello from sendfile test!"; size_t len = strlen(content); - EXPECT_TRUE(to_disk("test_sendfile_basic.txt", content, len)); + EXPECT_TRUE(file_write_to_disk("test_sendfile_basic.txt", content, len, false, false)); File* file = file_create("test_sendfile_basic.txt"); EXPECT_NOT_NULL(file); @@ -77,7 +77,7 @@ static void test_sendfile_basic() { static void test_sendfile_empty_file() { const char* content = ""; size_t len = 0; - EXPECT_TRUE(to_disk("test_sendfile_empty.txt", content, len)); + EXPECT_TRUE(file_write_to_disk("test_sendfile_empty.txt", content, len, false, false)); File* file = file_create("test_sendfile_empty.txt"); EXPECT_NOT_NULL(file); @@ -156,7 +156,7 @@ static void test_sendfile_missing_file() { static void test_sendfile_compression_fallback() { const char* content = "Compression fallback content"; size_t len = strlen(content); - EXPECT_TRUE(to_disk("test_sendfile_comp.txt", content, len)); + EXPECT_TRUE(file_write_to_disk("test_sendfile_comp.txt", content, len, false, false)); struct stat st; EXPECT_EQ_INT(stat("test_sendfile_comp.txt", &st), 0); @@ -222,7 +222,7 @@ static void test_sendfile_compression_fallback() { static void test_sendfile_no_path() { const char* content = "No path sendfile test"; size_t len = strlen(content); - EXPECT_TRUE(to_disk("test_sendfile_nopath.txt", content, len)); + EXPECT_TRUE(file_write_to_disk("test_sendfile_nopath.txt", content, len, false, false)); File* file = file_create("test_sendfile_nopath.txt"); EXPECT_NOT_NULL(file); diff --git a/tests/test_fuzz_smoke.c b/tests/test_fuzz_smoke.c index 94280f7..50d6c29 100644 --- a/tests/test_fuzz_smoke.c +++ b/tests/test_fuzz_smoke.c @@ -1,16 +1,24 @@ #include "test_fuzz_smoke.h" #include "chunk.h" #include "compression.h" +#include "config.h" #include "data.h" #include "delta.h" #include "metadata.h" +#include "protocol.h" #include "test_utils.h" #include "utils.h" +#include +#include #include #include #include +#include #include +/* P8 config-frame tail: super_mode (4) + copy-as presence (4) + uid (4) + gid (4). */ +#define P8_TAIL_BYTES 16 + /* Smoke test for chunk_deserialize fuzz target */ static void test_fuzz_chunk_deserialize() { /* Create a minimal valid chunk to serialize and deserialize */ @@ -101,12 +109,12 @@ static void test_fuzz_delta_deserialize() { /* Smoke test for metadata_from_buf fuzz target */ static void test_fuzz_metadata_from_buf() { /* Create a real file to get metadata from */ - EXPECT_TRUE(to_disk("fuzz_meta_test.txt", "metadata test", 13)); + EXPECT_TRUE(file_write_to_disk("fuzz_meta_test.txt", "metadata test", 13, false, false)); struct stat st; EXPECT_EQ_INT(stat("fuzz_meta_test.txt", &st), 0); - FileMetadata* meta = file_metadata_create(&st); + FileMetadata* meta = file_metadata_create("fuzz_meta_test.txt", &st, false, false); EXPECT_NOT_NULL(meta); EXPECT_EQ_INT((int)meta->mode, (int)st.st_mode); EXPECT_EQ_INT((int)meta->mtime_sec, (int)st.st_mtime); @@ -170,6 +178,272 @@ static void test_fuzz_glob_match() { EXPECT_FALSE(glob_match("*.md", "readme.txt")); } +/* ---- Deterministic config-frame receive hardening (P8) ---- + * + * The P8 tail (--super / --copy-as) and the identity-map count only parse after + * the entire preceding frame validates, which random bytes almost never reach. + * These tests capture one valid frame with the production sender and then + * mutate/truncate the exact tail bytes. */ + +/* Serialize cfg with the production sender into a heap buffer. The frame is + * written into a pipe (64 KiB kernel buffer, far larger than one config frame) + * whose read end is drained afterwards; the required STATUS_OK ack is + * pre-loaded into a second pipe, so a single thread suffices. */ +static bool capture_config_frame(const Config* cfg, unsigned char** out, size_t* out_len) { + *out = NULL; + *out_len = 0; + + int frame_pipe[2]; + int status_pipe[2]; + if (pipe(frame_pipe) != 0) + return false; + if (pipe(status_pipe) != 0) { + close(frame_pipe[0]); + close(frame_pipe[1]); + return false; + } + + int ack = STATUS_OK; + bool ok = write(status_pipe[1], &ack, sizeof(ack)) == (ssize_t)sizeof(ack); + if (ok) { + io_set_fds(status_pipe[0], frame_pipe[1]); + io_set_bwlimit(0); + ok = config_send(frame_pipe[1], cfg); + } + close(frame_pipe[1]); + close(status_pipe[0]); + close(status_pipe[1]); + + unsigned char* buf = NULL; + if (ok) { + size_t cap = 4096; + size_t len = 0; + buf = malloc(cap); + if (!buf) { + ok = false; + } + while (ok) { + if (len == cap) { + size_t grown = cap * 2; + unsigned char* bigger = realloc(buf, grown); + if (!bigger) { + ok = false; + break; + } + buf = bigger; + cap = grown; + } + ssize_t n = read(frame_pipe[0], buf + len, cap - len); + if (n > 0) { + len += (size_t)n; + continue; + } + if (n < 0 && errno == EINTR) + continue; + break; + } + if (ok && len > 0) { + *out = buf; + *out_len = len; + buf = NULL; + } + } + close(frame_pipe[0]); + free(buf); + return *out != NULL; +} + +/* Feed a raw config frame to config_receive over a socketpair. The write half + * is shut down (not closed) after the data so the receiver sees EOF but its + * STATUS_ERROR replies do not hit a closed peer. */ +static bool receive_config_frame(const unsigned char* buf, size_t len) { + int sv[2]; + if (socketpair(AF_UNIX, SOCK_STREAM, 0, sv) != 0) + return false; + + size_t off = 0; + while (off < len) { + ssize_t n = write(sv[0], buf + off, len - off); + if (n > 0) { + off += (size_t)n; + continue; + } + if (n < 0 && errno == EINTR) + continue; + break; + } + shutdown(sv[0], SHUT_WR); + io_set_fds(sv[1], sv[1]); + io_set_bwlimit(0); + Config* cfg = config_receive(sv[1]); + bool accepted = cfg != NULL; + config_delete(cfg); + close(sv[0]); + close(sv[1]); + return accepted; +} + +static void put_i32(unsigned char* buf, size_t off, int32_t value) { + memcpy(buf + off, &value, sizeof(value)); +} + +static size_t find_bytes(const unsigned char* haystack, size_t haystack_len, + const unsigned char* needle, size_t needle_len) { + if (needle_len == 0 || haystack_len < needle_len) + return SIZE_MAX; + for (size_t i = 0; i + needle_len <= haystack_len; i++) { + if (memcmp(haystack + i, needle, needle_len) == 0) + return i; + } + return SIZE_MAX; +} + +static Config* make_copy_as_config(void) { + Config* c = config_create(); + if (!c) + return NULL; + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->copy_as_set = true; + c->copy_as_uid = 0; + c->copy_as_gid = 0; + c->use_metadata = true; /* --copy-as requires the metadata path */ + return c; +} + +/* The P8 tail must reject an out-of-range super_mode, a negative copy-as id and + * any truncation inside the tail, while the untouched frame is accepted. */ +static void test_fuzz_config_receive_p8_tail() { + Config* c = make_copy_as_config(); + EXPECT_NOT_NULL(c); + + unsigned char* frame = NULL; + size_t len = 0; + bool captured = capture_config_frame(c, &frame, &len); + config_delete(c); + if (!captured || len <= P8_TAIL_BYTES) { + free(frame); + EXPECT_TRUE(false); + return; + } + + /* Baseline: the untouched frame is accepted. */ + EXPECT_TRUE(receive_config_frame(frame, len)); + + unsigned char* mut = malloc(len); + EXPECT_NOT_NULL(mut); + + /* super_mode outside the 0..2 tri-state is refused. */ + memcpy(mut, frame, len); + put_i32(mut, len - P8_TAIL_BYTES, 99); + EXPECT_FALSE(receive_config_frame(mut, len)); + put_i32(mut, len - P8_TAIL_BYTES, -1); + EXPECT_FALSE(receive_config_frame(mut, len)); + + /* A negative (sentinel) and an extreme copy-as uid/gid are refused. */ + memcpy(mut, frame, len); + put_i32(mut, len - P8_TAIL_BYTES, SUPER_MODE_AUTO); + put_i32(mut, len - P8_TAIL_BYTES + 4, 1); + put_i32(mut, len - P8_TAIL_BYTES + 8, -1); + put_i32(mut, len - P8_TAIL_BYTES + 12, 0); + EXPECT_FALSE(receive_config_frame(mut, len)); + put_i32(mut, len - P8_TAIL_BYTES + 8, 0); + put_i32(mut, len - P8_TAIL_BYTES + 12, INT32_MIN); + EXPECT_FALSE(receive_config_frame(mut, len)); + + /* A presence int that is not a wire bool is refused. */ + memcpy(mut, frame, len); + put_i32(mut, len - P8_TAIL_BYTES, SUPER_MODE_AUTO); + put_i32(mut, len - P8_TAIL_BYTES + 4, 2); + EXPECT_FALSE(receive_config_frame(mut, len)); + + /* Truncating anywhere inside the P8 tail is refused. */ + EXPECT_FALSE(receive_config_frame(frame, len - 2)); + EXPECT_FALSE(receive_config_frame(frame, len - P8_TAIL_BYTES)); + + free(mut); + free(frame); +} + +/* A huge or negative --usermap count must be refused up front, never driving a + * giant allocation. The count is located by searching for a sentinel entry. */ +static void test_fuzz_config_receive_huge_map_count() { + Config* c = make_copy_as_config(); + EXPECT_NOT_NULL(c); + int32_t sentinel_from = 0x11223344; + int32_t sentinel_to = 0x55667788; + c->usermap = malloc(sizeof(IdentityMap)); + if (!c->usermap) { + config_delete(c); + EXPECT_TRUE(false); + return; + } + c->usermap_count = 1; + c->usermap[0].from = sentinel_from; + c->usermap[0].to = sentinel_to; + + unsigned char* frame = NULL; + size_t len = 0; + bool captured = capture_config_frame(c, &frame, &len); + config_delete(c); + if (!captured) { + EXPECT_TRUE(false); + return; + } + + unsigned char pattern[8]; + memcpy(pattern, &sentinel_from, sizeof(sentinel_from)); + memcpy(pattern + sizeof(sentinel_from), &sentinel_to, sizeof(sentinel_to)); + size_t entry_off = find_bytes(frame, len, pattern, sizeof(pattern)); + if (entry_off == SIZE_MAX || entry_off < sizeof(int32_t)) { + free(frame); + EXPECT_TRUE(false); + return; + } + size_t count_off = entry_off - sizeof(int32_t); + + /* Baseline accepted. */ + EXPECT_TRUE(receive_config_frame(frame, len)); + + unsigned char* mut = malloc(len); + EXPECT_NOT_NULL(mut); + memcpy(mut, frame, len); + put_i32(mut, count_off, INT_MAX); + EXPECT_FALSE(receive_config_frame(mut, len)); + put_i32(mut, count_off, -1); + EXPECT_FALSE(receive_config_frame(mut, len)); + put_i32(mut, count_off, MAX_IDENTITY_MAP + 1); + EXPECT_FALSE(receive_config_frame(mut, len)); + + free(mut); + free(frame); +} + +/* A mismatched version and a matching version followed by a wrong-order field + * (an int that is not a wire bool) are both refused at/just after the gate. */ +static void test_fuzz_config_receive_version_gate() { + unsigned char buf[64]; + + size_t off = 0; + const char* bad_version = "1.2.3"; + size_t bad_len = strlen(bad_version); + memcpy(buf + off, &bad_len, sizeof(bad_len)); + off += sizeof(bad_len); + memcpy(buf + off, bad_version, bad_len); + off += bad_len; + EXPECT_FALSE(receive_config_frame(buf, off)); + + off = 0; + size_t good_len = strlen(PROTOCOL_VERSION); + memcpy(buf + off, &good_len, sizeof(good_len)); + off += sizeof(good_len); + memcpy(buf + off, PROTOCOL_VERSION, good_len); + off += good_len; + put_i32(buf, off, -1); + off += sizeof(int32_t); + EXPECT_FALSE(receive_config_frame(buf, off)); +} + void test_fuzz_smoke() { test_fuzz_chunk_deserialize(); test_fuzz_compress_decompress(); @@ -177,4 +451,7 @@ void test_fuzz_smoke() { test_fuzz_metadata_from_buf(); test_fuzz_delta_signature_deserialize(); test_fuzz_glob_match(); + test_fuzz_config_receive_p8_tail(); + test_fuzz_config_receive_huge_map_count(); + test_fuzz_config_receive_version_gate(); } diff --git a/tests/test_iconv.c b/tests/test_iconv.c new file mode 100644 index 0000000..faee11b --- /dev/null +++ b/tests/test_iconv.c @@ -0,0 +1,217 @@ +#include "test_iconv.h" +#include "charset.h" +#include "protocol.h" +#include "test_utils.h" +#include "utils.h" +#include +#include +#include +#include +#include +#include + +/* --- CONVERT_SPEC parsing ------------------------------------------------ */ + +static void test_iconv_spec_parse_split() { + char* local = NULL; + char* remote = NULL; + EXPECT_EQ_INT(charset_spec_parse("utf-8,iso-8859-1", &local, &remote), 0); + EXPECT_EQ_STR(local, "utf-8"); + EXPECT_EQ_STR(remote, "iso-8859-1"); + free(local); + free(remote); +} + +static void test_iconv_spec_parse_single_defaults_to_local() { + char* local = NULL; + char* remote = NULL; + EXPECT_EQ_INT(charset_spec_parse("utf-8", &local, &remote), 0); + EXPECT_EQ_STR(local, "utf-8"); + EXPECT_EQ_STR(remote, "utf-8"); + free(local); + free(remote); +} + +static void test_iconv_spec_parse_garbage() { + char* local = NULL; + char* remote = NULL; + EXPECT_EQ_INT(charset_spec_parse(NULL, &local, &remote), -1); + EXPECT_EQ_INT(charset_spec_parse("", &local, &remote), -1); + EXPECT_EQ_INT(charset_spec_parse(",", &local, &remote), -1); + EXPECT_EQ_INT(charset_spec_parse("utf-8,", &local, &remote), -1); + EXPECT_EQ_INT(charset_spec_parse(",utf-8", &local, &remote), -1); +} + +static void test_iconv_spec_valid() { + EXPECT_TRUE(charset_spec_valid(NULL)); + EXPECT_TRUE(charset_spec_valid("utf-8")); + EXPECT_TRUE(charset_spec_valid("utf-8,iso-8859-1")); + EXPECT_TRUE(charset_spec_valid("iso-8859-1,ascii")); + EXPECT_FALSE(charset_spec_valid("no-such-charset,utf-8")); + EXPECT_FALSE(charset_spec_valid("utf-8,no-such-charset")); + EXPECT_FALSE(charset_spec_valid(",,,")); + EXPECT_FALSE(charset_spec_valid("utf-8,")); + /* A target charset whose conversion emits embedded NUL bytes would be + truncated by the C-string wire helpers; it must be rejected up front. */ + EXPECT_FALSE(charset_spec_valid("utf-8,utf-16")); + EXPECT_FALSE(charset_spec_valid("utf-16")); + EXPECT_FALSE(charset_spec_valid("iso-8859-1,utf-16")); +} + +/* --- one-shot conversion ------------------------------------------------ */ + +static void test_iconv_utf8_to_latin1() { + void* conv = charset_conversion_open("utf-8", "iso-8859-1"); + EXPECT_NOT_NULL(conv); + char* out = charset_convert(conv, "caf\xc3\xa9", NULL); + EXPECT_NOT_NULL(out); + EXPECT_EQ_INT(strcmp(out, "caf\xe9"), 0); + free(out); + charset_conversion_close(conv); +} + +static void test_iconv_latin1_to_utf8() { + void* conv = charset_conversion_open("iso-8859-1", "utf-8"); + EXPECT_NOT_NULL(conv); + char* out = charset_convert(conv, "caf\xe9", NULL); + EXPECT_NOT_NULL(out); + EXPECT_EQ_INT(strcmp(out, "caf\xc3\xa9"), 0); + free(out); + charset_conversion_close(conv); +} + +static void test_iconv_invalid_sequence_fails() { + int err = 0; + /* 0xff is not a valid UTF-8 sequence. */ + void* conv = charset_conversion_open("utf-8", "ascii"); + EXPECT_NOT_NULL(conv); + EXPECT_TRUE(charset_convert(conv, "bad\xff", &err) == NULL); + EXPECT_TRUE(err == EILSEQ || err == EINVAL); + charset_conversion_close(conv); +} + +static void test_iconv_unrepresentable_fails() { + /* "caf\xc3\xa9" (UTF-8 for cafe) has no ASCII representation. */ + void* conv = charset_conversion_open("utf-8", "ascii"); + EXPECT_NOT_NULL(conv); + EXPECT_TRUE(charset_convert(conv, "caf\xc3\xa9", NULL) == NULL); + charset_conversion_close(conv); +} + +/* A latin1 high-bit byte expands to two UTF-8 bytes. With exactly 16 high + * bytes the output is exactly cap = in_len + 16, so the final iconv call fills + * the buffer completely and a naive NUL-terminator write would overflow. */ +static void test_iconv_exact_fill_no_overflow() { + char name[64]; + strcpy(name, "dir/"); + int n = 4; + for (int i = 0; i < 16; i++) + name[n++] = (char)(0x80 + i); + name[n] = '\0'; + + void* conv = charset_conversion_open("iso-8859-1", "utf-8"); + EXPECT_NOT_NULL(conv); + char* out = charset_convert(conv, name, NULL); + EXPECT_NOT_NULL(out); + EXPECT_EQ_INT((int)strlen(out), n + 16); + charset_conversion_close(conv); + free(out); +} + +/* Many high-bit bytes force the output buffer past its initial cap, exercising + * the E2BIG growth path (input partially consumed/produced before the grow). */ +static void test_iconv_growth_expanding_name() { + char name[256]; + strcpy(name, "dir/"); + int n = 4; + for (int i = 0; i < 80; i++) + name[n++] = (char)(0x80 + (i % 0x80)); + name[n] = '\0'; + + void* conv = charset_conversion_open("iso-8859-1", "utf-8"); + EXPECT_NOT_NULL(conv); + char* out = charset_convert(conv, name, NULL); + EXPECT_NOT_NULL(out); + EXPECT_EQ_INT((int)strlen(out), n + 80); + charset_conversion_close(conv); + free(out); +} + +/* --- process-wide wire conversion ---------------------------------------- */ + +static void test_iconv_wire_sender_converts_local_to_remote() { + EXPECT_TRUE(charset_wire_init_sender("utf-8,iso-8859-1")); + char* wire = charset_wire_apply("caf\xc3\xa9"); + EXPECT_NOT_NULL(wire); + EXPECT_EQ_INT(strcmp(wire, "caf\xe9"), 0); + free(wire); + charset_wire_free(); +} + +static void test_iconv_wire_receiver_converts_remote_to_local() { + EXPECT_TRUE(charset_wire_init_receiver("utf-8,iso-8859-1", NULL)); + char* local = charset_wire_apply("caf\xe9"); + EXPECT_NOT_NULL(local); + EXPECT_EQ_INT(strcmp(local, "caf\xc3\xa9"), 0); + free(local); + charset_wire_free(); +} + +static void test_iconv_wire_disabled_passthrough() { + charset_wire_init_sender(NULL); + EXPECT_FALSE(charset_wire_active()); + char* out = charset_wire_apply("plain/name\xff"); + EXPECT_NOT_NULL(out); + EXPECT_EQ_INT(strcmp(out, "plain/name\xff"), 0); + free(out); + charset_wire_free(); +} + +static void test_iconv_wire_str_roundtrip() { + EXPECT_TRUE(charset_wire_init_sender("utf-8,iso-8859-1")); + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + charset_wire_free(); + charset_wire_init_receiver("utf-8,iso-8859-1", NULL); + char* got = receive_wire_str(p[0]); + bool ok = got != NULL && strcmp(got, "caf\xc3\xa9") == 0; + free(got); + charset_wire_free(); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + bool sent = send_wire_str(p[1], "caf\xc3\xa9"); + int status; + waitpid(pid, &status, 0); + close(p[1]); + charset_wire_free(); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +void test_iconv() { + test_iconv_spec_parse_split(); + test_iconv_spec_parse_single_defaults_to_local(); + test_iconv_spec_parse_garbage(); + test_iconv_spec_valid(); + test_iconv_utf8_to_latin1(); + test_iconv_latin1_to_utf8(); + test_iconv_invalid_sequence_fails(); + test_iconv_unrepresentable_fails(); + test_iconv_exact_fill_no_overflow(); + test_iconv_growth_expanding_name(); + test_iconv_wire_sender_converts_local_to_remote(); + test_iconv_wire_receiver_converts_remote_to_local(); + test_iconv_wire_disabled_passthrough(); + test_iconv_wire_str_roundtrip(); +} \ No newline at end of file diff --git a/tests/test_iconv.h b/tests/test_iconv.h new file mode 100644 index 0000000..8d54ec6 --- /dev/null +++ b/tests/test_iconv.h @@ -0,0 +1,6 @@ +#ifndef TEST_ICONV_H +#define TEST_ICONV_H + +void test_iconv(void); + +#endif \ No newline at end of file diff --git a/tests/test_log.c b/tests/test_log.c index da828dc..f9caf19 100644 --- a/tests/test_log.c +++ b/tests/test_log.c @@ -1,6 +1,8 @@ #include "test_log.h" #include "log.h" #include "test_utils.h" +#include +#include /* Test default log level: WARNING and ERROR should print, DEBUG and INFO should not. * We can't easily capture stderr in unit tests, so we verify the functions don't crash @@ -90,6 +92,29 @@ static void test_log_filtering() { EXPECT_TRUE(true); } +static void test_log_stderr_mode_all() { + int pipe_fds[2]; + EXPECT_EQ_INT(pipe(pipe_fds), 0); + int saved_stderr = dup(STDERR_FILENO); + EXPECT_TRUE(saved_stderr >= 0); + EXPECT_TRUE(dup2(pipe_fds[1], STDERR_FILENO) >= 0); + close(pipe_fds[1]); + + set_log_level(LOG_LEVEL_WARNING); + log_set_stderr_mode(LOG_STDERR_ALL); + log_message(LOG_LEVEL_WARNING, "warning routed to stderr"); + fflush(stderr); + + EXPECT_TRUE(dup2(saved_stderr, STDERR_FILENO) >= 0); + close(saved_stderr); + char output[128] = {0}; + ssize_t length = read(pipe_fds[0], output, sizeof(output) - 1); + close(pipe_fds[0]); + EXPECT_TRUE(length > 0); + EXPECT_TRUE(strstr(output, "warning routed to stderr") != NULL); + log_set_stderr_mode(LOG_STDERR_ERRORS); +} + /* Test that log_message handles various format strings */ static void test_log_message_formats() { set_log_level(LOG_LEVEL_DEBUG); @@ -103,6 +128,25 @@ static void test_log_message_formats() { EXPECT_TRUE(true); } +/* log_debug_enabled is the lazy-formatting gate for log_debug_message: it must + * be true only at DEBUG level with the requested flag selected, exactly + * mirroring the filter inside log_debug_message itself. */ +static void test_log_debug_enabled_matches_gate() { + set_log_level(LOG_LEVEL_WARNING); + set_log_debug_flags(LOG_DEBUG_ALL); + EXPECT_FALSE(log_debug_enabled(LOG_DEBUG_PROTO)); + + set_log_level(LOG_LEVEL_DEBUG); + set_log_debug_flags(LOG_DEBUG_PROTO); + EXPECT_TRUE(log_debug_enabled(LOG_DEBUG_PROTO)); + EXPECT_FALSE(log_debug_enabled(LOG_DEBUG_IO)); + + set_log_debug_flags(0); + EXPECT_FALSE(log_debug_enabled(LOG_DEBUG_PROTO)); + + set_log_debug_flags(LOG_DEBUG_ALL); +} + void test_log() { test_log_message_debug(); test_log_message_info(); @@ -112,5 +156,7 @@ void test_log() { test_log_set_level_info(); test_log_set_level_error(); test_log_filtering(); + test_log_stderr_mode_all(); test_log_message_formats(); + test_log_debug_enabled_matches_gate(); } diff --git a/tests/test_metadata.c b/tests/test_metadata.c index 3a79a21..74a5f27 100644 --- a/tests/test_metadata.c +++ b/tests/test_metadata.c @@ -1,4 +1,5 @@ #include "test_metadata.h" +#include "chmod.h" #include "metadata.h" #include "protocol.h" #include "test_utils.h" @@ -14,6 +15,12 @@ static void test_metadata_to_from_buf_roundtrip() { original.gid = 1000; original.mtime_sec = 1234567890; original.mtime_nsec = 500000000; + original.atime_valid = true; + original.atime_sec = 1234567000; + original.atime_nsec = 250000000; + original.crtime_valid = true; + original.crtime_sec = 1200000000; + original.crtime_nsec = 750000000; char* buf = malloc(FILE_METADATA_WIRE_SIZE + sizeof(int)); EXPECT_NOT_NULL(buf); @@ -29,6 +36,14 @@ static void test_metadata_to_from_buf_roundtrip() { EXPECT_EQ_INT(result->gid, 1000); EXPECT_EQ_INT(result->mtime_sec, 1234567890); EXPECT_EQ_INT(result->mtime_nsec, 500000000); + EXPECT_TRUE(result->atime_valid); + EXPECT_EQ_INT(result->atime_sec, 1234567000); + EXPECT_EQ_INT(result->atime_nsec, 250000000); + EXPECT_TRUE(result->crtime_valid); + EXPECT_EQ_INT(result->crtime_sec, 1200000000); + EXPECT_EQ_INT(result->crtime_nsec, 750000000); + + EXPECT_EQ_INT((int)(read_ptr - buf), (int)FILE_METADATA_WIRE_SIZE + (int)sizeof(int)); free(result); free(buf); @@ -74,6 +89,12 @@ static void test_metadata_send_receive_roundtrip() { original.gid = 1000; original.mtime_sec = 1234567890; original.mtime_nsec = 500000000; + original.atime_valid = false; + original.atime_sec = 0; + original.atime_nsec = 0; + original.crtime_valid = true; + original.crtime_sec = 1200000000; + original.crtime_nsec = 750000000; EXPECT_TRUE(metadata_send(p[1], &original)); @@ -86,6 +107,10 @@ static void test_metadata_send_receive_roundtrip() { EXPECT_EQ_INT(received->gid, 1000); EXPECT_EQ_INT(received->mtime_sec, 1234567890); EXPECT_EQ_INT(received->mtime_nsec, 500000000); + EXPECT_FALSE(received->atime_valid); + EXPECT_TRUE(received->crtime_valid); + EXPECT_EQ_INT(received->crtime_sec, 1200000000); + EXPECT_EQ_INT(received->crtime_nsec, 750000000); free(received); close(p[0]); @@ -109,10 +134,106 @@ static void test_metadata_send_null() { close(p[1]); } +static void test_metadata_rejects_invalid_values() { + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + io_set_fds(p[0], p[1]); + int32_t present = 2; + EXPECT_TRUE(send_n_data(p[1], &present, sizeof(present))); + int ok = 1; + EXPECT_NULL(metadata_receive(p[0], &ok)); + EXPECT_EQ_INT(ok, 0); + close(p[0]); + close(p[1]); +} + +/* metadata_receive must reject an out-of-range atime/crtime nsec even when the + * flag would otherwise be valid (defense-in-depth on the -U/-N wire fields). */ +static void test_metadata_receive_rejects_bad_optional_times() { + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + io_set_fds(p[0], p[1]); + + int32_t present = 1; + int32_t mode = 0644; + int32_t uid = 1000; + int32_t gid = 1000; + int64_t mtime_sec = 1; + int64_t mtime_nsec = 0; + int32_t atime_valid = 1; + int64_t atime_sec = 1; + int64_t atime_nsec = 2000000000; /* invalid: >= 1e9 */ + EXPECT_TRUE(send_n_data(p[1], &present, sizeof(present))); + EXPECT_TRUE(send_n_data(p[1], &mode, sizeof(mode))); + EXPECT_TRUE(send_n_data(p[1], &uid, sizeof(uid))); + EXPECT_TRUE(send_n_data(p[1], &gid, sizeof(gid))); + EXPECT_TRUE(send_n_data(p[1], &mtime_sec, sizeof(mtime_sec))); + EXPECT_TRUE(send_n_data(p[1], &mtime_nsec, sizeof(mtime_nsec))); + EXPECT_TRUE(send_n_data(p[1], &atime_valid, sizeof(atime_valid))); + EXPECT_TRUE(send_n_data(p[1], &atime_sec, sizeof(atime_sec))); + EXPECT_TRUE(send_n_data(p[1], &atime_nsec, sizeof(atime_nsec))); + int32_t crtime_valid = 0; + int64_t crtime_sec = 0; + int64_t crtime_nsec = 0; + EXPECT_TRUE(send_n_data(p[1], &crtime_valid, sizeof(crtime_valid))); + EXPECT_TRUE(send_n_data(p[1], &crtime_sec, sizeof(crtime_sec))); + EXPECT_TRUE(send_n_data(p[1], &crtime_nsec, sizeof(crtime_nsec))); + int ok = 1; + EXPECT_NULL(metadata_receive(p[0], &ok)); + EXPECT_EQ_INT(ok, 0); + close(p[0]); + close(p[1]); +} + +/* file_restore_metadata applies the source atime alongside mtime when -U + * captured it (atime_valid set). */ +static void test_file_restore_metadata_applies_atime() { + const char* path = "temp_meta_atime_test.txt"; + EXPECT_TRUE(file_write_to_disk(path, "atime", 5, false, false)); + + FileMetadata m; + m.mode = 0644; + m.uid = getuid(); + m.gid = getgid(); + m.mtime_sec = 1234567890; + m.mtime_nsec = 0; + m.atime_valid = true; + m.atime_sec = 999999999; + m.atime_nsec = 123456789; + m.crtime_valid = false; + m.crtime_sec = 0; + m.crtime_nsec = 0; + + file_restore_metadata(path, &m, false); + + struct stat st; + EXPECT_EQ_INT(stat(path, &st), 0); + EXPECT_EQ_INT((int)st.st_mtime, 1234567890); +#ifdef __linux__ + EXPECT_EQ_INT((int)st.st_atime, 999999999); +#else + EXPECT_EQ_INT((int)st.st_atime, 999999999); +#endif + + unlink(path); +} + +static void test_metadata_mtime_window() { + EXPECT_TRUE(metadata_mtime_matches(100, 100000000, 101, 600000000, 2)); + EXPECT_FALSE(metadata_mtime_matches(100, 100000000, 102, 600000000, 2)); + EXPECT_TRUE(metadata_mtime_matches(100, 100000000, 102, 100000000, 2)); + EXPECT_TRUE(metadata_mtime_matches(100, 900000000, 102, 100000000, 2)); + EXPECT_FALSE(metadata_mtime_matches(100, 100000000, 102, 900000000, 2)); + EXPECT_TRUE(metadata_mtime_matches(100, 900000000, 102, 900000000, 2)); + EXPECT_FALSE(metadata_mtime_matches(100, 900000000, 101, 100000001, 0)); + EXPECT_TRUE(metadata_mtime_matches(100, 100000000, 100, 100000001, 0)); + EXPECT_TRUE(metadata_mtime_matches(100, 100000000, 100, 100000000, 0)); +} + static void test_file_restore_metadata() { const char* path = "temp_meta_restore_test.txt"; const char* content = "test content"; - EXPECT_TRUE(to_disk(path, content, strlen(content))); + EXPECT_TRUE(file_write_to_disk(path, content, strlen(content), false, false)); FileMetadata m; m.mode = 0644; @@ -121,7 +242,7 @@ static void test_file_restore_metadata() { m.mtime_sec = 1234567890; m.mtime_nsec = 0; - file_restore_metadata(path, &m); + file_restore_metadata(path, &m, false); struct stat st; EXPECT_EQ_INT(stat(path, &st), 0); @@ -131,11 +252,108 @@ static void test_file_restore_metadata() { unlink(path); } +static void test_file_restore_executability_only() { + const char* path = "temp_exec_restore_test.txt"; + EXPECT_TRUE(file_write_to_disk(path, "x", 1, false, false)); + EXPECT_EQ_INT(chmod(path, 0644), 0); + + FileMetadata m = { + .mode = 0751, .uid = getuid(), .gid = getgid(), .mtime_sec = 0, .mtime_nsec = 0}; + file_restore_metadata(path, &m, true); + + struct stat st; + EXPECT_EQ_INT(stat(path, &st), 0); + EXPECT_EQ_INT(st.st_mode & 0777, 0755); + unlink(path); +} + +static void test_directory_restore_executability_only() { + const char* path = "temp_exec_restore_test_dir"; + EXPECT_EQ_INT(mkdir(path, 0700), 0); + + FileMetadata m = { + .mode = 0755, .uid = getuid(), .gid = getgid(), .mtime_sec = 0, .mtime_nsec = 0}; + file_restore_metadata(path, &m, true); + + struct stat st; + EXPECT_EQ_INT(stat(path, &st), 0); + EXPECT_EQ_INT(st.st_mode & 0777, 0711); + rmdir(path); +} + +static void test_chmod_changes() { + mode_t result; + EXPECT_TRUE(chmod_apply(0777, "u=rw,go=r", &result)); + EXPECT_EQ_INT(result, 0644); + EXPECT_TRUE(chmod_apply(0644, "a+x", &result)); + EXPECT_EQ_INT(result, 0755); + result = 0777; + EXPECT_TRUE(chmod_apply(0777, "0000", &result)); + EXPECT_EQ_INT(result, 0000); + result = 0777; + EXPECT_TRUE(chmod_apply(0777, "7777", &result)); + EXPECT_EQ_INT(result, 07777); + result = 0777; + EXPECT_TRUE(chmod_apply(0777, "755", &result)); + EXPECT_EQ_INT(result, 0755); + EXPECT_FALSE(chmod_apply(0777, "888", &result)); + EXPECT_FALSE(chmod_apply(0777, "10000", &result)); + EXPECT_FALSE(chmod_apply(0777, "a+X", &result)); + EXPECT_FALSE(chmod_apply(0777, "a+r,", &result)); +} + +/* P7 Wave D: symlink metadata is applied with no-follow primitives, and -J + * (omit_link_times) suppresses the timestamp. The positive apply path is + * asserted when the filesystem actually stores symlink timestamps; a filesystem + * that silently ignores them (or a platform where utimensat AT_SYMLINK_NOFOLLOW + * is unsupported) is tolerated, in which case only the omit-path invariant is + * checked. */ +static void test_file_restore_symlink_metadata() { + const char* dir = "temp_symlink_md_test"; + const char* target = "temp_symlink_md_test/target"; + const char* link = "temp_symlink_md_test/link"; + EXPECT_EQ_INT(mkdir(dir, 0755), 0); + FILE* f = fopen(target, "w"); + EXPECT_NOT_NULL(f); + fputs("t", f); + fclose(f); + EXPECT_EQ_INT(symlink("target", link), 0); + + /* Positive path: a non-omitted apply stamps the link's own mtime. */ + FileMetadata applied = {.mtime_sec = 1000000000, .mtime_nsec = 0}; + file_restore_symlink_metadata(link, &applied, false); + struct stat st; + EXPECT_EQ_INT(lstat(link, &st), 0); + EXPECT_TRUE(S_ISLNK(st.st_mode)); + bool symlink_times_supported = ((int)st.st_mtime == 1000000000); + time_t t1 = st.st_mtime; + + /* -J: a different time must be left untouched. */ + FileMetadata newer = {.mtime_sec = 1234567890, .mtime_nsec = 0}; + file_restore_symlink_metadata(link, &newer, true); + EXPECT_EQ_INT(lstat(link, &st), 0); + EXPECT_EQ_INT((int)st.st_mtime, (int)t1); + if (symlink_times_supported) + EXPECT_EQ_INT((int)st.st_mtime, 1000000000); + + unlink(link); + unlink(target); + rmdir(dir); +} + void test_metadata() { test_metadata_to_from_buf_roundtrip(); test_metadata_to_buf_null(); test_metadata_from_buf_null(); test_metadata_send_receive_roundtrip(); test_metadata_send_null(); + test_metadata_rejects_invalid_values(); + test_metadata_receive_rejects_bad_optional_times(); + test_metadata_mtime_window(); test_file_restore_metadata(); + test_file_restore_metadata_applies_atime(); + test_file_restore_executability_only(); + test_directory_restore_executability_only(); + test_file_restore_symlink_metadata(); + test_chmod_changes(); } diff --git a/tests/test_motd.c b/tests/test_motd.c new file mode 100644 index 0000000..3b1c650 --- /dev/null +++ b/tests/test_motd.c @@ -0,0 +1,149 @@ +#include "test_motd.h" +#include "motd.h" +#include "protocol.h" +#include "test_utils.h" +#include +#include +#include +#include + +/* Write `body` (len bytes) to a fresh temp file; returns its heap path. */ +static int write_file(const char* body, size_t len, char** out_path) { + char tmpl[] = "/tmp/fastsync_motd_XXXXXX"; + int fd = mkstemp(tmpl); + if (fd < 0) + return -1; + if (write(fd, body, len) != (ssize_t)len) { + close(fd); + unlink(tmpl); + return -1; + } + close(fd); + *out_path = strdup(tmpl); + return *out_path ? 0 : -1; +} + +static void test_motd_read_present() { + char* path; + char body[] = "Welcome to FastSync\nBe excellent to each other.\n"; + EXPECT_EQ_INT(write_file(body, strlen(body), &path), 0); + char* motd = motd_read_file(path); + unlink(path); + free(path); + EXPECT_NOT_NULL(motd); + EXPECT_EQ_STR(motd, body); + free(motd); +} + +static void test_motd_read_absent() { + EXPECT_NULL(motd_read_file("/nonexistent/fastsync_motd_zzz")); + EXPECT_NULL(motd_read_file("")); + EXPECT_NULL(motd_read_file(NULL)); +} + +static void test_motd_read_unreadable() { + /* Reading a directory through fopen succeeds for the open but fread fails + * with EISDIR, which is a reliable "unreadable" probe even for root. */ + const char* dir = "/tmp"; + EXPECT_NULL(motd_read_file(dir)); +} + +static void test_motd_read_large_truncated() { + size_t total = MOTD_MAX_BYTES + 100; + char* body = malloc(total); + EXPECT_NOT_NULL(body); + memset(body, 'x', total); + body[0] = 'h'; + char* path; + EXPECT_EQ_INT(write_file(body, total, &path), 0); + char* motd = motd_read_file(path); + unlink(path); + free(path); + EXPECT_NOT_NULL(motd); + EXPECT_EQ_INT((int)strlen(motd), MOTD_MAX_BYTES); + EXPECT_EQ_INT(motd[0], 'h'); + EXPECT_EQ_INT(motd[MOTD_MAX_BYTES - 1], 'x'); + EXPECT_EQ_STR(motd + MOTD_MAX_BYTES, ""); + free(motd); + free(body); +} + +static void test_motd_render_escaping() { + /* Newlines/tabs survive; control bytes (ESC included) become \NNN octal. */ + char* rendered = motd_render("line1\n\tansi\033[31m", false); + EXPECT_NOT_NULL(rendered); + EXPECT_EQ_STR(rendered, "line1\n\tansi\\#033[31m"); + free(rendered); + + /* High-bit bytes are escaped without --8-bit-output. */ + rendered = motd_render("\xC3\xA9", false); + EXPECT_NOT_NULL(rendered); + EXPECT_EQ_STR(rendered, "\\#303\\#251"); + free(rendered); + + /* --8-bit-output keeps bytes >= 0x80 verbatim. */ + rendered = motd_render("\xC3\xA9", true); + EXPECT_NOT_NULL(rendered); + EXPECT_EQ_STR(rendered, "\xC3\xA9"); + free(rendered); + + EXPECT_NULL(motd_render(NULL, false)); +} + +static void test_motd_frame_roundtrip() { + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + const char* motd = "Greetings from the module server.\nEnjoy your stay.\n"; + EXPECT_TRUE(motd_send(0, motd)); + char* received = motd_receive(0); + EXPECT_NOT_NULL(received); + EXPECT_EQ_STR(received, motd); + free(received); + + /* An unset MOTD is an empty (but present) frame, not an error. */ + EXPECT_TRUE(motd_send(0, NULL)); + received = motd_receive(0); + EXPECT_NOT_NULL(received); + EXPECT_EQ_STR(received, ""); + free(received); + + close(p[0]); + close(p[1]); +} + +static void test_motd_receive_over_bound_rejected() { + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + size_t size = MOTD_MAX_BYTES + 100; + char* big = malloc(size); + EXPECT_NOT_NULL(big); + memset(big, 'a', size); + big[size - 1] = '\0'; + /* A (hostile/oversized) peer frame within MAX_STRING_SIZE but above the MOTD + * bound is consumed and discarded: motd_receive returns NULL and the stream + * stays framed for the next message. */ + EXPECT_TRUE(send_str(0, big)); + EXPECT_NULL(motd_receive(0)); + EXPECT_TRUE(send_str(0, "after")); + char* next = receive_str(0); + EXPECT_NOT_NULL(next); + EXPECT_EQ_STR(next, "after"); + free(next); + free(big); + close(p[0]); + close(p[1]); +} + +void test_motd() { + test_motd_read_present(); + test_motd_read_absent(); + test_motd_read_unreadable(); + test_motd_read_large_truncated(); + test_motd_render_escaping(); + test_motd_frame_roundtrip(); + test_motd_receive_over_bound_rejected(); +} \ No newline at end of file diff --git a/tests/test_motd.h b/tests/test_motd.h new file mode 100644 index 0000000..731cf06 --- /dev/null +++ b/tests/test_motd.h @@ -0,0 +1,6 @@ +#ifndef TEST_MOTD_H +#define TEST_MOTD_H + +void test_motd(void); + +#endif \ No newline at end of file diff --git a/tests/test_multiprocessing.c b/tests/test_multiprocessing.c index 54ed5d6..fb0241d 100644 --- a/tests/test_multiprocessing.c +++ b/tests/test_multiprocessing.c @@ -82,7 +82,7 @@ static void test_sender_queue_capacities() { pipeline_context_sender_destroy(ctx); } -/* Test that create handles zero-capacity queues */ +/* Invalid queue capacities must not create unusable pipeline queues. */ static void test_sender_zero_capacity() { Config* cfg = config_create(); EXPECT_NOT_NULL(cfg); @@ -91,13 +91,13 @@ static void test_sender_zero_capacity() { cfg->send_directory = str_dup("/src"); cfg->receive_root_directory = str_dup("/dst"); - Queue* q1 = queue_create(0, NULL); - Queue* q2 = queue_create(0, NULL); - PipelineContextSender* ctx = pipeline_context_sender_create(cfg, q1, q2); - EXPECT_NOT_NULL(ctx); - EXPECT_EQ_INT(ctx->queue_scanner->capacity, 0); - EXPECT_EQ_INT(ctx->queue_loader->capacity, 0); - pipeline_context_sender_destroy(ctx); + // cppcheck-suppress constVariablePointer + Queue* const q1 = queue_create(0, NULL); + // cppcheck-suppress constVariablePointer + Queue* const q2 = queue_create(0, NULL); + EXPECT_NULL(q1); + EXPECT_NULL(q2); + config_delete(cfg); } /* Test receiver with zero file_descriptor */ @@ -168,6 +168,41 @@ static void test_receive_thread_finished() { } } +/* A malformed terminal status must wake a writer waiting on an empty queue. */ +static void test_receive_thread_failure_wakes_writer() { + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + free(cfg->version); + cfg->version = str_dup(PROTOCOL_VERSION); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/tmp/dst"); + + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + Queue* q = queue_create(1, file_destroy); + EXPECT_NOT_NULL(q); + PipelineContextReceiver* ctx = pipeline_context_receiver_create(cfg, q, p[0], NULL); + EXPECT_NOT_NULL(ctx); + + thrd_t receiver; + thrd_t writer; + EXPECT_EQ_INT(thrd_create(&writer, write_thread, ctx), thrd_success); + EXPECT_EQ_INT(thrd_create(&receiver, receive_thread, ctx), thrd_success); + EXPECT_TRUE(send_status(p[1], STATUS_OK)); + close(p[1]); + + int receiver_result; + int writer_result; + EXPECT_EQ_INT(thrd_join(receiver, &receiver_result), thrd_success); + EXPECT_EQ_INT(thrd_join(writer, &writer_result), thrd_success); + EXPECT_EQ_INT(receiver_result, thrd_error); + EXPECT_EQ_INT(writer_result, thrd_success); + EXPECT_TRUE(ctx->receiver_done); + + close(p[0]); + pipeline_context_receiver_destroy(ctx); +} + /* Test that write_thread completes cleanly when queue signals done */ static void test_write_thread_done() { Config* cfg = config_create(); @@ -215,6 +250,88 @@ static void test_write_thread_done() { config_delete(cfg); } +typedef struct { + PipelineContextReceiver* context; + File* file; + atomic_bool* done; + atomic_bool* result; +} ByteBudgetEnqueueArg; + +static int byte_budget_enqueue_worker(void* arg) { + ByteBudgetEnqueueArg* worker = arg; + bool ok = pipeline_context_receiver_enqueue_file(worker->context, worker->file); + atomic_store(worker->result, ok); + atomic_store(worker->done, true); + return thrd_success; +} + +/* A receiver must not buffer more decompressed/copied payload bytes ahead of + the (slow) disk writer than the configured byte budget: an enqueue that + would exceed the budget blocks until the writer releases bytes. */ +static void test_receiver_enqueue_byte_budget() { + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + free(cfg->version); + cfg->version = str_dup(PROTOCOL_VERSION); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + cfg->save_to_disk = false; + + Queue* q = queue_create(16, file_destroy); + EXPECT_NOT_NULL(q); + PipelineContextReceiver* ctx = pipeline_context_receiver_create(cfg, q, -1, NULL); + EXPECT_NOT_NULL(ctx); + pipeline_context_receiver_set_queue_byte_limit(ctx, 3000); + ctx->receiver_done = false; + + File* first = file_create("budget_file_1"); + EXPECT_NOT_NULL(first); + first->data->size = 2000; + EXPECT_TRUE(pipeline_context_receiver_enqueue_file(ctx, first)); + EXPECT_EQ_INT((int)ctx->queued_bytes, 2000); + + /* Second 2000-byte payload would push the pipeline to 4000 > 3000 budget, + so the enqueue must block until the first payload is released. */ + File* second = file_create("budget_file_2"); + EXPECT_NOT_NULL(second); + second->data->size = 2000; + atomic_bool done; + atomic_bool result; + atomic_init(&done, false); + atomic_init(&result, false); + ByteBudgetEnqueueArg arg = {ctx, second, &done, &result}; + thrd_t enqueuer; + EXPECT_EQ_INT(thrd_create(&enqueuer, byte_budget_enqueue_worker, &arg), thrd_success); + + /* Give a broken (unbounded) implementation every chance to enqueue. */ + struct timespec wait = {0, 200 * 1000000L}; + thrd_sleep(&wait, NULL); + EXPECT_FALSE(atomic_load(&done)); + EXPECT_EQ_INT((int)ctx->queued_bytes, 2000); /* budget still honored */ + + /* Simulate the disk writer: dequeue + destroy + release the first file. + Releasing bytes unblocks the waiting enqueuer, which then admits the + second payload, so only the post-join state (below) is deterministic. */ + File* drained = queue_dequeue_multithreaded(q, &ctx->mutex, &ctx->condition_not_empty, + &ctx->condition_not_full, &ctx->receiver_done); + EXPECT_NOT_NULL(drained); + file_destroy(drained); + pipeline_context_receiver_note_bytes_released(ctx, 2000); + + EXPECT_EQ_INT(thrd_join(enqueuer, NULL), thrd_success); + EXPECT_TRUE(atomic_load(&done)); + EXPECT_TRUE(atomic_load(&result)); + EXPECT_EQ_INT((int)ctx->queued_bytes, 2000); /* second payload now in flight */ + + /* Tear down: the second file is still queued and is freed by queue_destroy. */ + mtx_destroy(&ctx->mutex); + cnd_destroy(&ctx->condition_not_full); + cnd_destroy(&ctx->condition_not_empty); + free(ctx); + queue_destroy(q); + config_delete(cfg); +} + void test_multiprocessing() { test_sender_create_destroy(); test_receiver_create_destroy(); @@ -223,6 +340,8 @@ void test_multiprocessing() { test_receiver_fd_zero(); if (!is_running_under_valgrind()) { test_receive_thread_finished(); + test_receive_thread_failure_wakes_writer(); } test_write_thread_done(); + test_receiver_enqueue_byte_budget(); } diff --git a/tests/test_property.c b/tests/test_property.c index 5840f4a..2a88dd5 100644 --- a/tests/test_property.c +++ b/tests/test_property.c @@ -14,6 +14,8 @@ static Data* random_data(int min_size, int max_size) { int size = min_size + rand() % (max_size - min_size + 1); char* buf = malloc(size); + if (!buf) + return NULL; for (int i = 0; i < size; i++) buf[i] = (char)(rand() % 256); return data_create(buf, size); @@ -81,10 +83,12 @@ static void test_property_chunk_roundtrip() { int content_len = 1 + rand() % 4096; char* content = malloc(content_len); + if (!content) + return; for (int i = 0; i < content_len; i++) content[i] = (char)(rand() % 256); - to_disk(path, content, content_len); + file_write_to_disk(path, content, content_len, false, false); struct stat st; stat(path, &st); diff --git a/tests/test_protocol.c b/tests/test_protocol.c index af1dc00..79fb8a0 100644 --- a/tests/test_protocol.c +++ b/tests/test_protocol.c @@ -3,6 +3,60 @@ #include #include #include +#include + +typedef struct { + ProtocolSession* session; + bool allocation_allowed; +} AllocationWorkerArg; + +static int allocation_worker(void* arg) { + AllocationWorkerArg* worker = arg; + protocol_session_bind(worker->session); + void* allocation = protocol_alloc(8); + worker->allocation_allowed = allocation != NULL; + free(allocation); + protocol_session_unbind(); + return thrd_success; +} + +typedef struct { + ProtocolSession* session; + int read_fd; + bool released; +} AccountingWorkerArg; + +typedef struct { + ProtocolSession* session; + atomic_int* ready; + atomic_bool* release; + bool received; +} ConcurrentAccountingWorkerArg; + +static int accounting_worker(void* arg) { + AccountingWorkerArg* worker = arg; + protocol_session_bind(worker->session); + Data* data = protocol_receive_data_limited(worker->session, 8); + if (data) { + data_destroy(data); + worker->released = atomic_load(&worker->session->total_allocated_bytes) == 0; + } + protocol_session_unbind(); + return data ? thrd_success : thrd_error; +} + +static int concurrent_accounting_worker(void* arg) { + ConcurrentAccountingWorkerArg* worker = arg; + protocol_session_bind(worker->session); + Data* data = protocol_receive_data_limited(worker->session, 8); + worker->received = data != NULL; + atomic_fetch_add(worker->ready, 1); + while (!atomic_load(worker->release)) + thrd_yield(); + data_destroy(data); + protocol_session_unbind(); + return thrd_success; +} static void test_send_receive_n_data() { int p[2]; @@ -38,6 +92,23 @@ static void test_send_receive_n_data_zero() { close(p[1]); } +static void test_explicit_session_context() { + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + ProtocolSession session; + protocol_session_init(&session, p[0], p[1]); + protocol_session_set_bwlimit(&session, 0); + + const char payload[] = "explicit context"; + char received[sizeof(payload)] = {0}; + EXPECT_TRUE(protocol_send_n_data(&session, payload, sizeof(payload))); + EXPECT_TRUE(protocol_receive_n_data(&session, received, sizeof(received))); + EXPECT_EQ_INT(memcmp(payload, received, sizeof(payload)), 0); + + close(p[0]); + close(p[1]); +} + static void test_send_receive_str() { int p[2]; EXPECT_EQ_INT(pipe(p), 0); @@ -170,14 +241,214 @@ static void test_receive_str_truncated() { close(p[0]); } +static void test_max_alloc_rejects_single_buffer() { + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + ProtocolSession session; + protocol_session_init(&session, p[0], p[1]); + protocol_session_set_max_alloc(&session, 4); + protocol_session_bind(&session); + char payload[8] = {0}; + EXPECT_TRUE(write(p[1], &(size_t){sizeof(payload)}, sizeof(size_t)) == sizeof(size_t)); + EXPECT_NULL(protocol_receive_str(&session)); + protocol_session_unbind(); + close(p[0]); + close(p[1]); +} + +static void test_explicit_session_max_alloc_cannot_be_bypassed() { + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + ProtocolSession explicit_session; + ProtocolSession unrelated_session; + protocol_session_init(&explicit_session, p[0], p[1]); + protocol_session_init(&unrelated_session, p[0], p[1]); + protocol_session_set_max_alloc(&explicit_session, 4); + protocol_session_set_max_alloc(&unrelated_session, 64); + protocol_session_bind(&unrelated_session); + + unsigned long long size = 8; + EXPECT_EQ_INT((int)write(p[1], &size, sizeof(size)), (int)sizeof(size)); + EXPECT_EQ_INT((int)write(p[1], "12345678", 8), 8); + EXPECT_NULL(protocol_receive_data_limited(&explicit_session, 8)); + EXPECT_EQ_INT((int)atomic_load(&explicit_session.total_allocated_bytes), 0); + + protocol_session_unbind(); + close(p[0]); + close(p[1]); +} + +static void test_max_alloc_allows_configured_buffer() { + ProtocolSession session; + protocol_session_init(&session, -1, -1); + protocol_session_set_max_alloc(&session, 4); + protocol_session_bind(&session); + void* allowed = protocol_alloc(4); + const void* rejected = protocol_alloc(5); + EXPECT_NOT_NULL(allowed); + EXPECT_NULL(rejected); + free(allowed); + protocol_session_unbind(); +} + +static void test_max_alloc_is_bound_in_worker_threads() { + enum { WORKER_COUNT = 4 }; + ProtocolSession sessions[WORKER_COUNT]; + AllocationWorkerArg args[WORKER_COUNT] = {0}; + thrd_t threads[WORKER_COUNT]; + for (int i = 0; i < WORKER_COUNT; i++) { + protocol_session_init(&sessions[i], -1, -1); + protocol_session_set_max_alloc(&sessions[i], 4); + args[i].session = &sessions[i]; + EXPECT_EQ_INT(thrd_create(&threads[i], allocation_worker, &args[i]), thrd_success); + } + for (int i = 0; i < WORKER_COUNT; i++) { + int result; + EXPECT_EQ_INT(thrd_join(threads[i], &result), thrd_success); + EXPECT_EQ_INT(result, thrd_success); + EXPECT_FALSE(args[i].allocation_allowed); + } +} + +static void test_protocol_accounting_is_released_in_worker_threads() { + enum { WORKER_COUNT = 4 }; + ProtocolSession sessions[WORKER_COUNT]; + AccountingWorkerArg args[WORKER_COUNT] = {0}; + thrd_t threads[WORKER_COUNT]; + for (int i = 0; i < WORKER_COUNT; i++) { + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + protocol_session_init(&sessions[i], p[0], p[1]); + protocol_session_set_max_alloc(&sessions[i], 64); + unsigned long long size = 8; + EXPECT_EQ_INT((int)write(p[1], &size, sizeof(size)), (int)sizeof(size)); + EXPECT_EQ_INT((int)write(p[1], "12345678", 8), 8); + close(p[1]); + args[i].session = &sessions[i]; + args[i].read_fd = p[0]; + EXPECT_EQ_INT(thrd_create(&threads[i], accounting_worker, &args[i]), thrd_success); + } + for (int i = 0; i < WORKER_COUNT; i++) { + int result; + EXPECT_EQ_INT(thrd_join(threads[i], &result), thrd_success); + EXPECT_EQ_INT(result, thrd_success); + EXPECT_TRUE(args[i].released); + EXPECT_EQ_INT((int)atomic_load(&sessions[i].total_allocated_bytes), 0); + close(args[i].read_fd); + } +} + +static void test_protocol_accounting_reservation_is_atomic() { + enum { WORKER_COUNT = 8 }; + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + ProtocolSession session; + protocol_session_init(&session, p[0], p[1]); + protocol_session_set_max_alloc(&session, 64); + const unsigned long long budget_before = MAX_SERVER_ALLOC - 8; + atomic_store(&session.total_allocated_bytes, budget_before); + + for (int i = 0; i < WORKER_COUNT; i++) { + unsigned long long size = 8; + EXPECT_EQ_INT((int)write(p[1], &size, sizeof(size)), (int)sizeof(size)); + EXPECT_EQ_INT((int)write(p[1], "12345678", 8), 8); + } + close(p[1]); + + atomic_int ready; + atomic_bool release; + atomic_init(&ready, 0); + atomic_init(&release, false); + ConcurrentAccountingWorkerArg args[WORKER_COUNT] = {0}; + thrd_t threads[WORKER_COUNT]; + for (int i = 0; i < WORKER_COUNT; i++) { + args[i].session = &session; + args[i].ready = &ready; + args[i].release = &release; + EXPECT_EQ_INT(thrd_create(&threads[i], concurrent_accounting_worker, &args[i]), thrd_success); + } + while (atomic_load(&ready) != WORKER_COUNT) + thrd_yield(); + bool budget_ok = atomic_load(&session.total_allocated_bytes) == budget_before + 8; + atomic_store(&release, true); + int received = 0; + for (int i = 0; i < WORKER_COUNT; i++) { + int result; + EXPECT_EQ_INT(thrd_join(threads[i], &result), thrd_success); + EXPECT_EQ_INT(result, thrd_success); + received += args[i].received ? 1 : 0; + } + EXPECT_EQ_INT(received, 1); + EXPECT_TRUE(budget_ok); + EXPECT_EQ_INT((int)atomic_load(&session.total_allocated_bytes), (int)budget_before); + close(p[0]); +} + +static void test_protocol_string_accounting_is_transient() { + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + ProtocolSession session; + protocol_session_init(&session, p[0], p[1]); + protocol_session_set_max_alloc(&session, 64); + EXPECT_TRUE(protocol_send_str(&session, "temporary")); + char* received = protocol_receive_str(&session); + EXPECT_NOT_NULL(received); + EXPECT_EQ_STR(received, "temporary"); + EXPECT_EQ_INT((int)atomic_load(&session.total_allocated_bytes), 0); + free(received); + close(p[0]); + close(p[1]); +} + +static void test_protocol_accounting_release_does_not_underflow() { + ProtocolSession session; + protocol_session_init(&session, -1, -1); + atomic_store(&session.total_allocated_bytes, 4); + protocol_session_bind(&session); + protocol_release_memory(8); + EXPECT_EQ_INT((int)atomic_load(&session.total_allocated_bytes), 0); + protocol_release_memory(1); + EXPECT_EQ_INT((int)atomic_load(&session.total_allocated_bytes), 0); + protocol_session_unbind(); +} + +static void test_send_receive_status_timed() { + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + /* The extended-deadline variant must read an ordinary status just like the + default window, and must fail cleanly on EOF rather than block. */ + EXPECT_TRUE(send_status(0, STATUS_OK)); + Status received = -1; + EXPECT_TRUE(receive_status_timed(0, &received, 5)); + EXPECT_EQ_INT((int)received, (int)STATUS_OK); + + close(p[1]); + EXPECT_FALSE(receive_status_timed(0, &received, 5)); + + close(p[0]); +} + void test_protocol() { test_send_receive_n_data(); test_send_receive_n_data_zero(); + test_explicit_session_context(); test_send_receive_str(); test_send_receive_str_normal(); test_send_receive_data(); test_send_receive_int(); test_send_receive_status(); + test_send_receive_status_timed(); test_receive_n_data_truncated(); test_receive_str_truncated(); + test_max_alloc_rejects_single_buffer(); + test_explicit_session_max_alloc_cannot_be_bypassed(); + test_max_alloc_allows_configured_buffer(); + test_max_alloc_is_bound_in_worker_threads(); + test_protocol_accounting_is_released_in_worker_threads(); + test_protocol_accounting_reservation_is_atomic(); + test_protocol_string_accounting_is_transient(); + test_protocol_accounting_release_does_not_underflow(); } diff --git a/tests/test_queue.c b/tests/test_queue.c index c392fcb..4c90153 100644 --- a/tests/test_queue.c +++ b/tests/test_queue.c @@ -56,6 +56,11 @@ static void test_queue_basic() { queue_destroy(q); } +static void test_queue_rejects_invalid_capacity() { + EXPECT_NULL(queue_create(0, NULL)); + EXPECT_NULL(queue_create(-1, NULL)); +} + static void test_queue_resize() { Queue* q = queue_create(3, NULL); EXPECT_NOT_NULL(q); @@ -117,6 +122,8 @@ static void test_queue_destroyer() { for (int i = 0; i < 3; i++) { int* val = malloc(sizeof(int)); + if (!val) + break; *val = i; queue_enqueue(q, val); } @@ -176,6 +183,8 @@ static void test_queue_multithreaded() { for (int i = 1; i <= 100; i++) { int* val = malloc(sizeof(int)); + if (!val) + break; *val = i; queue_enqueue_multithreaded(q, val, &mutex, &cnd_empty, &cnd_full); } @@ -199,6 +208,7 @@ static void test_queue_multithreaded() { void test_queue() { test_queue_basic(); + test_queue_rejects_invalid_capacity(); test_queue_resize(); test_queue_destroyer(); test_queue_multithreaded(); diff --git a/tests/test_robustness.c b/tests/test_robustness.c index 8a3668a..1e1cd09 100644 --- a/tests/test_robustness.c +++ b/tests/test_robustness.c @@ -13,7 +13,7 @@ static void test_chunk_deserialize_truncated() { char* path = "test_rob_trunc.txt"; char* content = "hello"; - to_disk(path, content, strlen(content)); + file_write_to_disk(path, content, strlen(content), false, false); struct stat st; stat(path, &st); @@ -109,6 +109,20 @@ static void test_delta_deserialize_garbage() { data_destroy(d); } +static void test_delta_deserialize_respects_max_alloc() { + unsigned char serialized[sizeof(uint64_t) + sizeof(uint32_t)] = {0}; + Data data = {.data = serialized, .size = sizeof(serialized)}; + ProtocolSession session; + protocol_session_init(&session, -1, -1); + protocol_session_set_max_alloc(&session, sizeof(Delta) - 1); + protocol_session_bind(&session); + + const Delta* result = delta_deserialize(&data); + EXPECT_NULL(result); + + protocol_session_unbind(); +} + static void test_delta_signature_deserialize_truncated() { char old_data[4096]; for (int i = 0; i < 4096; i++) @@ -206,6 +220,7 @@ void test_robustness() { test_delta_deserialize_truncated(); test_delta_deserialize_empty(); test_delta_deserialize_garbage(); + test_delta_deserialize_respects_max_alloc(); test_delta_deserialize_truncated_instructions(); test_delta_signature_deserialize_truncated(); test_delta_apply_null(); diff --git a/tests/test_scanner.c b/tests/test_scanner.c index 3b68cb4..06e74ad 100644 --- a/tests/test_scanner.c +++ b/tests/test_scanner.c @@ -1,13 +1,17 @@ #include "test_utils.h" #include "scanner.h" +#include "array_list.h" #include "file.h" +#include "file_list.h" +#include "filter.h" #include "utils.h" +#include #include #include #include static void create_test_file(const char* path, const char* content) { - (void)to_disk(path, content, strlen(content)); + (void)file_write_to_disk(path, content, strlen(content), false, false); } static void test_scanner_single_file() { @@ -19,7 +23,7 @@ static void test_scanner_single_file() { create_test_file(file1, content1); DirectoryScanner* scanner = directory_scanner_create((char*)dir, false, 0, NULL, 0, NULL, 0, 0, 0, - 0, false, false, false, false); + 0, false, false, false, false, false); EXPECT_NOT_NULL(scanner); Chunk* chunk = directory_scanner_next(scanner); @@ -48,7 +52,7 @@ static void test_scanner_multiple_files() { create_test_file(file2, content2); DirectoryScanner* scanner = directory_scanner_create((char*)dir, false, 0, NULL, 0, NULL, 0, 0, 0, - 0, false, false, false, false); + 0, false, false, false, false, false); EXPECT_NOT_NULL(scanner); const Chunk* chunk = directory_scanner_next(scanner); @@ -88,7 +92,7 @@ static void test_scanner_subdirectory() { create_test_file(sub_file, content); DirectoryScanner* scanner = directory_scanner_create((char*)root, false, 0, NULL, 0, NULL, 0, 0, - 0, 0, false, false, false, false); + 0, 0, false, false, false, false, false); EXPECT_NOT_NULL(scanner); int total_files = 0; @@ -112,7 +116,7 @@ static void test_scanner_empty_directory() { EXPECT_EQ_INT(mkdir(dir, 0755), 0); DirectoryScanner* scanner = directory_scanner_create((char*)dir, false, 0, NULL, 0, NULL, 0, 0, 0, - 0, false, false, false, false); + 0, false, false, false, false, false); EXPECT_NOT_NULL(scanner); const Chunk* chunk = directory_scanner_next(scanner); @@ -136,7 +140,7 @@ static void test_scanner_exclude_pattern() { char* exclude[] = {"*.tmp"}; DirectoryScanner* scanner = directory_scanner_create((char*)dir, false, 0, exclude, 1, NULL, 0, 0, - 0, 0, false, false, false, false); + 0, 0, false, false, false, false, false); EXPECT_NOT_NULL(scanner); Chunk* chunk = directory_scanner_next(scanner); @@ -169,7 +173,7 @@ static void test_scanner_exclude_subdirectory() { char* exclude[] = {"*.tmp"}; DirectoryScanner* scanner = directory_scanner_create((char*)root, false, 0, exclude, 1, NULL, 0, - 0, 0, 0, false, false, false, false); + 0, 0, 0, false, false, false, false, false); EXPECT_NOT_NULL(scanner); int total = 0; @@ -207,7 +211,7 @@ static void test_scanner_include_and_exclude() { char* exclude[] = {"*.bak"}; char* include[] = {"*.txt", "*.log"}; DirectoryScanner* scanner = directory_scanner_create((char*)dir, false, 0, exclude, 1, include, 2, - 0, 0, 0, false, false, false, false); + 0, 0, 0, false, false, false, false, false); EXPECT_NOT_NULL(scanner); Chunk* chunk = directory_scanner_next(scanner); @@ -244,7 +248,7 @@ static void test_scanner_max_size() { /* max_size = 10 — only files <= 10 bytes */ DirectoryScanner* scanner = directory_scanner_create((char*)dir, false, 0, NULL, 0, NULL, 0, 10, - 0, 0, false, false, false, false); + 0, 0, false, false, false, false, false); EXPECT_NOT_NULL(scanner); Chunk* chunk = directory_scanner_next(scanner); @@ -272,7 +276,7 @@ static void test_scanner_min_size() { /* min_size = 1 — only files >= 1 byte */ DirectoryScanner* scanner = directory_scanner_create((char*)dir, false, 0, NULL, 0, NULL, 0, 0, 1, - 0, false, false, false, false); + 0, false, false, false, false, false); EXPECT_NOT_NULL(scanner); Chunk* chunk = directory_scanner_next(scanner); @@ -302,7 +306,7 @@ static void test_scanner_size_range() { /* Only files between 3 and 20 bytes */ DirectoryScanner* scanner = directory_scanner_create((char*)dir, false, 0, NULL, 0, NULL, 0, 20, - 3, 0, false, false, false, false); + 3, 0, false, false, false, false, false); EXPECT_NOT_NULL(scanner); Chunk* chunk = directory_scanner_next(scanner); @@ -338,7 +342,7 @@ static void test_scanner_mixed_patterns() { char* exclude[] = {"*.bak"}; char* include[] = {"*.txt"}; DirectoryScanner* scanner = directory_scanner_create((char*)dir, false, 0, exclude, 1, include, 1, - 10, 3, 0, false, false, false, false); + 10, 3, 0, false, false, false, false, false); EXPECT_NOT_NULL(scanner); Chunk* chunk = directory_scanner_next(scanner); @@ -369,7 +373,7 @@ static void test_scanner_no_patterns() { create_test_file(f2, "second"); DirectoryScanner* scanner = directory_scanner_create((char*)dir, false, 0, NULL, 0, NULL, 0, 0, 0, - 0, false, false, false, false); + 0, false, false, false, false, false); EXPECT_NOT_NULL(scanner); Chunk* chunk = directory_scanner_next(scanner); @@ -385,6 +389,934 @@ static void test_scanner_no_patterns() { rmdir(dir); } +static void test_parallel_scanner_root_chunks_without_workers() { + const char* dir = "test_parallel_scan_root"; + const char* file1 = "test_parallel_scan_root/a.txt"; + const char* file2 = "test_parallel_scan_root/b.txt"; + + EXPECT_EQ_INT(mkdir(dir, 0755), 0); + create_test_file(file1, "a"); + create_test_file(file2, "b"); + + ScannerOptions options = {0}; + options.chunk_size = 1; + ParallelScanner* scanner = parallel_scanner_create_with_options(dir, &options, NULL); + EXPECT_NOT_NULL(scanner); + + int total_files = 0; + Chunk* chunk; + while ((chunk = parallel_scanner_next(scanner)) != NULL) { + total_files += chunk->element_count; + chunk_destroy(chunk); + } + EXPECT_EQ_INT(total_files, 2); + EXPECT_FALSE(parallel_scanner_failed(scanner)); + + parallel_scanner_destroy(scanner); + unlink(file1); + unlink(file2); + rmdir(dir); +} + +/* --one-file-system (-x) decision is a pure device comparison. */ +static void test_scanner_one_file_system_decision() { + /* Option disabled: every device is allowed (unchanged default behavior). */ + EXPECT_TRUE(scanner_same_filesystem(false, 0, 123)); + EXPECT_TRUE(scanner_same_filesystem(false, 7, 999)); + /* Option enabled: only entries on the root device may be descended into. */ + EXPECT_TRUE(scanner_same_filesystem(true, 7, 7)); + EXPECT_FALSE(scanner_same_filesystem(true, 7, 8)); +} + +/* With -x over an ordinary tree (all one device) nothing may be skipped. */ +static void test_scanner_one_file_system_same_device() { + const char* root = "test_scan_ofs"; + const char* sub = "test_scan_ofs/sub"; + const char* deeper = "test_scan_ofs/sub/deeper"; + const char* root_file = "test_scan_ofs/root.txt"; + const char* sub_file = "test_scan_ofs/sub/inner.txt"; + const char* deep_file = "test_scan_ofs/sub/deeper/deep.txt"; + + EXPECT_EQ_INT(mkdir(root, 0755), 0); + EXPECT_EQ_INT(mkdir(sub, 0755), 0); + EXPECT_EQ_INT(mkdir(deeper, 0755), 0); + create_test_file(root_file, "root"); + create_test_file(sub_file, "inner"); + create_test_file(deep_file, "deep"); + + ScannerOptions options = {0}; + options.one_file_system = true; + DirectoryScanner* scanner = directory_scanner_create_with_options(root, &options); + EXPECT_NOT_NULL(scanner); + + int total_files = 0; + Chunk* chunk; + while ((chunk = directory_scanner_next(scanner)) != NULL) { + total_files += chunk->element_count; + chunk_destroy(chunk); + } + EXPECT_EQ_INT(total_files, 3); + EXPECT_FALSE(directory_scanner_failed(scanner)); + + directory_scanner_destroy(scanner); + unlink(root_file); + unlink(sub_file); + unlink(deep_file); + rmdir(deeper); + rmdir(sub); + rmdir(root); +} + +/* Multithreaded (-m) scan with -x over a single-device tree must match the + * single-threaded result. */ +static void test_parallel_scanner_one_file_system_same_device() { + const char* root = "test_parallel_scan_ofs"; + const char* sub = "test_parallel_scan_ofs/sub"; + const char* sub2 = "test_parallel_scan_ofs/sub2"; + const char* root_file = "test_parallel_scan_ofs/root.txt"; + const char* sub_file = "test_parallel_scan_ofs/sub/inner.txt"; + const char* sub2_file = "test_parallel_scan_ofs/sub2/inner2.txt"; + + EXPECT_EQ_INT(mkdir(root, 0755), 0); + EXPECT_EQ_INT(mkdir(sub, 0755), 0); + EXPECT_EQ_INT(mkdir(sub2, 0755), 0); + create_test_file(root_file, "root"); + create_test_file(sub_file, "inner"); + create_test_file(sub2_file, "inner2"); + + ScannerOptions options = {0}; + options.one_file_system = true; + options.num_threads = 2; + ParallelScanner* scanner = parallel_scanner_create_with_options(root, &options, NULL); + EXPECT_NOT_NULL(scanner); + + int total_files = 0; + Chunk* chunk; + while ((chunk = parallel_scanner_next(scanner)) != NULL) { + total_files += chunk->element_count; + chunk_destroy(chunk); + } + EXPECT_EQ_INT(total_files, 3); + EXPECT_FALSE(parallel_scanner_failed(scanner)); + + parallel_scanner_destroy(scanner); + unlink(root_file); + unlink(sub_file); + unlink(sub2_file); + rmdir(sub); + rmdir(sub2); + rmdir(root); +} + +/* Scan a tree with copy_links semantics, collecting every emitted path. + * Returns 0 on success, -1 on scanner failure. */ +static int collect_directory_scan(const char* root, bool one_file_system, const char* needle, + bool* found, int* total) { + ScannerOptions options = {0}; + options.copy_links = true; + options.one_file_system = one_file_system; + DirectoryScanner* scanner = directory_scanner_create_with_options(root, &options); + if (!scanner) + return -1; + *found = false; + *total = 0; + Chunk* chunk; + while ((chunk = directory_scanner_next(scanner)) != NULL) { + for (int i = 0; i < chunk->element_count; i++) { + (*total)++; + if (strstr(chunk->items[i]->path, needle) != NULL) + *found = true; + } + chunk_destroy(chunk); + } + bool failed = directory_scanner_failed(scanner); + directory_scanner_destroy(scanner); + return failed ? -1 : 0; +} + +static int collect_parallel_scan(const char* root, bool one_file_system, const char* needle, + bool* found, int* total) { + ScannerOptions options = {0}; + options.copy_links = true; + options.one_file_system = one_file_system; + options.num_threads = 2; + ParallelScanner* scanner = parallel_scanner_create_with_options(root, &options, NULL); + if (!scanner) + return -1; + *found = false; + *total = 0; + Chunk* chunk; + while ((chunk = parallel_scanner_next(scanner)) != NULL) { + for (int i = 0; i < chunk->element_count; i++) { + (*total)++; + if (strstr(chunk->items[i]->path, needle) != NULL) + *found = true; + } + chunk_destroy(chunk); + } + bool failed = parallel_scanner_failed(scanner); + parallel_scanner_destroy(scanner); + return failed ? -1 : 0; +} + +/* Rootless cross-filesystem test: a symlink nested under the scan root points + * at a directory on another device (typically /dev/shm, a tmpfs distinct from + * the build filesystem). With --copy-links semantics the scanner resolves the + * link and must descend into it only when -x is off. The nested placement + * exercises the skip decision in the sequential walker and in the parallel + * worker (depth > 1). Skips when no cross-device target is available. */ +static void test_scanner_one_file_system_cross_device() { + struct stat local_stat; + if (stat(".", &local_stat) != 0) + return; + + char shm_dir[64] = "/dev/shm/fastsync_ofs_shm_XXXXXX"; + if (mkdtemp(shm_dir) == NULL) + return; + struct stat shm_stat; + if (stat(shm_dir, &shm_stat) != 0 || shm_stat.st_dev == local_stat.st_dev) { + rmdir(shm_dir); + return; + } + + char root_dir[64] = "./fastsync_ofs_root_XXXXXX"; + if (mkdtemp(root_dir) == NULL) { + rmdir(shm_dir); + return; + } + + char nested[96]; + snprintf(nested, sizeof(nested), "%s/nested", root_dir); + char link_path[128]; + snprintf(link_path, sizeof(link_path), "%s/link", nested); + char root_file[96]; + snprintf(root_file, sizeof(root_file), "%s/keep.txt", root_dir); + char shm_file[96]; + snprintf(shm_file, sizeof(shm_file), "%s/inside.txt", shm_dir); + + bool ready = mkdir(nested, 0755) == 0 && symlink(shm_dir, link_path) == 0; + if (ready) { + create_test_file(root_file, "keep"); + create_test_file(shm_file, "cross"); + } + + int seq_off_rc, seq_off_total, seq_on_rc, seq_on_total; + bool seq_off_found, seq_on_found; + int par_off_rc, par_off_total, par_on_rc, par_on_total; + bool par_off_found, par_on_found; + if (!ready) { + seq_off_rc = seq_on_rc = par_off_rc = par_on_rc = -1; + seq_off_total = seq_on_total = par_off_total = par_on_total = 0; + seq_off_found = seq_on_found = par_off_found = par_on_found = false; + } else { + int rc, total; + bool found; + rc = collect_directory_scan(root_dir, false, "inside.txt", &found, &total); + seq_off_rc = rc; + seq_off_total = total; + seq_off_found = found; + rc = collect_directory_scan(root_dir, true, "inside.txt", &found, &total); + seq_on_rc = rc; + seq_on_total = total; + seq_on_found = found; + rc = collect_parallel_scan(root_dir, false, "inside.txt", &found, &total); + par_off_rc = rc; + par_off_total = total; + par_off_found = found; + rc = collect_parallel_scan(root_dir, true, "inside.txt", &found, &total); + par_on_rc = rc; + par_on_total = total; + par_on_found = found; + } + + /* Hermetic cleanup regardless of scan outcome, before any assertions. */ + unlink(shm_file); + rmdir(shm_dir); + unlink(link_path); + unlink(root_file); + rmdir(nested); + rmdir(root_dir); + + if (!ready) + return; + + /* Sequential: without -x the symlinked foreign subtree is included. */ + EXPECT_EQ_INT(seq_off_rc, 0); + EXPECT_TRUE(seq_off_found); + EXPECT_EQ_INT(seq_off_total, 2); + /* Sequential: with -x the cross-device subtree is dropped, keep.txt remains. */ + EXPECT_EQ_INT(seq_on_rc, 0); + EXPECT_FALSE(seq_on_found); + EXPECT_EQ_INT(seq_on_total, 1); + /* Parallel: same behavior, worker path (depth > 1). */ + EXPECT_EQ_INT(par_off_rc, 0); + EXPECT_TRUE(par_off_found); + EXPECT_EQ_INT(par_off_total, 2); + EXPECT_EQ_INT(par_on_rc, 0); + EXPECT_FALSE(par_on_found); + EXPECT_EQ_INT(par_on_total, 1); +} + +/* Collect emitted file paths (relative to `root`) from a sequential scan. + * Returns 0 on success with *out and *count set (caller frees *out). */ +static int collect_files(const char* root, const ScannerOptions* options, char*** out, + int* out_count) { + DirectoryScanner* scanner = directory_scanner_create_with_options(root, options); + if (!scanner) + return -1; + size_t root_len = strlen(root); + while (root_len > 0 && root[root_len - 1] == '/') + root_len--; + int cap = 16; + int count = 0; + char** paths = malloc((size_t)cap * sizeof(char*)); + if (!paths) { + directory_scanner_destroy(scanner); + return -1; + } + Chunk* chunk; + while ((chunk = directory_scanner_next(scanner)) != NULL) { + for (int i = 0; i < chunk->element_count; i++) { + const char* rel = chunk->items[i]->path + root_len; + if (*rel == '/') + rel++; + if (count == cap) { + cap *= 2; + char** grown = realloc(paths, (size_t)cap * sizeof(char*)); + if (!grown) { + for (int k = 0; k < count; k++) + free(paths[k]); + free(paths); + chunk_destroy(chunk); + directory_scanner_destroy(scanner); + return -1; + } + paths = grown; + } + paths[count++] = str_dup(rel); + } + chunk_destroy(chunk); + } + bool failed = directory_scanner_failed(scanner); + directory_scanner_destroy(scanner); + if (failed) { + for (int k = 0; k < count; k++) + free(paths[k]); + free(paths); + return -1; + } + *out = paths; + *out_count = count; + return 0; +} + +static int collect_files_parallel(const char* root, const ScannerOptions* options, char*** out, + int* out_count) { + ParallelScanner* scanner = parallel_scanner_create_with_options(root, options, NULL); + if (!scanner) + return -1; + size_t root_len = strlen(root); + while (root_len > 0 && root[root_len - 1] == '/') + root_len--; + int cap = 16; + int count = 0; + char** paths = malloc((size_t)cap * sizeof(char*)); + if (!paths) { + parallel_scanner_destroy(scanner); + return -1; + } + Chunk* chunk; + while ((chunk = parallel_scanner_next(scanner)) != NULL) { + for (int i = 0; i < chunk->element_count; i++) { + const char* rel = chunk->items[i]->path + root_len; + if (*rel == '/') + rel++; + if (count == cap) { + cap *= 2; + char** grown = realloc(paths, (size_t)cap * sizeof(char*)); + if (!grown) { + for (int k = 0; k < count; k++) + free(paths[k]); + free(paths); + chunk_destroy(chunk); + parallel_scanner_destroy(scanner); + return -1; + } + paths = grown; + } + paths[count++] = str_dup(rel); + } + chunk_destroy(chunk); + } + bool failed = parallel_scanner_failed(scanner); + parallel_scanner_destroy(scanner); + if (failed) { + for (int k = 0; k < count; k++) + free(paths[k]); + free(paths); + return -1; + } + *out = paths; + *out_count = count; + return 0; +} + +static bool has_path(char** paths, int count, const char* rel) { + for (int i = 0; i < count; i++) + if (strcmp(paths[i], rel) == 0) + return true; + return false; +} + +static void free_paths(char** paths, int count) { + for (int i = 0; i < count; i++) + free(paths[i]); + free(paths); +} + +static const char* FILE_LIST_PATH = "test_scan_files_from.txt"; + +/* --files-from: only the listed files (and the subtree of a listed directory) + * are emitted; unrelated files and directories are pruned. */ +static void test_files_from_subset(bool parallel) { + const char* root = "test_scan_ff"; + const char* sub = "test_scan_ff/sub"; + const char* other = "test_scan_ff/other"; + EXPECT_EQ_INT(mkdir(root, 0755), 0); + EXPECT_EQ_INT(mkdir(sub, 0755), 0); + EXPECT_EQ_INT(mkdir(other, 0755), 0); + create_test_file("test_scan_ff/root.txt", "root"); + create_test_file("test_scan_ff/sub/keep.txt", "keep"); + create_test_file("test_scan_ff/sub/skip.bin", "skip"); + create_test_file("test_scan_ff/other/unrelated.txt", "unrelated"); + + /* List a root file and a file under sub: sub is descended but its other file + * is not listed, and the whole `other` directory is pruned. */ + create_test_file(FILE_LIST_PATH, "root.txt\nsub/keep.txt\n"); + char err[160]; + FileListSet* set = file_list_load(FILE_LIST_PATH, false, err, sizeof(err)); + EXPECT_NOT_NULL(set); + + ScannerOptions options = {0}; + options.file_list = set; + if (parallel) + options.num_threads = 2; + char** paths = NULL; + int count = 0; + int rc = parallel ? collect_files_parallel(root, &options, &paths, &count) + : collect_files(root, &options, &paths, &count); + EXPECT_EQ_INT(rc, 0); + EXPECT_EQ_INT(count, 2); + EXPECT_TRUE(has_path(paths, count, "root.txt")); + EXPECT_TRUE(has_path(paths, count, "sub/keep.txt")); + EXPECT_FALSE(has_path(paths, count, "sub/skip.bin")); + EXPECT_FALSE(has_path(paths, count, "other/unrelated.txt")); + free_paths(paths, count); + file_list_destroy(set); + remove(FILE_LIST_PATH); + + /* Listing a directory transfers its whole subtree. */ + create_test_file(FILE_LIST_PATH, "sub\n"); + set = file_list_load(FILE_LIST_PATH, false, err, sizeof(err)); + EXPECT_NOT_NULL(set); + options.file_list = set; + rc = parallel ? collect_files_parallel(root, &options, &paths, &count) + : collect_files(root, &options, &paths, &count); + EXPECT_EQ_INT(rc, 0); + EXPECT_EQ_INT(count, 2); + EXPECT_TRUE(has_path(paths, count, "sub/keep.txt")); + EXPECT_TRUE(has_path(paths, count, "sub/skip.bin")); + EXPECT_FALSE(has_path(paths, count, "root.txt")); + EXPECT_FALSE(has_path(paths, count, "other/unrelated.txt")); + free_paths(paths, count); + file_list_destroy(set); + remove(FILE_LIST_PATH); + + unlink("test_scan_ff/root.txt"); + unlink("test_scan_ff/sub/keep.txt"); + unlink("test_scan_ff/sub/skip.bin"); + unlink("test_scan_ff/other/unrelated.txt"); + rmdir(other); + rmdir(sub); + rmdir(root); +} + +/* Filter layer: '-' excludes, first-match-wins ordering with '+', anchored + * rules, and dir-only rules all prune during the scan. */ +static void test_filter_rules(bool parallel) { + const char* root = "test_scan_filter"; + EXPECT_EQ_INT(mkdir(root, 0755), 0); + create_test_file("test_scan_filter/a.txt", "a"); + create_test_file("test_scan_filter/b.tmp", "b"); + create_test_file("test_scan_filter/c.txt", "c"); + + /* - *.tmp excludes only the tmp file; other files remain (default include). */ + const char* exclude_only[] = {"- *.tmp"}; + char err[160]; + FilterRuleList* base = filter_base_build(exclude_only, 1, false, err, sizeof(err)); + EXPECT_NOT_NULL(base); + ScannerOptions options = {0}; + options.base_filters = base; + if (parallel) + options.num_threads = 2; + char** paths = NULL; + int count = 0; + int rc = parallel ? collect_files_parallel(root, &options, &paths, &count) + : collect_files(root, &options, &paths, &count); + EXPECT_EQ_INT(rc, 0); + EXPECT_EQ_INT(count, 2); + EXPECT_TRUE(has_path(paths, count, "a.txt")); + EXPECT_TRUE(has_path(paths, count, "c.txt")); + EXPECT_FALSE(has_path(paths, count, "b.tmp")); + free_paths(paths, count); + filter_rule_list_free(base); + + /* Anchored include then exclude-all: only root-level keep* survives. */ + const char* anchored[] = {"+ /a.txt", "- *"}; + base = filter_base_build(anchored, 2, false, err, sizeof(err)); + EXPECT_NOT_NULL(base); + options.base_filters = base; + rc = parallel ? collect_files_parallel(root, &options, &paths, &count) + : collect_files(root, &options, &paths, &count); + EXPECT_EQ_INT(rc, 0); + EXPECT_EQ_INT(count, 1); + EXPECT_TRUE(has_path(paths, count, "a.txt")); + free_paths(paths, count); + filter_rule_list_free(base); + + unlink("test_scan_filter/a.txt"); + unlink("test_scan_filter/b.tmp"); + unlink("test_scan_filter/c.txt"); + rmdir(root); +} + +/* Anchored dir-only rules prune a whole subtree. */ +static void test_filter_dir_only_and_anchored(bool parallel) { + const char* root = "test_scan_filter_dir"; + const char* sub = "test_scan_filter_dir/sub"; + EXPECT_EQ_INT(mkdir(root, 0755), 0); + EXPECT_EQ_INT(mkdir(sub, 0755), 0); + create_test_file("test_scan_filter_dir/sub/inner.txt", "x"); + create_test_file("test_scan_filter_dir/keep.txt", "keep"); + + const char* rules[] = {"- /sub/"}; + char err[160]; + FilterRuleList* base = filter_base_build(rules, 1, false, err, sizeof(err)); + EXPECT_NOT_NULL(base); + ScannerOptions options = {0}; + options.base_filters = base; + if (parallel) + options.num_threads = 2; + char** paths = NULL; + int count = 0; + int rc = parallel ? collect_files_parallel(root, &options, &paths, &count) + : collect_files(root, &options, &paths, &count); + EXPECT_EQ_INT(rc, 0); + EXPECT_EQ_INT(count, 1); + EXPECT_TRUE(has_path(paths, count, "keep.txt")); + EXPECT_FALSE(has_path(paths, count, "sub/inner.txt")); + free_paths(paths, count); + filter_rule_list_free(base); + + unlink("test_scan_filter_dir/sub/inner.txt"); + unlink("test_scan_filter_dir/keep.txt"); + rmdir(sub); + rmdir(root); +} + +/* -C default CVS excludes prune .git/ directories and *.o files. */ +static void test_cvs_defaults(bool parallel) { + const char* root = "test_scan_cvs"; + const char* git = "test_scan_cvs/.git"; + EXPECT_EQ_INT(mkdir(root, 0755), 0); + EXPECT_EQ_INT(mkdir(git, 0755), 0); + create_test_file("test_scan_cvs/.git/config", "cfg"); + create_test_file("test_scan_cvs/object.o", "o"); + create_test_file("test_scan_cvs/keep.txt", "keep"); + + char err[160]; + FilterRuleList* base = filter_base_build(NULL, 0, true, err, sizeof(err)); + EXPECT_NOT_NULL(base); + ScannerOptions options = {0}; + options.base_filters = base; + if (parallel) + options.num_threads = 2; + char** paths = NULL; + int count = 0; + int rc = parallel ? collect_files_parallel(root, &options, &paths, &count) + : collect_files(root, &options, &paths, &count); + EXPECT_EQ_INT(rc, 0); + EXPECT_EQ_INT(count, 1); + EXPECT_TRUE(has_path(paths, count, "keep.txt")); + EXPECT_FALSE(has_path(paths, count, ".git/config")); + EXPECT_FALSE(has_path(paths, count, "object.o")); + free_paths(paths, count); + filter_rule_list_free(base); + + unlink("test_scan_cvs/.git/config"); + unlink("test_scan_cvs/object.o"); + unlink("test_scan_cvs/keep.txt"); + rmdir(git); + rmdir(root); +} + +/* -F: a .rsync-filter placed in a directory governs its subtree and the file + * itself is never transferred. */ +static void test_per_dir_filter(bool parallel) { + const char* root = "test_scan_perdir"; + const char* sub = "test_scan_perdir/sub"; + EXPECT_EQ_INT(mkdir(root, 0755), 0); + EXPECT_EQ_INT(mkdir(sub, 0755), 0); + create_test_file("test_scan_perdir/drop.tmp", "tmp"); + create_test_file("test_scan_perdir/keep.txt", "keep"); + create_test_file("test_scan_perdir/sub/nested.tmp", "tmp"); + create_test_file("test_scan_perdir/.rsync-filter", "- *.tmp\n"); + + ScannerOptions options = {0}; + options.per_dir_filters = true; + if (parallel) + options.num_threads = 2; + char** paths = NULL; + int count = 0; + int rc = parallel ? collect_files_parallel(root, &options, &paths, &count) + : collect_files(root, &options, &paths, &count); + EXPECT_EQ_INT(rc, 0); + EXPECT_EQ_INT(count, 1); + EXPECT_TRUE(has_path(paths, count, "keep.txt")); + EXPECT_FALSE(has_path(paths, count, "drop.tmp")); + EXPECT_FALSE(has_path(paths, count, "sub/nested.tmp")); + EXPECT_FALSE(has_path(paths, count, ".rsync-filter")); + free_paths(paths, count); + + unlink("test_scan_perdir/drop.tmp"); + unlink("test_scan_perdir/keep.txt"); + unlink("test_scan_perdir/sub/nested.tmp"); + unlink("test_scan_perdir/.rsync-filter"); + rmdir(sub); + rmdir(root); +} + +/* scanner_path_relative maps an on-disk path to its transfer-relative path, + * including the "/" transfer-root edge case (regression: children of "/" used + * to abort the scan because the suffix was mis-read). */ +static void test_scanner_path_relative() { + char* rel = NULL; + + rel = scanner_path_relative("/", "/"); + EXPECT_NOT_NULL(rel); + EXPECT_EQ_STR(rel, ""); + free(rel); + + rel = scanner_path_relative("/", "/etc"); + EXPECT_NOT_NULL(rel); + EXPECT_EQ_STR(rel, "etc"); + free(rel); + + rel = scanner_path_relative("/", "/etc/passwd"); + EXPECT_NOT_NULL(rel); + EXPECT_EQ_STR(rel, "etc/passwd"); + free(rel); + + /* Normal roots: with and without a trailing slash on the root. */ + rel = scanner_path_relative("/tmp/foo", "/tmp/foo"); + EXPECT_NOT_NULL(rel); + EXPECT_EQ_STR(rel, ""); + free(rel); + + rel = scanner_path_relative("/tmp/foo", "/tmp/foo/bar"); + EXPECT_NOT_NULL(rel); + EXPECT_EQ_STR(rel, "bar"); + free(rel); + + rel = scanner_path_relative("/tmp/foo/", "/tmp/foo/bar/baz.txt"); + EXPECT_NOT_NULL(rel); + EXPECT_EQ_STR(rel, "bar/baz.txt"); + free(rel); + + /* A path outside the root maps to NULL. */ + EXPECT_NULL(scanner_path_relative("/tmp/foo", "/tmp")); + EXPECT_NULL(scanner_path_relative("/tmp/foo", "/tmp/foobar")); +} + +/* rsync precedence: a deeper .rsync-filter overrides a shallower one, so an + * inner "+ *.tmp" re-includes what the outer "- *.tmp" excluded. */ +static void test_per_dir_filter_override(bool parallel) { + const char* root = "test_scan_perdir_ovr"; + const char* sub = "test_scan_perdir_ovr/sub"; + EXPECT_EQ_INT(mkdir(root, 0755), 0); + EXPECT_EQ_INT(mkdir(sub, 0755), 0); + create_test_file("test_scan_perdir_ovr/.rsync-filter", "- *.tmp\n"); + create_test_file("test_scan_perdir_ovr/sub/.rsync-filter", "+ *.tmp\n"); + create_test_file("test_scan_perdir_ovr/top.tmp", "x"); + create_test_file("test_scan_perdir_ovr/keep.txt", "keep"); + create_test_file("test_scan_perdir_ovr/sub/inside.tmp", "x"); + + ScannerOptions options = {0}; + options.per_dir_filters = true; + if (parallel) + options.num_threads = 2; + char** paths = NULL; + int count = 0; + int rc = parallel ? collect_files_parallel(root, &options, &paths, &count) + : collect_files(root, &options, &paths, &count); + EXPECT_EQ_INT(rc, 0); + /* top.tmp is still excluded by the root file; inside.tmp is re-included by + * the subdir file; .rsync-filter files are never transferred. */ + EXPECT_EQ_INT(count, 2); + EXPECT_TRUE(has_path(paths, count, "keep.txt")); + EXPECT_TRUE(has_path(paths, count, "sub/inside.tmp")); + EXPECT_FALSE(has_path(paths, count, "top.tmp")); + EXPECT_FALSE(has_path(paths, count, ".rsync-filter")); + EXPECT_FALSE(has_path(paths, count, "sub/.rsync-filter")); + free_paths(paths, count); + + unlink("test_scan_perdir_ovr/top.tmp"); + unlink("test_scan_perdir_ovr/keep.txt"); + unlink("test_scan_perdir_ovr/sub/inside.tmp"); + unlink("test_scan_perdir_ovr/.rsync-filter"); + unlink("test_scan_perdir_ovr/sub/.rsync-filter"); + rmdir(sub); + rmdir(root); +} + +typedef struct { + char rel[512]; + char send[512]; + bool is_dir; +} ScanInfo; + +/* Collect every scanner entry below `root` into `out` (at most `max`), mapping + * paths to their root-relative form and capturing send_path and is_dir. */ +static int collect_scan_info(const char* root, const ScannerOptions* options, ScanInfo out[], + int max) { + DirectoryScanner* scanner = directory_scanner_create_with_options(root, options); + if (!scanner) + return -1; + size_t root_len = strlen(root); + while (root_len > 0 && root[root_len - 1] == '/') + root_len--; + int count = 0; + Chunk* chunk; + while ((chunk = directory_scanner_next(scanner)) != NULL) { + for (int i = 0; i < chunk->element_count && count < max; i++) { + const File* f = chunk->items[i]; + const char* rel = f->path + root_len; + if (*rel == '/') + rel++; + snprintf(out[count].rel, sizeof(out[count].rel), "%s", rel); + snprintf(out[count].send, sizeof(out[count].send), "%s", f->send_path ? f->send_path : ""); + out[count].is_dir = f->is_dir; + count++; + } + chunk_destroy(chunk); + } + bool failed = directory_scanner_failed(scanner); + directory_scanner_destroy(scanner); + return failed ? -1 : count; +} + +static bool scan_info_present(const ScanInfo* infos, int count, const char* rel, bool is_dir, + const char* send) { + for (int i = 0; i < count; i++) { + if (strcmp(infos[i].rel, rel) == 0 && infos[i].is_dir == is_dir && + strcmp(infos[i].send, send ? send : "") == 0) + return true; + } + return false; +} + +/* Parallel variant of collect_scan_info; drains `scanner` fully and destroys + * it. */ +static int collect_scan_info_parallel(ParallelScanner* scanner, const char* root, ScanInfo out[], + int max) { + if (!scanner) + return -1; + size_t root_len = strlen(root); + while (root_len > 0 && root[root_len - 1] == '/') + root_len--; + int count = 0; + Chunk* chunk; + while ((chunk = parallel_scanner_next(scanner)) != NULL) { + for (int i = 0; i < chunk->element_count && count < max; i++) { + const File* f = chunk->items[i]; + const char* rel = f->path + root_len; + if (*rel == '/') + rel++; + snprintf(out[count].rel, sizeof(out[count].rel), "%s", rel); + snprintf(out[count].send, sizeof(out[count].send), "%s", f->send_path ? f->send_path : ""); + out[count].is_dir = f->is_dir; + count++; + } + chunk_destroy(chunk); + } + bool failed = parallel_scanner_failed(scanner); + parallel_scanner_destroy(scanner); + return failed ? -1 : count; +} + +/* -d without --files-from emits exactly the source-root directory (empty) and + * never descends. */ +static void test_dirs_no_descent() { + const char* root = "test_scan_dirs_root"; + EXPECT_EQ_INT(mkdir(root, 0755), 0); + EXPECT_EQ_INT(mkdir("test_scan_dirs_root/sub", 0755), 0); + create_test_file("test_scan_dirs_root/a.txt", "a"); + create_test_file("test_scan_dirs_root/sub/b.txt", "b"); + + ScannerOptions options = {0}; + options.dirs = true; + ScanInfo infos[8]; + int count = collect_scan_info(root, &options, infos, 8); + EXPECT_EQ_INT(count, 1); + EXPECT_TRUE(scan_info_present(infos, count, "", true, NULL)); + EXPECT_FALSE(scan_info_present(infos, count, "a.txt", false, "")); + EXPECT_FALSE(scan_info_present(infos, count, "sub/b.txt", false, "")); + + unlink("test_scan_dirs_root/a.txt"); + unlink("test_scan_dirs_root/sub/b.txt"); + rmdir("test_scan_dirs_root/sub"); + rmdir(root); +} + +/* -d with --files-from transfers exactly the listed directory (empty) and the + * listed file; nothing is descended into. */ +static void test_dirs_files_from() { + const char* root = "test_scan_dirs_ff"; + const char* list_path = "test_scan_dirs_ff.list"; + EXPECT_EQ_INT(mkdir(root, 0755), 0); + EXPECT_EQ_INT(mkdir("test_scan_dirs_ff/sub", 0755), 0); + create_test_file("test_scan_dirs_ff/sub/keep.txt", "keep"); + create_test_file("test_scan_dirs_ff/sub/skip.bin", "skip"); + create_test_file("test_scan_dirs_ff/top.txt", "top"); + + char err[160]; + create_test_file(list_path, "sub\nsub/keep.txt\n"); + FileListSet* set = file_list_load(list_path, false, err, sizeof(err)); + EXPECT_NOT_NULL(set); + + for (int relative = 0; relative <= 1; relative++) { + ScannerOptions options = {0}; + options.dirs = true; + options.file_list = set; + options.relative = relative != 0; + ScanInfo infos[8]; + int count = collect_scan_info(root, &options, infos, 8); + EXPECT_EQ_INT(count, 2); + if (relative) { + EXPECT_TRUE(scan_info_present(infos, count, "sub", true, "sub")); + EXPECT_TRUE(scan_info_present(infos, count, "sub/keep.txt", false, "sub/keep.txt")); + } else { + EXPECT_TRUE(scan_info_present(infos, count, "sub", true, NULL)); + EXPECT_TRUE(scan_info_present(infos, count, "sub/keep.txt", false, NULL)); + } + EXPECT_FALSE(scan_info_present(infos, count, "sub/skip.bin", false, "")); + EXPECT_FALSE(scan_info_present(infos, count, "top.txt", false, "")); + } + file_list_destroy(set); + remove(list_path); + unlink("test_scan_dirs_ff/sub/keep.txt"); + unlink("test_scan_dirs_ff/sub/skip.bin"); + unlink("test_scan_dirs_ff/top.txt"); + rmdir("test_scan_dirs_ff/sub"); + rmdir(root); +} + +/* -R with --files-from (no -d): every file keeps its bare relative path as the + * send_path while the local scan path stays absolute-under-root. */ +static void test_files_from_relative_send_path() { + const char* root = "test_scan_rel_ff"; + const char* list_path = "test_scan_rel_ff.list"; + EXPECT_EQ_INT(mkdir(root, 0755), 0); + EXPECT_EQ_INT(mkdir("test_scan_rel_ff/sub", 0755), 0); + create_test_file("test_scan_rel_ff/root.txt", "root"); + create_test_file("test_scan_rel_ff/sub/keep.txt", "keep"); + + char err[160]; + create_test_file(list_path, "root.txt\nsub/keep.txt\n"); + FileListSet* set = file_list_load(list_path, false, err, sizeof(err)); + EXPECT_NOT_NULL(set); + + for (int parallel = 0; parallel <= 1; parallel++) { + ScannerOptions options = {0}; + options.file_list = set; + options.relative = true; + if (parallel) + options.num_threads = 2; + ScanInfo infos[8]; + int count; + if (parallel) { + ParallelScanner* scanner = parallel_scanner_create_with_options(root, &options, NULL); + EXPECT_NOT_NULL(scanner); + count = collect_scan_info_parallel(scanner, root, infos, 8); + } else { + count = collect_scan_info(root, &options, infos, 8); + } + EXPECT_EQ_INT(count, 2); + EXPECT_TRUE(scan_info_present(infos, count, "root.txt", false, "root.txt")); + EXPECT_TRUE(scan_info_present(infos, count, "sub/keep.txt", false, "sub/keep.txt")); + } + file_list_destroy(set); + remove(list_path); + unlink("test_scan_rel_ff/root.txt"); + unlink("test_scan_rel_ff/sub/keep.txt"); + rmdir("test_scan_rel_ff/sub"); + rmdir(root); +} + +/* P7 Wave D: the recursive scan captures every traversed source directory as an + * is_dir File (metadata, no payload) in the shared dir_entries list, including + * the transfer root and an EMPTY directory. The empty dir is captured even + * though the receiver deliberately never creates it, so its time can still be + * applied when the destination already holds that directory. */ +static void test_scanner_captures_directory_times() { + const char* root = "test_scan_dirtime"; + const char* sub = "test_scan_dirtime/sub"; + const char* empty = "test_scan_dirtime/empty"; + const char* file1 = "test_scan_dirtime/sub/a.txt"; + EXPECT_EQ_INT(mkdir(root, 0755), 0); + EXPECT_EQ_INT(mkdir(sub, 0755), 0); + EXPECT_EQ_INT(mkdir(empty, 0755), 0); + create_test_file(file1, "x"); + + ArrayList* dirs = array_list_create(file_destroy); + EXPECT_NOT_NULL(dirs); + ScannerOptions options = {0}; + options.use_metadata = true; + options.capture_dir_times = true; + options.dir_entries = dirs; + DirectoryScanner* scanner = directory_scanner_create_with_options(root, &options); + EXPECT_NOT_NULL(scanner); + Chunk* chunk; + while ((chunk = directory_scanner_next(scanner)) != NULL) + chunk_destroy(chunk); + EXPECT_FALSE(directory_scanner_failed(scanner)); + + int found_root = 0; + int found_sub = 0; + int found_empty = 0; + for (int i = 0; i < dirs->size; i++) { + const File* file = (const File*)dirs->items[i]; + EXPECT_TRUE(file->is_dir); + EXPECT_NOT_NULL(file->metadata); + if (strcmp(file->path, root) == 0) + found_root = 1; + if (strcmp(file->path, sub) == 0) + found_sub = 1; + if (strcmp(file->path, empty) == 0) + found_empty = 1; + } + EXPECT_TRUE(found_root); + EXPECT_TRUE(found_sub); + EXPECT_TRUE(found_empty); + + directory_scanner_destroy(scanner); + array_list_delete(dirs); + unlink(file1); + rmdir(empty); + rmdir(sub); + rmdir(root); +} + void test_scanner() { test_scanner_single_file(); test_scanner_multiple_files(); @@ -399,4 +1331,26 @@ void test_scanner() { test_scanner_size_range(); test_scanner_mixed_patterns(); test_scanner_no_patterns(); + test_parallel_scanner_root_chunks_without_workers(); + test_scanner_one_file_system_decision(); + test_scanner_one_file_system_same_device(); + test_parallel_scanner_one_file_system_same_device(); + test_scanner_one_file_system_cross_device(); + test_files_from_subset(false); + test_files_from_subset(true); + test_filter_rules(false); + test_filter_rules(true); + test_filter_dir_only_and_anchored(false); + test_filter_dir_only_and_anchored(true); + test_cvs_defaults(false); + test_cvs_defaults(true); + test_per_dir_filter(false); + test_per_dir_filter(true); + test_scanner_path_relative(); + test_per_dir_filter_override(false); + test_per_dir_filter_override(true); + test_dirs_no_descent(); + test_dirs_files_from(); + test_files_from_relative_send_path(); + test_scanner_captures_directory_times(); } diff --git a/tests/test_server.c b/tests/test_server.c index 9226e00..82b842e 100644 --- a/tests/test_server.c +++ b/tests/test_server.c @@ -1,21 +1,22 @@ #include "test_server.h" #include "config.h" +#include "delta.h" #include "file.h" +#include "log.h" #include "protocol.h" #include "test_utils.h" #include "utils.h" +#include #include #include #include #include +#include #include +#include #include -/* Include server.c but rename main to avoid conflict with test runner's main */ -#define main server_main_ -#define FASTSYNC_SERVER_AS_LIB -#include "server.c" -#undef main +#include "receiver.h" /* Test receive_files with immediate FINISHED status */ static void test_receive_files_finished() { @@ -36,7 +37,7 @@ static void test_receive_files_finished() { /* Child: use p[0] for both read and write */ close(p[1]); io_set_fds(p[0], p[0]); - int ret = receive_files(cfg, p[0]); + int ret = receiver_receive_files(cfg, p[0]); close(p[0]); config_delete(cfg); _exit(ret == 0 ? 0 : 1); @@ -88,7 +89,7 @@ static void test_receive_files_single_file() { /* Child: use p[0] for both read and write */ close(p[1]); io_set_fds(p[0], p[0]); - int ret = receive_files(cfg, p[0]); + int ret = receiver_receive_files(cfg, p[0]); close(p[0]); config_delete(cfg); _exit(ret == 0 ? 0 : 1); @@ -149,7 +150,7 @@ static void test_receive_files_abort() { if (pid == 0) { close(p[1]); io_set_fds(p[0], p[0]); - int ret = receive_files(cfg, p[0]); + int ret = receiver_receive_files(cfg, p[0]); close(p[0]); config_delete(cfg); /* Should return -1 on abort */ @@ -171,10 +172,611 @@ static void test_receive_files_abort() { } } +static void test_receive_manifest_rejects_traversal() { + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + cfg->receive_root_directory = str_dup("/tmp/dst"); + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + EXPECT_TRUE(send_int(p[1], 1)); + EXPECT_TRUE(send_str(p[1], "../outside")); + EXPECT_NULL(receive_manifest_entries(p[0])); + Status status; + EXPECT_TRUE(receive_status(p[1], &status)); + EXPECT_EQ_INT(status, STATUS_ERROR); + close(p[0]); + close(p[1]); + config_delete(cfg); +} + +static void test_receive_incremental_check_rejects_invalid_nanoseconds() { + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + cfg->receive_root_directory = str_dup("/tmp/dst"); + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + EXPECT_TRUE(send_str(p[1], "file.txt")); + unsigned long long size = 0; + long long mtime = 100; + long long mtime_nsec = 1000000000LL; + EXPECT_TRUE(send_n_data(p[1], &size, sizeof(size))); + EXPECT_TRUE(send_n_data(p[1], &mtime, sizeof(mtime))); + EXPECT_TRUE(send_n_data(p[1], &mtime_nsec, sizeof(mtime_nsec))); + + bool skipped = false; + EXPECT_NULL(receive_incremental_check(p[0], cfg, &skipped)); + Status status; + EXPECT_TRUE(receive_status(p[1], &status)); + EXPECT_EQ_INT(status, STATUS_ERROR); + EXPECT_FALSE(skipped); + + close(p[0]); + close(p[1]); + config_delete(cfg); +} + +static char* make_check_root(const char* tag) { + char tmpl[128]; + snprintf(tmpl, sizeof(tmpl), "/tmp/fastsync_%s_XXXXXX", tag); + char* path = str_dup(tmpl); + if (!path) + return NULL; + if (!mkdtemp(path)) { + free(path); + return NULL; + } + return path; +} + +static void write_check_file(const char* dir, const char* name, const char* content) { + /* Sized so a caller that passes a PATH_MAX-bounded `dir` (e.g. one of the + test's own char[1024] stack buffers) still provably fits with the joined + name, keeping -Werror=format-truncation quiet. */ + char path[4096]; + snprintf(path, sizeof(path), "%s/%s", dir, name); + int fd = open(path, O_WRONLY | O_CREAT | O_TRUNC, 0644); + if (fd >= 0) { + size_t len = strlen(content); + if (write(fd, content, len) != (ssize_t)len) { + /* intentionally ignored in tests */ + } + close(fd); + } +} + +/* Issue #255: a same-size/mtime match is decided from metadata alone, so the + receiver answers STATUS_OK (skip) and never asks for a data body. */ +static void test_incremental_check_quick_skip_by_mtime() { + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + char* root = make_check_root("qskip"); + EXPECT_NOT_NULL(root); + cfg->receive_root_directory = str_dup(root); + write_check_file(root, "file.txt", "0123456789abcdef"); + + char path[1024]; + snprintf(path, sizeof(path), "%s/file.txt", root); + struct stat st; + EXPECT_EQ_INT(stat(path, &st), 0); + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + alarm(30); + close(p[1]); + io_set_fds(p[0], p[0]); + bool skipped = false; + File* file = receive_incremental_check(p[0], cfg, &skipped); + bool ok = file == NULL && skipped; + file_destroy(file); + config_delete(cfg); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + EXPECT_TRUE(send_str(p[1], "file.txt")); + unsigned long long size = (unsigned long long)st.st_size; + long long mtime = (long long)st.st_mtime; + long long mtime_nsec = 0; +#ifdef __linux__ + mtime_nsec = (long long)st.st_mtim.tv_nsec; +#endif + EXPECT_TRUE(send_n_data(p[1], &size, sizeof(size))); + EXPECT_TRUE(send_n_data(p[1], &mtime, sizeof(mtime))); + EXPECT_TRUE(send_n_data(p[1], &mtime_nsec, sizeof(mtime_nsec))); + Status s; + EXPECT_TRUE(receive_status(p[1], &s)); + EXPECT_EQ_INT(s, STATUS_OK); + + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(cfg); + unlink(path); + rmdir(root); + free(root); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +/* Issue #255: a size mismatch cannot be a skip, so the receiver answers + STATUS_NEXT and consumes the full data body that follows. */ +static void test_incremental_check_size_mismatch_full_transfer() { + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + char* root = make_check_root("qnext"); + EXPECT_NOT_NULL(root); + cfg->receive_root_directory = str_dup(root); + write_check_file(root, "file.txt", "0123456789abcdef"); + + char path[1024]; + snprintf(path, sizeof(path), "%s/file.txt", root); + struct stat st; + EXPECT_EQ_INT(stat(path, &st), 0); + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + alarm(30); + close(p[1]); + io_set_fds(p[0], p[0]); + bool skipped = false; + File* file = receive_incremental_check(p[0], cfg, &skipped); + bool ok = file != NULL && !skipped && file->path != NULL && strcmp(file->path, "file.txt") == 0; + file_destroy(file); + config_delete(cfg); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + EXPECT_TRUE(send_str(p[1], "file.txt")); + unsigned long long size = (unsigned long long)st.st_size + 1; + long long mtime = (long long)st.st_mtime; + long long mtime_nsec = 0; +#ifdef __linux__ + mtime_nsec = (long long)st.st_mtim.tv_nsec; +#endif + EXPECT_TRUE(send_n_data(p[1], &size, sizeof(size))); + EXPECT_TRUE(send_n_data(p[1], &mtime, sizeof(mtime))); + EXPECT_TRUE(send_n_data(p[1], &mtime_nsec, sizeof(mtime_nsec))); + Status s; + EXPECT_TRUE(receive_status(p[1], &s)); + EXPECT_EQ_INT(s, STATUS_NEXT); + + Data* body = data_create_reserve(8); + EXPECT_NOT_NULL(body); + body->data = malloc(8); + EXPECT_NOT_NULL(body->data); + memcpy(body->data, "replaced", 8); + body->size = 8; + EXPECT_TRUE(send_data(p[1], body)); + data_destroy(body); + + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(cfg); + unlink(path); + rmdir(root); + free(root); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +/* Issue #256: when a received delta claims a result above the whole-file cap, + receive_delta_file must mark the operation failed so the caller aborts with + STATUS_ERROR instead of emitting STATUS_NEXT and waiting for a body that + never arrives. */ +static void test_incremental_check_delta_oversize_reports_failure() { + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + char* root = make_check_root("qdelta"); + EXPECT_NOT_NULL(root); + cfg->receive_root_directory = str_dup(root); + cfg->use_delta = true; + + char content[20000]; + memset(content, 'a', sizeof(content)); + content[sizeof(content) - 1] = '\0'; + write_check_file(root, "file.txt", content); + + char path[1024]; + snprintf(path, sizeof(path), "%s/file.txt", root); + struct stat st; + EXPECT_EQ_INT(stat(path, &st), 0); + EXPECT_EQ_INT((int)st.st_size, 19999); + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + alarm(30); + close(p[1]); + io_set_fds(p[0], p[0]); + bool skipped = false; + File* file = receive_incremental_check(p[0], cfg, &skipped); + bool ok = file == NULL && !skipped; + if (ok) + send_status(p[0], STATUS_ERROR); /* mirror the server error path */ + file_destroy(file); + config_delete(cfg); + close(p[0]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + EXPECT_TRUE(send_str(p[1], "file.txt")); + unsigned long long size = (unsigned long long)st.st_size; + long long mtime = 1; /* different from the file mtime: force a transfer */ + long long mtime_nsec = 0; + EXPECT_TRUE(send_n_data(p[1], &size, sizeof(size))); + EXPECT_TRUE(send_n_data(p[1], &mtime, sizeof(mtime))); + EXPECT_TRUE(send_n_data(p[1], &mtime_nsec, sizeof(mtime_nsec))); + + Status s; + EXPECT_TRUE(receive_status(p[1], &s)); + EXPECT_EQ_INT(s, STATUS_DELTA_SIGNATURE); + Data* sig_data = receive_data(p[1]); + EXPECT_NOT_NULL(sig_data); + DeltaSignature* sig = delta_signature_deserialize(sig_data); + EXPECT_NOT_NULL(sig); + delta_signature_destroy(sig); + data_destroy(sig_data); + + /* Send a delta whose claimed output size exceeds the whole-file cap. */ + Delta delta; + memset(&delta, 0, sizeof(delta)); + delta.new_file_size = MAX_RECEIVE_WHOLE_FILE_SIZE + 1; + Data* bogus = delta_serialize(&delta); + EXPECT_NOT_NULL(bogus); + EXPECT_TRUE(send_status(p[1], STATUS_DELTA_DATA)); + EXPECT_TRUE(bogus != NULL && send_data(p[1], bogus)); + data_destroy(bogus); + + /* The receiver must answer with an error, never with STATUS_NEXT. */ + EXPECT_TRUE(receive_status(p[1], &s)); + EXPECT_EQ_INT(s, STATUS_ERROR); + + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(cfg); + unlink(path); + rmdir(root); + free(root); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +/* Late-timing keep-set leak guard: a manifest parked by the commit path must + be freed on every error exit, never leaked. These tests drive + receiver_process_pending() through an error AFTER the manifest was parked and + are exercised under ASan/valgrind to prove the list is released. */ + +static Config* make_late_delete_config(const char* root) { + Config* cfg = config_create(); + if (!cfg) + return NULL; + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup(root); + cfg->use_delete = true; + cfg->delete_after = true; + return cfg; +} + +static int run_pending_receiver(Config* cfg, int fd, DeleteManifest** pending) { + ReceiverSink sink = {0}; + return receiver_process_pending(cfg, fd, &sink, pending); +} + +static void test_late_manifest_abort_frees_keepset() { + Config* cfg = make_late_delete_config("/tmp/fastsync_late_abort"); + EXPECT_NOT_NULL(cfg); + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + EXPECT_TRUE(send_status(p[1], STATUS_MANIFEST)); + EXPECT_TRUE(send_int(p[1], 1)); + EXPECT_TRUE(send_str(p[1], "keep.txt")); + EXPECT_TRUE(send_int(p[1], 0)); /* protected-prefix section is empty */ + EXPECT_TRUE(send_int(p[1], 0)); /* missing-args section is empty */ + EXPECT_TRUE(send_status(p[1], STATUS_ABORT)); + + DeleteManifest* pending = NULL; + EXPECT_EQ_INT(run_pending_receiver(cfg, p[0], &pending), -1); + EXPECT_NULL(pending); + + close(p[0]); + close(p[1]); + config_delete(cfg); +} + +static void test_late_manifest_eof_frees_keepset() { + Config* cfg = make_late_delete_config("/tmp/fastsync_late_eof"); + EXPECT_NOT_NULL(cfg); + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + EXPECT_TRUE(send_status(p[1], STATUS_MANIFEST)); + EXPECT_TRUE(send_int(p[1], 1)); + EXPECT_TRUE(send_str(p[1], "keep.txt")); + EXPECT_TRUE(send_int(p[1], 0)); /* protected-prefix section is empty */ + EXPECT_TRUE(send_int(p[1], 0)); /* missing-args section is empty */ + shutdown(p[1], SHUT_WR); + + DeleteManifest* pending = NULL; + EXPECT_EQ_INT(run_pending_receiver(cfg, p[0], &pending), -1); + EXPECT_NULL(pending); + + close(p[0]); + close(p[1]); + config_delete(cfg); +} + +static void test_late_second_manifest_frees_both() { + Config* cfg = make_late_delete_config("/tmp/fastsync_late_second"); + EXPECT_NOT_NULL(cfg); + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + EXPECT_TRUE(send_status(p[1], STATUS_MANIFEST)); + EXPECT_TRUE(send_int(p[1], 1)); + EXPECT_TRUE(send_str(p[1], "first.txt")); + EXPECT_TRUE(send_int(p[1], 0)); /* protected-prefix section is empty */ + EXPECT_TRUE(send_int(p[1], 0)); /* missing-args section is empty */ + EXPECT_TRUE(send_status(p[1], STATUS_MANIFEST)); + EXPECT_TRUE(send_int(p[1], 1)); + EXPECT_TRUE(send_str(p[1], "second.txt")); + EXPECT_TRUE(send_int(p[1], 0)); /* protected-prefix section is empty */ + EXPECT_TRUE(send_int(p[1], 0)); /* missing-args section is empty */ + + DeleteManifest* pending = NULL; + EXPECT_EQ_INT(run_pending_receiver(cfg, p[0], &pending), -1); + EXPECT_NULL(pending); + + close(p[0]); + close(p[1]); + config_delete(cfg); +} + +/* A delete-manifest frame with a third (missing-args) section round-trips: the + receiver keeps all three sections and the missing paths are confined exactly + like the keep-set (a traversal entry in the missing section is rejected). + receive_manifest_entries() reads the counts directly (the leading + STATUS_MANIFEST code is consumed by the caller, so these frames do not send + it). */ +static void test_receive_manifest_three_sections() { + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + cfg->receive_root_directory = str_dup("/tmp/dst"); + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + + EXPECT_TRUE(send_int(p[1], 1)); + EXPECT_TRUE(send_str(p[1], "keep.txt")); + EXPECT_TRUE(send_int(p[1], 1)); + EXPECT_TRUE(send_str(p[1], "protected.txt")); + EXPECT_TRUE(send_int(p[1], 2)); + EXPECT_TRUE(send_str(p[1], "gone.txt")); + EXPECT_TRUE(send_str(p[1], "dir/gone.bin")); + + DeleteManifest* manifest = receive_manifest_entries(p[0]); + EXPECT_NOT_NULL(manifest); + EXPECT_EQ_INT(manifest->keeps->size, 1); + EXPECT_EQ_STR((char*)manifest->keeps->items[0], "keep.txt"); + EXPECT_EQ_INT(manifest->protected->size, 1); + EXPECT_EQ_STR((char*)manifest->protected->items[0], "protected.txt"); + EXPECT_EQ_INT(manifest->missing->size, 2); + EXPECT_EQ_STR((char*)manifest->missing->items[0], "gone.txt"); + EXPECT_EQ_STR((char*)manifest->missing->items[1], "dir/gone.bin"); + delete_manifest_free(manifest); + + /* A traversal entry in the third section is rejected like every other. */ + EXPECT_TRUE(send_int(p[1], 0)); + EXPECT_TRUE(send_int(p[1], 0)); + EXPECT_TRUE(send_int(p[1], 1)); + EXPECT_TRUE(send_str(p[1], "../escape")); + EXPECT_NULL(receive_manifest_entries(p[0])); + Status status; + EXPECT_TRUE(receive_status(p[1], &status)); + EXPECT_EQ_INT(status, STATUS_ERROR); + + close(p[0]); + close(p[1]); + config_delete(cfg); +} + +/* --delete-missing-args exact-path deletions: regular files and empty + directories are removed, a non-empty directory survives without + --force/--delete and is recursively removed with --force or --delete, and a + missing mirror is a no-op. */ +static void test_manifest_delete_missing_args() { + char* root = make_check_root("qmissing"); + EXPECT_NOT_NULL(root); + write_check_file(root, "gone.txt", "stale"); + char empty_dir[1024], full_dir[1024], inner[1024]; + snprintf(empty_dir, sizeof(empty_dir), "%s/empty_dir", root); + snprintf(full_dir, sizeof(full_dir), "%s/full_dir", root); + snprintf(inner, sizeof(inner), "%s/full_dir/inner.txt", root); + EXPECT_EQ_INT(mkdir(empty_dir, 0755), 0); + EXPECT_EQ_INT(mkdir(full_dir, 0755), 0); + write_check_file(full_dir, "inner.txt", "content"); + + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + cfg->receive_root_directory = str_dup(root); + cfg->delete_missing_args = true; + + DeleteManifest* manifest = calloc(1, sizeof(DeleteManifest)); + EXPECT_NOT_NULL(manifest); + manifest->keeps = array_list_create(free); + manifest->protected = array_list_create(free); + manifest->missing = array_list_create(free); + EXPECT_TRUE(array_list_add(manifest->missing, str_dup("gone.txt"))); + EXPECT_TRUE(array_list_add(manifest->missing, str_dup("empty_dir"))); + EXPECT_TRUE(array_list_add(manifest->missing, str_dup("full_dir"))); + EXPECT_TRUE(array_list_add(manifest->missing, str_dup("never_here.txt"))); + /* A deeper entry whose destination parent directory does not exist is a + no-op (nothing to delete), never a failure. */ + EXPECT_TRUE(array_list_add(manifest->missing, str_dup("no_parent_here/gone.txt"))); + + /* Without --delete/--force the non-empty directory survives (rsync parity). */ + EXPECT_TRUE(manifest_delete_missing_args(cfg, manifest)); + char path[1024]; + snprintf(path, sizeof(path), "%s/gone.txt", root); + EXPECT_EQ_INT(access(path, F_OK), -1); + snprintf(path, sizeof(path), "%s/empty_dir", root); + EXPECT_EQ_INT(access(path, F_OK), -1); + snprintf(path, sizeof(path), "%s/full_dir", root); + EXPECT_EQ_INT(access(path, F_OK), 0); + EXPECT_EQ_INT(access(inner, F_OK), 0); + + /* With --force the non-empty directory mirror is removed recursively. */ + cfg->force_delete = true; + EXPECT_TRUE(array_list_add(manifest->missing, str_dup("full_dir"))); + EXPECT_TRUE(manifest_delete_missing_args(cfg, manifest)); + EXPECT_EQ_INT(access(full_dir, F_OK), -1); + snprintf(path, sizeof(path), "%s/no_parent_here", root); + EXPECT_EQ_INT(access(path, F_OK), -1); + + delete_manifest_free(manifest); + config_delete(cfg); + remove(full_dir); + rmdir(empty_dir); + rmdir(root); + free(root); +} + +/* A delete-missing-args manifest parked by the commit path is committed after + STATUS_FINISHED: the mirror that exists is removed, a missing mirror is a + no-op, and unrelated destination content is untouched (no --delete). */ +static void test_receiver_pending_commits_missing_args() { + char* root = make_check_root("qmisscomm"); + EXPECT_NOT_NULL(root); + write_check_file(root, "gone.txt", "stale"); + write_check_file(root, "extra.txt", "unrelated"); + + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup(root); + cfg->delete_missing_args = true; + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + EXPECT_TRUE(send_status(p[1], STATUS_MANIFEST)); + EXPECT_TRUE(send_int(p[1], 0)); /* keep-set empty */ + EXPECT_TRUE(send_int(p[1], 0)); /* protected empty */ + EXPECT_TRUE(send_int(p[1], 2)); + EXPECT_TRUE(send_str(p[1], "gone.txt")); + EXPECT_TRUE(send_str(p[1], "never_here.txt")); + EXPECT_TRUE(send_status(p[1], STATUS_FINISHED)); + + /* NULL pending: the single-threaded commit path deletes at FINISHED. The + sink sends the terminal STATUS_OK success frame. */ + ReceiverSink sink = {.send_success = true}; + EXPECT_EQ_INT(receiver_process_pending(cfg, p[0], &sink, NULL), 0); + Status ack; + EXPECT_TRUE(receive_status(p[1], &ack)); + EXPECT_EQ_INT(ack, STATUS_OK); + + char path[1024]; + snprintf(path, sizeof(path), "%s/gone.txt", root); + EXPECT_EQ_INT(access(path, F_OK), -1); + snprintf(path, sizeof(path), "%s/extra.txt", root); + EXPECT_EQ_INT(access(path, F_OK), 0); + + close(p[0]); + close(p[1]); + config_delete(cfg); + { + /* remove fixtures */ + char pth[1024]; + snprintf(pth, sizeof(pth), "%s/extra.txt", root); + remove(pth); + rmdir(root); + } + free(root); +} + +/* A6: an attacker-controlled file path appearing in a log line must be escaped + so a control byte cannot forge a second log record. The socket special-node + branch logs file->path before touching the filesystem, making it a cheap way + to exercise an escaped site. The captured line must contain the escaped path + (`\#012` for the newline), never the raw control byte. */ +static void test_special_socket_path_log_escaped() { + set_log_level(LOG_LEVEL_WARNING); + log_set_8_bit_output(false); + + FILE* capture = tmpfile(); + EXPECT_NOT_NULL(capture); + log_set_file(capture); + + File* file = file_create("evil\npath"); + EXPECT_NOT_NULL(file); + file->is_special = true; + file->metadata = calloc(1, sizeof(FileMetadata)); + EXPECT_NOT_NULL(file->metadata); + file->metadata->mode = S_IFSOCK | 0644; + + FileSaveResult result = file_save_to_disk_full("/tmp/dst", file, NULL); + EXPECT_EQ_INT(result, FILE_SAVE_SKIPPED); + + fflush(capture); + rewind(capture); + char output[512] = {0}; + size_t length = fread(output, 1, sizeof(output) - 1, capture); + output[length] = '\0'; + + log_set_file(NULL); + fclose(capture); + file_destroy(file); + + EXPECT_NOT_NULL(strstr(output, "socket not recreated: evil\\#012path")); +} + void test_server() { + test_special_socket_path_log_escaped(); if (!is_running_under_valgrind()) { test_receive_files_finished(); test_receive_files_single_file(); test_receive_files_abort(); + test_receive_manifest_rejects_traversal(); + test_receive_incremental_check_rejects_invalid_nanoseconds(); + test_incremental_check_quick_skip_by_mtime(); + test_incremental_check_size_mismatch_full_transfer(); + test_incremental_check_delta_oversize_reports_failure(); + test_late_manifest_abort_frees_keepset(); + test_late_manifest_eof_frees_keepset(); + test_late_second_manifest_frees_both(); + test_receive_manifest_three_sections(); + test_manifest_delete_missing_args(); + test_receiver_pending_commits_missing_args(); } } diff --git a/tests/test_server_cli.c b/tests/test_server_cli.c new file mode 100644 index 0000000..c057ebd --- /dev/null +++ b/tests/test_server_cli.c @@ -0,0 +1,228 @@ +#include "test_server_cli.h" +#include "server_cli.h" +#include "test_utils.h" +#include +#include + +static int parse_ok(const char* const* args, int count, ServerCliOptions* opts) { + char err[512]; + int r = server_cli_parse(count, (char**)args, opts, err, sizeof(err)); + if (r == 0) + return 0; + if (r < 0) + return -1; + return 1; +} + +static void test_server_cli_defaults() { + const char* args[] = {"fastsync-server"}; + ServerCliOptions opts; + EXPECT_EQ_INT(parse_ok(args, 1, &opts), 0); + EXPECT_FALSE(opts.stdio_mode); + EXPECT_FALSE(opts.daemon_mode); + EXPECT_FALSE(opts.no_detach); + EXPECT_FALSE(opts.verbose); + EXPECT_FALSE(opts.use_tls); + EXPECT_EQ_INT(opts.port, 8080); + EXPECT_FALSE(opts.port_set); + EXPECT_EQ_STR(opts.destination_root, "."); + EXPECT_NULL(opts.config_path); + EXPECT_EQ_INT(opts.dparam_count, 0); + EXPECT_EQ_INT(opts.bind_family, AF_UNSPEC); + EXPECT_FALSE(opts.allow_delete); + EXPECT_FALSE(opts.allow_unauthenticated); + EXPECT_FALSE(opts.no_super); + server_cli_options_free(&opts); +} + +/* --no-super is a standalone/SSH operator veto (does not require --daemon): + it forces SUPER_MODE_OFF for every connection and refuses client --copy-as. */ +static void test_server_cli_no_super() { + const char* args[] = {"fastsync-server", "--no-super", "--destination-root", "/srv"}; + ServerCliOptions opts; + EXPECT_EQ_INT(parse_ok(args, 4, &opts), 0); + EXPECT_TRUE(opts.no_super); + EXPECT_EQ_STR(opts.destination_root, "/srv"); + server_cli_options_free(&opts); + + const char* args2[] = {"fastsync-server", "--daemon", "--config=/tmp/x.conf", "--no-super"}; + ServerCliOptions opts2; + EXPECT_EQ_INT(parse_ok(args2, 4, &opts2), 0); + EXPECT_TRUE(opts2.no_super); + EXPECT_TRUE(opts2.daemon_mode); + server_cli_options_free(&opts2); +} + +static void test_server_cli_daemon_flags() { + const char* args[] = {"fastsync-server", "--daemon", "--no-detach", "--allow-unauthenticated"}; + ServerCliOptions opts; + EXPECT_EQ_INT(parse_ok(args, 4, &opts), 0); + EXPECT_TRUE(opts.daemon_mode); + EXPECT_TRUE(opts.no_detach); + EXPECT_TRUE(opts.allow_unauthenticated); + server_cli_options_free(&opts); +} + +static void test_server_cli_config_and_dparam_forms() { + const char* args[] = {"fastsync-server", "--daemon", "--config=/tmp/x.conf", + "--dparam=port=8734", "--dparam", "address=127.0.0.1"}; + ServerCliOptions opts; + EXPECT_EQ_INT(parse_ok(args, 6, &opts), 0); + EXPECT_EQ_STR(opts.config_path, "/tmp/x.conf"); + EXPECT_EQ_INT(opts.dparam_count, 2); + EXPECT_EQ_STR(opts.dparams[0], "port=8734"); + EXPECT_EQ_STR(opts.dparams[1], "address=127.0.0.1"); + + const char* args2[] = {"fastsync-server", "--daemon", "--config", "/tmp/y.conf"}; + ServerCliOptions opts2; + EXPECT_EQ_INT(parse_ok(args2, 4, &opts2), 0); + EXPECT_EQ_STR(opts2.config_path, "/tmp/y.conf"); + server_cli_options_free(&opts); + server_cli_options_free(&opts2); +} + +static void test_server_cli_preserves_existing_flags() { + const char* args[] = {"fastsync-server", + "-p", + "9000", + "--tls", + "--cert", + "/c", + "--key", + "/k", + "--ca", + "/ca", + "--client-cn", + "cn", + "--allow-delete", + "--trust-sender", + "--address", + "127.0.0.1", + "-6", + "--destination-root", + "/srv"}; + ServerCliOptions opts; + EXPECT_EQ_INT(parse_ok(args, 19, &opts), 0); + EXPECT_EQ_INT(opts.port, 9000); + EXPECT_TRUE(opts.port_set); + EXPECT_TRUE(opts.use_tls); + EXPECT_EQ_STR(opts.tls_cert, "/c"); + EXPECT_EQ_STR(opts.tls_key, "/k"); + EXPECT_EQ_STR(opts.tls_ca, "/ca"); + EXPECT_EQ_STR(opts.client_cn, "cn"); + EXPECT_TRUE(opts.allow_delete); + EXPECT_TRUE(opts.trust_sender); + EXPECT_EQ_STR(opts.bind_address, "127.0.0.1"); + EXPECT_EQ_INT(opts.bind_family, AF_INET6); + EXPECT_TRUE(opts.destination_root_set); + EXPECT_EQ_STR(opts.destination_root, "/srv"); + server_cli_options_free(&opts); +} + +static void test_server_cli_conflicts() { + /* server_cli_parse zero-initializes opts (server_cli_options_default) before + * parsing, so `opts` is still safe to pass to server_cli_options_free even + * when every parse below returns -1 on failure. */ + char err[256]; + ServerCliOptions opts; + const char* a1[] = {"s", "--daemon", "--stdio"}; + EXPECT_EQ_INT(server_cli_parse(3, (char**)a1, &opts, err, sizeof(err)), -1); + EXPECT_TRUE(strstr(err, "mutually exclusive") != NULL); + + const char* a2[] = {"s", "--daemon", "--destination-root", "/x"}; + EXPECT_EQ_INT(server_cli_parse(4, (char**)a2, &opts, err, sizeof(err)), -1); + EXPECT_TRUE(strstr(err, "module paths") != NULL); + + const char* a3[] = {"s", "--config", "/x.conf"}; + EXPECT_EQ_INT(server_cli_parse(3, (char**)a3, &opts, err, sizeof(err)), -1); + EXPECT_TRUE(strstr(err, "require --daemon") != NULL); + + const char* a4[] = {"s", "--no-detach"}; + EXPECT_EQ_INT(server_cli_parse(2, (char**)a4, &opts, err, sizeof(err)), -1); + EXPECT_TRUE(strstr(err, "require --daemon") != NULL); + server_cli_options_free(&opts); +} + +static void test_server_cli_invalid() { + char err[256]; + ServerCliOptions opts; + const char* a1[] = {"s", "-p", "notaport"}; + EXPECT_EQ_INT(server_cli_parse(3, (char**)a1, &opts, err, sizeof(err)), -1); + EXPECT_TRUE(strstr(err, "invalid port") != NULL); + + const char* a2[] = {"s", "-4", "-6"}; + EXPECT_EQ_INT(server_cli_parse(3, (char**)a2, &opts, err, sizeof(err)), -1); + EXPECT_TRUE(strstr(err, "mutually exclusive") != NULL); + + const char* a3[] = {"s", "--nope"}; + EXPECT_EQ_INT(server_cli_parse(2, (char**)a3, &opts, err, sizeof(err)), -1); + EXPECT_TRUE(strstr(err, "unknown option") != NULL); + + const char* a4[] = {"s", "--cert"}; + EXPECT_EQ_INT(server_cli_parse(2, (char**)a4, &opts, err, sizeof(err)), -1); + server_cli_options_free(&opts); +} + +static void test_server_cli_password_and_early_input() { + const char* args[] = {"s", + "--daemon", + "--password-file=/etc/fast.pw", + "--early-input", + "/run/secrets", + "--iconv=utf-8"}; + ServerCliOptions opts; + EXPECT_EQ_INT(parse_ok(args, 6, &opts), 0); + EXPECT_EQ_STR(opts.password_file, "/etc/fast.pw"); + EXPECT_EQ_STR(opts.early_input_file, "/run/secrets"); + EXPECT_EQ_STR(opts.iconv_spec, "utf-8"); + + const char* args2[] = { + "s", "--daemon", "--password-file", "/etc/fast.pw", "--early-input=/secrets", + "--iconv", "utf-8,iso-8859-1"}; + ServerCliOptions opts2; + EXPECT_EQ_INT(parse_ok(args2, 7, &opts2), 0); + EXPECT_EQ_STR(opts2.password_file, "/etc/fast.pw"); + EXPECT_EQ_STR(opts2.early_input_file, "/secrets"); + EXPECT_EQ_STR(opts2.iconv_spec, "utf-8,iso-8859-1"); + server_cli_options_free(&opts); + server_cli_options_free(&opts2); +} + +static void test_server_cli_password_requires_daemon() { + char err[256]; + ServerCliOptions opts; + const char* a1[] = {"s", "--password-file", "/etc/fast.pw"}; + EXPECT_EQ_INT(server_cli_parse(3, (char**)a1, &opts, err, sizeof(err)), -1); + EXPECT_TRUE(strstr(err, "require --daemon") != NULL); + + const char* a2[] = {"s", "--early-input", "/secrets"}; + EXPECT_EQ_INT(server_cli_parse(3, (char**)a2, &opts, err, sizeof(err)), -1); + EXPECT_TRUE(strstr(err, "require --daemon") != NULL); + + const char* a3[] = {"s", "--daemon", "--password-file"}; + EXPECT_EQ_INT(server_cli_parse(3, (char**)a3, &opts, err, sizeof(err)), -1); + EXPECT_TRUE(strstr(err, "missing argument") != NULL); + server_cli_options_free(&opts); +} + +static void test_server_cli_help() { + char err[256]; + const char* a1[] = {"s", "--help"}; + ServerCliOptions opts; + EXPECT_EQ_INT(server_cli_parse(2, (char**)a1, &opts, err, sizeof(err)), 1); + EXPECT_TRUE(opts.show_help); + server_cli_options_free(&opts); +} + +void test_server_cli() { + test_server_cli_defaults(); + test_server_cli_daemon_flags(); + test_server_cli_config_and_dparam_forms(); + test_server_cli_preserves_existing_flags(); + test_server_cli_conflicts(); + test_server_cli_invalid(); + test_server_cli_password_and_early_input(); + test_server_cli_password_requires_daemon(); + test_server_cli_no_super(); + test_server_cli_help(); +} diff --git a/tests/test_server_cli.h b/tests/test_server_cli.h new file mode 100644 index 0000000..5023a65 --- /dev/null +++ b/tests/test_server_cli.h @@ -0,0 +1,6 @@ +#ifndef TEST_SERVER_CLI_H +#define TEST_SERVER_CLI_H + +void test_server_cli(); + +#endif \ No newline at end of file diff --git a/tests/test_shared_utils.c b/tests/test_shared_utils.c index 57b1c74..c9da6cf 100644 --- a/tests/test_shared_utils.c +++ b/tests/test_shared_utils.c @@ -1,10 +1,422 @@ #include "test_shared_utils.h" #include "utils.h" +#include "protocol.h" #include "test_utils.h" +#include +#include +#include +#include +#include +#include #include #include +#include +#include +#include +#include + +/* ---- delete-walker tests ---- */ + +static char* make_walk_root(const char* tag) { + char* path = malloc(256); + if (!path) + return NULL; + snprintf(path, 256, "/tmp/fastsync_walk_%s_%d", tag, (int)getpid()); + rmdir(path); + if (mkdir(path, 0755) != 0) { + free(path); + return NULL; + } + return path; +} + +static bool write_file_at(const char* dir, const char* name, const char* content) { + char* path = path_cat(dir, name); + if (!path) + return false; + int fd = open(path, O_WRONLY | O_CREAT | O_TRUNC, 0644); + bool ok = fd >= 0; + if (fd >= 0) { + if (content) { + const char* p = content; + size_t remaining = strlen(content); + while (remaining > 0) { + ssize_t n = write(fd, p, remaining); + if (n <= 0) { + ok = false; + break; + } + p += n; + remaining -= (size_t)n; + } + } + close(fd); + } + free(path); + return ok; +} + +static bool file_exists(const char* dir, const char* name) { + char* path = path_cat(dir, name); + bool exists = path && access(path, F_OK) == 0; + free(path); + return exists; +} + +static bool dir_exists(const char* dir, const char* name) { + char* path = path_cat(dir, name); + struct stat st; + bool exists = path && stat(path, &st) == 0 && S_ISDIR(st.st_mode); + free(path); + return exists; +} + +static int make_subdir(const char* root, const char* name) { + char* path = path_cat(root, name); + int rc = -1; + if (path) { + rc = mkdir(path, 0755); + free(path); + } + return rc; +} + +static void remove_walk_tree(const char* path) { + DIR* dir = opendir(path); + if (!dir) { + rmdir(path); + return; + } + const struct dirent* entry; + while ((entry = readdir(dir)) != NULL) { + if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) + continue; + char* child = path_cat(path, entry->d_name); + if (child) { + struct stat st; + if (lstat(child, &st) == 0 && S_ISDIR(st.st_mode)) + remove_walk_tree(child); + else + unlink(child); + free(child); + } + } + closedir(dir); + rmdir(path); +} + +static ArrayList* make_manifest_strings(const char* const* entries, int count) { + ArrayList* manifest = array_list_create(free); + if (!manifest) + return NULL; + for (int i = 0; i < count; i++) { + char* dup = str_dup(entries[i]); + if (!dup || !array_list_add(manifest, dup)) { + free(dup); + array_list_delete(manifest); + return NULL; + } + } + return manifest; +} + +static void test_walker_removes_extras_keeps_manifest_and_protected() { + char* root = make_walk_root("basic"); + EXPECT_NOT_NULL(root); + EXPECT_TRUE(write_file_at(root, "a.txt", "extra")); + EXPECT_TRUE(write_file_at(root, "keep.txt", "kept")); + EXPECT_EQ_INT(make_subdir(root, "d"), 0); + EXPECT_TRUE(write_file_at(root, "d/e.txt", "extra")); + EXPECT_TRUE(write_file_at(root, "d/k.txt", "kept")); + EXPECT_EQ_INT(make_subdir(root, "prot"), 0); + EXPECT_TRUE(write_file_at(root, "prot/f.txt", "untouched")); + + const char* keeps[] = {"keep.txt", "d/k.txt"}; + ArrayList* manifest = make_manifest_strings(keeps, 2); + EXPECT_NOT_NULL(manifest); + DeleteSkipEntry skip = {"prot", false}; + size_t deleted = 0; + DeleteWalkResult result = delete_extras_limited(root, manifest, 100000, &skip, 1, &deleted); + EXPECT_EQ_INT((int)result, (int)DELETE_WALK_OK); + EXPECT_FALSE(file_exists(root, "a.txt")); + EXPECT_TRUE(file_exists(root, "keep.txt")); + EXPECT_FALSE(file_exists(root, "d/e.txt")); + EXPECT_TRUE(file_exists(root, "d/k.txt")); + EXPECT_TRUE(dir_exists(root, "d")); + EXPECT_TRUE(file_exists(root, "prot/f.txt")); + EXPECT_TRUE(deleted >= 2); + array_list_delete(manifest); + remove_walk_tree(root); + free(root); +} + +static void test_walker_max_delete_exceeded_deletes_nothing() { + char* root = make_walk_root("maxdel"); + EXPECT_NOT_NULL(root); + EXPECT_TRUE(write_file_at(root, "a.txt", "extra")); + EXPECT_TRUE(write_file_at(root, "b.txt", "extra")); + EXPECT_TRUE(write_file_at(root, "c.txt", "extra")); + const char* keeps[1] = {NULL}; + ArrayList* manifest = make_manifest_strings(keeps, 0); + EXPECT_NOT_NULL(manifest); + size_t deleted = 999; + DeleteWalkResult result = delete_extras_limited(root, manifest, 2, NULL, 0, &deleted); + EXPECT_EQ_INT((int)result, (int)DELETE_WALK_LIMIT_EXCEEDED); + EXPECT_EQ_INT((int)deleted, 0); + EXPECT_TRUE(file_exists(root, "a.txt")); + EXPECT_TRUE(file_exists(root, "b.txt")); + EXPECT_TRUE(file_exists(root, "c.txt")); + array_list_delete(manifest); + remove_walk_tree(root); + free(root); +} + +static void test_walker_max_delete_exact_bound_deletes() { + char* root = make_walk_root("maxdel2"); + EXPECT_NOT_NULL(root); + EXPECT_TRUE(write_file_at(root, "a.txt", "extra")); + EXPECT_TRUE(write_file_at(root, "b.txt", "extra")); + const char* keeps[1] = {NULL}; + ArrayList* manifest = make_manifest_strings(keeps, 0); + EXPECT_NOT_NULL(manifest); + size_t deleted = 0; + DeleteWalkResult result = delete_extras_limited(root, manifest, 2, NULL, 0, &deleted); + EXPECT_EQ_INT((int)result, (int)DELETE_WALK_OK); + EXPECT_EQ_INT((int)deleted, 2); + EXPECT_FALSE(file_exists(root, "a.txt")); + EXPECT_FALSE(file_exists(root, "b.txt")); + array_list_delete(manifest); + remove_walk_tree(root); + free(root); +} + +static void test_walker_unlimited_deletes_all() { + char* root = make_walk_root("unlim"); + EXPECT_NOT_NULL(root); + EXPECT_TRUE(write_file_at(root, "a.txt", "extra")); + EXPECT_TRUE(write_file_at(root, "b.txt", "extra")); + EXPECT_EQ_INT(make_subdir(root, "emptydir"), 0); + const char* keeps[1] = {NULL}; + ArrayList* manifest = make_manifest_strings(keeps, 0); + EXPECT_NOT_NULL(manifest); + EXPECT_TRUE(delete_extras(root, manifest)); + EXPECT_FALSE(file_exists(root, "a.txt")); + EXPECT_FALSE(file_exists(root, "b.txt")); + EXPECT_FALSE(dir_exists(root, "emptydir")); + array_list_delete(manifest); + remove_walk_tree(root); + free(root); +} + +/* The 100000-entry server hard bound (MAX_SERVER_DELETE_COUNT, which this test + exercises through a literal to avoid reaching into file_receive.c) is also + all-or-nothing: a destination holding more extras than the bound must be left + completely untouched. Skipped under valgrind: 100k file creations would be + far too slow under instrumentation. */ +static void test_walker_hard_bound_all_or_nothing() { + if (is_running_under_valgrind()) + return; + enum { HARD_BOUND = 100000 }; + char* root = make_walk_root("hardbound"); + EXPECT_NOT_NULL(root); + int rootfd = open(root, O_RDONLY | O_DIRECTORY | O_CLOEXEC); + EXPECT_TRUE(rootfd >= 0); + bool created = true; + for (int i = 0; created && i < HARD_BOUND + 1; i++) { + char name[32]; + snprintf(name, sizeof(name), "f%d", i); + int fd = openat(rootfd, name, O_WRONLY | O_CREAT | O_TRUNC, 0644); + if (fd < 0) + created = false; + else + close(fd); + } + EXPECT_TRUE(created); + const char* keeps[1] = {NULL}; + ArrayList* manifest = make_manifest_strings(keeps, 0); + EXPECT_NOT_NULL(manifest); + size_t deleted = 999; + DeleteWalkResult result = delete_extras_limited(root, manifest, HARD_BOUND, NULL, 0, &deleted); + EXPECT_EQ_INT((int)result, (int)DELETE_WALK_LIMIT_EXCEEDED); + EXPECT_EQ_INT((int)deleted, 0); + EXPECT_TRUE(file_exists(root, "f0")); + EXPECT_TRUE(file_exists(root, "f100000")); + array_list_delete(manifest); + /* Fast cleanup: unlink every created name through the still-open root fd. */ + if (rootfd >= 0) { + for (int i = 0; i < HARD_BOUND + 1; i++) { + char name[32]; + snprintf(name, sizeof(name), "f%d", i); + (void)unlinkat(rootfd, name, 0); + } + close(rootfd); + } + rmdir(root); + free(root); +} + +typedef struct { + bool eight_bit_output; + const char* expected; + int failed; +} EscapeThreadArgs; + +static int escape_thread(void* arg) { + EscapeThreadArgs* args = arg; + for (int i = 0; i < 1000; i++) { + char* escaped = output_escape("x\xc3\xa9\n", args->eight_bit_output); + if (!escaped || strcmp(escaped, args->expected) != 0) + args->failed = 1; + free(escaped); + } + return 0; +} + +/* A7-3/S1 transport classification: the daemon auth gate and the client + credential rule both key off these helpers, so cover the exact accepted + forms plus the negative cases. */ +static void test_loopback_helpers() { + /* Host strings. */ + EXPECT_TRUE(utils_host_is_loopback("localhost")); + EXPECT_TRUE(utils_host_is_loopback("127.0.0.1")); + EXPECT_TRUE(utils_host_is_loopback("127.255.255.254")); + EXPECT_TRUE(utils_host_is_loopback("127.0.0.0")); + EXPECT_TRUE(utils_host_is_loopback("::1")); + EXPECT_TRUE(utils_host_is_loopback("[::1]")); + EXPECT_FALSE(utils_host_is_loopback("128.0.0.1")); + EXPECT_FALSE(utils_host_is_loopback("10.0.0.1")); + EXPECT_FALSE(utils_host_is_loopback("0.0.0.0")); + EXPECT_FALSE(utils_host_is_loopback("example.com")); + EXPECT_FALSE(utils_host_is_loopback("")); + EXPECT_FALSE(utils_host_is_loopback(NULL)); + + /* Raw sockaddr classification. */ + struct sockaddr_in v4; + memset(&v4, 0, sizeof(v4)); + v4.sin_family = AF_INET; + EXPECT_TRUE(inet_pton(AF_INET, "127.0.0.1", &v4.sin_addr) == 1); + EXPECT_TRUE(utils_sockaddr_is_loopback((const struct sockaddr*)&v4)); + EXPECT_TRUE(inet_pton(AF_INET, "127.5.5.5", &v4.sin_addr) == 1); + EXPECT_TRUE(utils_sockaddr_is_loopback((const struct sockaddr*)&v4)); + EXPECT_TRUE(inet_pton(AF_INET, "128.0.0.1", &v4.sin_addr) == 1); + EXPECT_FALSE(utils_sockaddr_is_loopback((const struct sockaddr*)&v4)); + + struct sockaddr_in6 v6; + memset(&v6, 0, sizeof(v6)); + v6.sin6_family = AF_INET6; + EXPECT_TRUE(inet_pton(AF_INET6, "::1", &v6.sin6_addr) == 1); + EXPECT_TRUE(utils_sockaddr_is_loopback((const struct sockaddr*)&v6)); + EXPECT_TRUE(inet_pton(AF_INET6, "::ffff:127.0.0.1", &v6.sin6_addr) == 1); + EXPECT_TRUE(utils_sockaddr_is_loopback((const struct sockaddr*)&v6)); + EXPECT_TRUE(inet_pton(AF_INET6, "::ffff:127.255.255.254", &v6.sin6_addr) == 1); + EXPECT_TRUE(utils_sockaddr_is_loopback((const struct sockaddr*)&v6)); + EXPECT_TRUE(inet_pton(AF_INET6, "::ffff:10.0.0.1", &v6.sin6_addr) == 1); + EXPECT_FALSE(utils_sockaddr_is_loopback((const struct sockaddr*)&v6)); + + EXPECT_FALSE(utils_sockaddr_is_loopback(NULL)); + + /* A pipe has no socket peer: getpeername fails with ENOTSOCK. The helper is + fail-closed, so an unprovable channel is NOT local (daemon auth modules are + daemon-only and never run over the --stdio pipe). */ + int pipe_fds[2]; + EXPECT_EQ_INT(pipe(pipe_fds), 0); + EXPECT_FALSE(utils_fd_peer_is_local(pipe_fds[0])); + close(pipe_fds[0]); + close(pipe_fds[1]); + EXPECT_FALSE(utils_fd_peer_is_local(-1)); + + /* A connected AF_UNIX socketpair is a socket, but its peer is not a loopback + IP address, so it is not local either. */ + int pair_fds[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, pair_fds), 0); + EXPECT_FALSE(utils_fd_peer_is_local(pair_fds[0])); + close(pair_fds[0]); + close(pair_fds[1]); + + /* A real loopback TCP peer is local. */ + int listener = socket(AF_INET, SOCK_STREAM, 0); + EXPECT_TRUE(listener >= 0); + struct sockaddr_in bind_addr; + memset(&bind_addr, 0, sizeof(bind_addr)); + bind_addr.sin_family = AF_INET; + bind_addr.sin_addr.s_addr = htonl(INADDR_LOOPBACK); + bind_addr.sin_port = 0; + EXPECT_EQ_INT(bind(listener, (const struct sockaddr*)&bind_addr, sizeof(bind_addr)), 0); + EXPECT_EQ_INT(listen(listener, 1), 0); + socklen_t addr_len = sizeof(bind_addr); + EXPECT_EQ_INT(getsockname(listener, (struct sockaddr*)&bind_addr, &addr_len), 0); + int dialer = socket(AF_INET, SOCK_STREAM, 0); + EXPECT_TRUE(dialer >= 0); + EXPECT_EQ_INT(connect(dialer, (const struct sockaddr*)&bind_addr, sizeof(bind_addr)), 0); + int accepted = accept(listener, NULL, NULL); + EXPECT_TRUE(accepted >= 0); + EXPECT_TRUE(utils_fd_peer_is_local(accepted)); + close(accepted); + close(dialer); + close(listener); +} void test_shared_utils() { + test_walker_removes_extras_keeps_manifest_and_protected(); + test_walker_max_delete_exceeded_deletes_nothing(); + test_walker_max_delete_exact_bound_deletes(); + test_walker_unlimited_deletes_all(); + test_walker_hard_bound_all_or_nothing(); + test_loopback_helpers(); + + /* --append / --append-verify tail-resume math: a resume is eligible only for + a shorter existing destination, and the tail length is then the difference. */ + EXPECT_TRUE(append_resume_eligible(0, 10)); + EXPECT_TRUE(append_resume_eligible(7, 10)); + EXPECT_FALSE(append_resume_eligible(10, 10)); + EXPECT_FALSE(append_resume_eligible(11, 10)); + + unsigned long long tail; + EXPECT_TRUE(append_tail_length(0, 10, &tail)); + EXPECT_EQ_INT((int)tail, 10); + EXPECT_TRUE(append_tail_length(7, 10, &tail)); + EXPECT_EQ_INT((int)tail, 3); + EXPECT_FALSE(append_tail_length(10, 10, &tail)); + EXPECT_FALSE(append_tail_length(11, 10, &tail)); + EXPECT_FALSE(append_tail_length(7, 10, NULL)); + + char formatted[32]; + EXPECT_TRUE(format_human_bytes(0, formatted, sizeof(formatted))); + EXPECT_EQ_STR(formatted, "0 B"); + EXPECT_TRUE(format_human_bytes(1024, formatted, sizeof(formatted))); + EXPECT_EQ_STR(formatted, "1.0 KB"); + EXPECT_TRUE(format_human_bytes(1536 * 1024, formatted, sizeof(formatted))); + EXPECT_EQ_STR(formatted, "1.5 MB"); + EXPECT_FALSE(format_human_bytes(1024, formatted, 4)); + + char high_bit[] = {'a', (char)0xc3, (char)0xa9, '\n', '\0'}; + char* escaped = output_escape(high_bit, false); + EXPECT_EQ_STR(escaped, "a\\#303\\#251\\#012"); + free(escaped); + escaped = output_escape(high_bit, true); + EXPECT_EQ_STR(escaped, "a\xc3\xa9\\#012"); + free(escaped); + + ProtocolSession safe_session; + ProtocolSession eight_bit_session; + protocol_session_init(&safe_session, -1, -1); + protocol_session_init(&eight_bit_session, -1, -1); + protocol_session_set_8_bit_output(&safe_session, false); + protocol_session_set_8_bit_output(&eight_bit_session, true); + EXPECT_FALSE(safe_session.eight_bit_output); + EXPECT_TRUE(eight_bit_session.eight_bit_output); + + EscapeThreadArgs safe_args = {false, "x\\#303\\#251\\#012", 0}; + EscapeThreadArgs eight_bit_args = {true, "x\xc3\xa9\\#012", 0}; + thrd_t safe_thread; + thrd_t eight_bit_thread; + EXPECT_EQ_INT(thrd_create(&safe_thread, escape_thread, &safe_args), thrd_success); + EXPECT_EQ_INT(thrd_create(&eight_bit_thread, escape_thread, &eight_bit_args), thrd_success); + EXPECT_EQ_INT(thrd_join(safe_thread, NULL), thrd_success); + EXPECT_EQ_INT(thrd_join(eight_bit_thread, NULL), thrd_success); + EXPECT_FALSE(safe_args.failed); + EXPECT_FALSE(eight_bit_args.failed); + // Test str_dup const char* dup_null = str_dup(NULL); EXPECT_NULL(dup_null); diff --git a/tests/test_stop.c b/tests/test_stop.c new file mode 100644 index 0000000..4fc71bf --- /dev/null +++ b/tests/test_stop.c @@ -0,0 +1,150 @@ +#include "test_stop.h" +#include "stop_condition.h" +#include "test_utils.h" +#include +#include + +static void test_stop_after_parse_valid() { + int minutes = 0; + EXPECT_TRUE(stop_parse_after_minutes("5", &minutes)); + EXPECT_EQ_INT(minutes, 5); + EXPECT_TRUE(stop_parse_after_minutes("1", &minutes)); + EXPECT_EQ_INT(minutes, 1); + EXPECT_TRUE(stop_parse_after_minutes("1440", &minutes)); + EXPECT_EQ_INT(minutes, 1440); + EXPECT_TRUE(stop_parse_after_minutes("2147483647", &minutes)); + EXPECT_EQ_INT(minutes, INT_MAX); +} + +static void test_stop_after_parse_invalid() { + int minutes = 0; + EXPECT_FALSE(stop_parse_after_minutes("0", &minutes)); + EXPECT_FALSE(stop_parse_after_minutes("-1", &minutes)); + EXPECT_FALSE(stop_parse_after_minutes("abc", &minutes)); + EXPECT_FALSE(stop_parse_after_minutes("", &minutes)); + EXPECT_FALSE(stop_parse_after_minutes("5x", &minutes)); + EXPECT_FALSE(stop_parse_after_minutes("1.5", &minutes)); + EXPECT_FALSE(stop_parse_after_minutes(" 5", &minutes)); + EXPECT_FALSE(stop_parse_after_minutes(" 5 ", &minutes)); + EXPECT_FALSE(stop_parse_after_minutes("+5", &minutes)); + EXPECT_FALSE(stop_parse_after_minutes("2147483648", &minutes)); + EXPECT_FALSE(stop_parse_after_minutes(NULL, &minutes)); +} + +static void test_stop_at_parse_hhmm() { + time_t now = 1700000000; + time_t deadline = 0; + + EXPECT_TRUE(stop_parse_at_time("12:30", now, &deadline)); + struct tm t; + EXPECT_NOT_NULL(localtime_r(&deadline, &t)); + EXPECT_EQ_INT(t.tm_hour, 12); + EXPECT_EQ_INT(t.tm_min, 30); + EXPECT_EQ_INT(t.tm_sec, 0); + + EXPECT_TRUE(stop_parse_at_time("12:30:59", now, &deadline)); + EXPECT_NOT_NULL(localtime_r(&deadline, &t)); + EXPECT_EQ_INT(t.tm_hour, 12); + EXPECT_EQ_INT(t.tm_min, 30); + EXPECT_EQ_INT(t.tm_sec, 59); + + EXPECT_TRUE(stop_parse_at_time("00:00", now, &deadline)); + EXPECT_NOT_NULL(localtime_r(&deadline, &t)); + EXPECT_EQ_INT(t.tm_hour, 0); + EXPECT_EQ_INT(t.tm_min, 0); + EXPECT_EQ_INT(t.tm_sec, 0); +} + +static void test_stop_at_parse_now_plus() { + time_t now = 1700000000; + time_t deadline = 0; + + EXPECT_TRUE(stop_parse_at_time("now+90s", now, &deadline)); + EXPECT_EQ_INT(deadline, now + 90); + EXPECT_TRUE(stop_parse_at_time("now+5m", now, &deadline)); + EXPECT_EQ_INT(deadline, now + 300); + EXPECT_TRUE(stop_parse_at_time("now+2h", now, &deadline)); + EXPECT_EQ_INT(deadline, now + 7200); + EXPECT_TRUE(stop_parse_at_time("now+1d", now, &deadline)); + EXPECT_EQ_INT(deadline, now + 86400); + EXPECT_TRUE(stop_parse_at_time("now+0s", now, &deadline)); + EXPECT_EQ_INT(deadline, now); +} + +static void test_stop_at_parse_invalid() { + time_t now = 1700000000; + time_t deadline = 0; + EXPECT_FALSE(stop_parse_at_time("12", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("12:3", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("1234", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("12:30:5", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("12:30:5x", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("24:00", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("12:60", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("12:30:61", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("12;00", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("now", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("now+", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("now+5", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("now+5x", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("now-5m", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("now+1w", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("now+ 5s", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("now++5s", now, &deadline)); + /* Signed overflow of the destination deadline must be rejected, not wrap. */ + EXPECT_FALSE(stop_parse_at_time("now+9223372036854775807s", now, &deadline)); + /* 10^15 days is well beyond LONG_MAX/86400, so the amount itself is rejected. */ + EXPECT_FALSE(stop_parse_at_time("now+1000000000000000d", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("abc", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time("", now, &deadline)); + EXPECT_FALSE(stop_parse_at_time(NULL, now, &deadline)); +} + +static void test_stop_deadline_latency() { + struct timespec now; + EXPECT_EQ_INT(clock_gettime(CLOCK_MONOTONIC, &now), 0); + + StopCondition future = stop_condition_make(true, 60, false, 0, now); + EXPECT_TRUE(future.has_monotonic); + EXPECT_EQ_INT(future.monotonic_deadline.tv_sec, now.tv_sec + 3600); + EXPECT_EQ_INT(future.monotonic_deadline.tv_nsec, now.tv_nsec); + EXPECT_FALSE(future.has_wall); + EXPECT_FALSE(stop_condition_reached(&future)); + + /* Move the 60-minute deadline into the past: the check now reports reached. */ + StopCondition past = stop_condition_make(true, 60, false, 0, now); + past.monotonic_deadline.tv_sec -= 7200; + EXPECT_TRUE(stop_condition_reached(&past)); + + StopCondition no_after = stop_condition_make(false, 0, false, 0, now); + EXPECT_FALSE(no_after.has_monotonic); + EXPECT_FALSE(no_after.has_wall); + EXPECT_FALSE(stop_condition_reached(&no_after)); + + /* An invalid (non-positive) after_minutes never arms the monotonic half. */ + StopCondition zero_after = stop_condition_make(true, 0, false, 0, now); + EXPECT_FALSE(zero_after.has_monotonic); + StopCondition neg_after = stop_condition_make(true, -5, false, 0, now); + EXPECT_FALSE(neg_after.has_monotonic); + + /* --stop-at: a wall-clock deadline in the past/now is reached; one in the + future is not, and it stays independent of the monotonic half. */ + StopCondition wall_future = stop_condition_make(false, 0, true, time(NULL) + 3600, now); + EXPECT_TRUE(wall_future.has_wall); + EXPECT_FALSE(wall_future.has_monotonic); + EXPECT_FALSE(stop_condition_reached(&wall_future)); + + StopCondition wall_past = stop_condition_make(false, 0, true, time(NULL) - 1, now); + EXPECT_TRUE(stop_condition_reached(&wall_past)); + + EXPECT_FALSE(stop_condition_reached(NULL)); +} + +void test_stop(void) { + test_stop_after_parse_valid(); + test_stop_after_parse_invalid(); + test_stop_at_parse_hhmm(); + test_stop_at_parse_now_plus(); + test_stop_at_parse_invalid(); + test_stop_deadline_latency(); +} \ No newline at end of file diff --git a/tests/test_stop.h b/tests/test_stop.h new file mode 100644 index 0000000..3c5b85e --- /dev/null +++ b/tests/test_stop.h @@ -0,0 +1,6 @@ +#ifndef TEST_STOP_H +#define TEST_STOP_H + +void test_stop(void); + +#endif \ No newline at end of file diff --git a/tests/test_stress.c b/tests/test_stress.c index c73f46e..bf0e3f3 100644 --- a/tests/test_stress.c +++ b/tests/test_stress.c @@ -32,6 +32,8 @@ static int mpmc_producer_func(void* arg) { ProducerCtx* ctx = (ProducerCtx*)arg; for (int i = 1; i <= ITEMS_PER_PRODUCER; i++) { int* val = malloc(sizeof(int)); + if (!val) + return thrd_error; *val = ctx->producer_id * ITEMS_PER_PRODUCER + i; queue_enqueue_multithreaded(ctx->q, val, ctx->mutex, ctx->cnd_empty, ctx->cnd_full); } @@ -121,6 +123,8 @@ static int bp_producer_func(void* arg) { BackpressureCtx* ctx = (BackpressureCtx*)arg; for (int i = 0; i < 5; i++) { int* val = malloc(sizeof(int)); + if (!val) + return thrd_error; *val = i + 1; queue_enqueue_multithreaded(ctx->q, val, ctx->mutex, ctx->cnd_empty, ctx->cnd_full); ctx->items_sent++; @@ -194,6 +198,8 @@ static void test_queue_rapid_create_destroy() { for (int j = 0; j < 3; j++) { int* val = malloc(sizeof(int)); + if (!val) + break; *val = j; queue_enqueue(q, val); } diff --git a/tests/test_transport_ssh.c b/tests/test_transport_ssh.c index e085cbb..94a896a 100644 --- a/tests/test_transport_ssh.c +++ b/tests/test_transport_ssh.c @@ -4,34 +4,41 @@ static void test_ssh_connect_invalid_dest_no_colon() { /* cppcheck-suppress constVariablePointer */ - Client* client = client_connect_ssh("invalid-destination-no-colon", 22, NULL); + Client* client = + client_connect_ssh("invalid-destination-no-colon", 22, NULL, false, NULL, false, NULL, 0); EXPECT_NULL(client); } static void test_ssh_connect_invalid_dest_empty() { /* cppcheck-suppress constVariablePointer */ - Client* client = client_connect_ssh("", 22, NULL); + Client* client = client_connect_ssh("", 22, NULL, false, NULL, false, NULL, 0); EXPECT_NULL(client); } -/* Test client_connect_ssh with malformed destination (just a colon). - * parse_remote_dest succeeds, ssh is exec'd and fails, but the function - * creates a Client that must be cleaned up. */ +/* A child that cannot exec ssh must not be returned as a successful client. */ static void test_ssh_connect_malformed() { - Client* client = client_connect_ssh(":", 22, NULL); - /* ssh binary exists, so exec succeeds; the function returns a Client. - * We just verify it doesn't crash and clean up properly. */ - if (client != NULL) { - client_disconnect(client); - client_delete(client); + const char* old_path = getenv("PATH"); + char* saved_path = old_path ? strdup(old_path) : NULL; + setenv("PATH", "", 1); + + /* cppcheck-suppress constVariablePointer */ + Client* client = client_connect_ssh(":", 22, NULL, false, NULL, false, NULL, 0); + + if (saved_path) { + setenv("PATH", saved_path, 1); + free(saved_path); + } else { + unsetenv("PATH"); } - EXPECT_TRUE(true); + + EXPECT_NULL(client); } /* Test client_connect_ssh with valid format but unreachable host. * The function launches ssh which will fail to connect, returns a Client. */ static void test_ssh_connect_unreachable() { - Client* client = client_connect_ssh("nonexistent.invalid:/remote/path", 22, NULL); + Client* client = + client_connect_ssh("nonexistent.invalid:/remote/path", 22, NULL, false, NULL, false, NULL, 0); if (client != NULL) { client_disconnect(client); client_delete(client); @@ -39,9 +46,126 @@ static void test_ssh_connect_unreachable() { EXPECT_TRUE(true); } +static void test_ssh_remote_command_argument_modes() { + char* command = ssh_build_remote_command("fast sync; touch /tmp/pwned", false, NULL, 0); + EXPECT_EQ_STR(command, "'fast sync; touch /tmp/pwned' --stdio"); + free(command); + + command = ssh_build_remote_command("fast'sync", false, NULL, 0); + EXPECT_EQ_STR(command, "'fast'\\''sync' --stdio"); + free(command); + + /* --old-args no longer disables injection-safe quoting: the path is still one + single-quoted word, even when it carries shell metacharacters. */ + command = ssh_build_remote_command("fast sync; touch /tmp/pwned", true, NULL, 0); + EXPECT_EQ_STR(command, "'fast sync; touch /tmp/pwned' --stdio"); + free(command); + + command = ssh_build_remote_command("fast'sync; rm -rf /", true, NULL, 0); + EXPECT_EQ_STR(command, "'fast'\\''sync; rm -rf /' --stdio"); + free(command); +} + +/* The build for a single-word argv is [prog, six -o args, user, command]. */ + +static void test_ssh_build_client_argv_default_is_ssh() { + char** argv = ssh_build_client_argv(NULL, 0, "u@h", "'srv' --stdio"); + EXPECT_NOT_NULL(argv); + EXPECT_EQ_STR(argv[0], "ssh"); + EXPECT_EQ_STR(argv[1], "-o"); + EXPECT_EQ_STR(argv[7], "u@h"); + EXPECT_EQ_STR(argv[8], "'srv' --stdio"); + EXPECT_NULL(argv[9]); + ssh_free_client_argv(argv); +} + +/* A configured rsh must replace "ssh" as argv[0] (and never leak the default). */ +static void test_ssh_build_client_argv_uses_custom_rsh() { + char** argv = ssh_build_client_argv("myrsh", 0, "u@h", "rc"); + EXPECT_NOT_NULL(argv); + EXPECT_EQ_STR(argv[0], "myrsh"); + EXPECT_NULL(argv[9]); + ssh_free_client_argv(argv); +} + +/* A multi-word rsh command line (rsync -e "ssh -p 2222") is split into the + * leading argv words; a non-default port adds a -p/value pair. */ +static void test_ssh_build_client_argv_whitespace_command_and_port() { + char** argv = ssh_build_client_argv("ssh -p 2222", 0, "u@h", "rc"); + EXPECT_NOT_NULL(argv); + EXPECT_EQ_STR(argv[0], "ssh"); + EXPECT_EQ_STR(argv[1], "-p"); + EXPECT_EQ_STR(argv[2], "2222"); + EXPECT_NULL(argv[11]); + ssh_free_client_argv(argv); + + argv = ssh_build_client_argv("ssh", 2222, "u@h", "rc"); + EXPECT_NOT_NULL(argv); + EXPECT_EQ_STR(argv[0], "ssh"); + /* Flat [prog, -o x6, -p, port, user, command]. */ + EXPECT_EQ_STR(argv[7], "-p"); + EXPECT_EQ_STR(argv[8], "2222"); + EXPECT_EQ_STR(argv[9], "u@h"); + EXPECT_EQ_STR(argv[10], "rc"); + EXPECT_NULL(argv[11]); + ssh_free_client_argv(argv); +} + +/* --remote-option=OPT appends OPT to the remote command line after " --stdio", + * each escaped as its own single-quoted shell word. Metacharacters that could + * break out of the quoting are neutralized (never injected), matching the + * ssh_build_remote_command safety boundary for the server path. */ +static void test_ssh_remote_command_with_remote_options() { + char* noop[] = {"--allow-delete"}; + char* command = ssh_build_remote_command("fastsync-server", false, noop, 1); + EXPECT_EQ_STR(command, "'fastsync-server' --stdio '--allow-delete'"); + free(command); + + /* Multiple options append in order, each as its own quoted word. */ + char* multi[] = {"-v", "--allow-delete"}; + command = ssh_build_remote_command("srv", false, multi, 2); + EXPECT_EQ_STR(command, "'srv' --stdio '-v' '--allow-delete'"); + free(command); + + /* A remote option containing a single quote and shell metacharacters is + escaped with the same "'\''" boundary, so it stays one word and cannot + break out into an arbitrary remote command. */ + char* val = strdup("--x=un'der; touch /tmp/pwned"); + char* dangerous[1] = {val}; + command = ssh_build_remote_command("srv", false, dangerous, 1); + EXPECT_EQ_STR(command, "'srv' --stdio '--x=un'\\''der; touch /tmp/pwned'"); + free(command); + free(val); + + /* --old-args still quotes both the server path and the remote options. */ + command = ssh_build_remote_command("srv", true, multi, 2); + EXPECT_EQ_STR(command, "'srv' --stdio '-v' '--allow-delete'"); + free(command); +} + +/* The remote command builder refuses to forward an empty or control-character + * remote option (defense-in-depth independent of the CLI validation). */ +static void test_ssh_remote_command_rejects_bad_options() { + char* empty[] = {""}; + EXPECT_NULL(ssh_build_remote_command("srv", false, empty, 1)); + + char nl = '\n'; + char* newline[] = {&nl}; + EXPECT_NULL(ssh_build_remote_command("srv", false, newline, 1)); + + char* with_null[] = {NULL}; + EXPECT_NULL(ssh_build_remote_command("srv", false, with_null, 1)); +} + void test_transport_ssh() { test_ssh_connect_invalid_dest_no_colon(); test_ssh_connect_invalid_dest_empty(); test_ssh_connect_malformed(); test_ssh_connect_unreachable(); -} + test_ssh_remote_command_argument_modes(); + test_ssh_build_client_argv_default_is_ssh(); + test_ssh_build_client_argv_uses_custom_rsh(); + test_ssh_build_client_argv_whitespace_command_and_port(); + test_ssh_remote_command_with_remote_options(); + test_ssh_remote_command_rejects_bad_options(); +} \ No newline at end of file diff --git a/tests/test_transport_tcp.c b/tests/test_transport_tcp.c index 7ec97fb..b4b7c89 100644 --- a/tests/test_transport_tcp.c +++ b/tests/test_transport_tcp.c @@ -2,14 +2,120 @@ #include "protocol.h" #include "test_utils.h" #include "transport_tcp.h" +#include +#include #include #include +#include + +/* -4/-6 map to a getaddrinfo ai_family hint: -4 -> AF_INET, -6 -> AF_INET6, + * and neither -> AF_UNSPEC. Both flags together are rejected earlier (in + * validate_config), so this helper never needs to prefer one over the other. */ +static void test_tcp_connect_family_hints() { + EXPECT_EQ_INT(tcp_connect_family(false, false), AF_UNSPEC); + EXPECT_EQ_INT(tcp_connect_family(true, false), AF_INET); + EXPECT_EQ_INT(tcp_connect_family(false, true), AF_INET6); +} + +/* --sockopts parsing+validation: every allowlisted KEY works, OPT=VAL values + * are captured, and an unknown option or a bad value is rejected (never + * silently ignored). */ +static void test_sockopts_parse_valid() { + SockOptEntry* out = NULL; + int count = 0; + EXPECT_EQ_INT(config_sockopts_parse("TCP_NODELAY=1,SO_KEEPALIVE=0", &out, &count), 0); + EXPECT_EQ_INT(count, 2); + EXPECT_EQ_INT(out[0].id, SOCKOPT_TCP_NODELAY); + EXPECT_EQ_INT(out[0].value, 1); + EXPECT_EQ_INT(out[1].id, SOCKOPT_SO_KEEPALIVE); + EXPECT_EQ_INT(out[1].value, 0); + free(out); + + out = NULL; + count = 0; + EXPECT_EQ_INT( + config_sockopts_parse("SO_RCVBUF=65536,SO_SNDBUF=131072,SO_REUSEADDR=1", &out, &count), 0); + EXPECT_EQ_INT(count, 3); + EXPECT_EQ_INT(out[0].id, SOCKOPT_SO_RCVBUF); + EXPECT_EQ_INT(out[0].value, 65536); + EXPECT_EQ_INT(out[1].id, SOCKOPT_SO_SNDBUF); + EXPECT_EQ_INT(out[1].value, 131072); + EXPECT_EQ_INT(out[2].id, SOCKOPT_SO_REUSEADDR); + EXPECT_EQ_INT(out[2].value, 1); + free(out); +} + +static void test_sockopts_parse_rejects() { + static const char* const bad[] = {"IP_TTL=1", /* unknown option name */ + "SO_KEEPALIVE", /* missing '=' */ + "=1", /* missing option name */ + "TCP_NODELAY=", /* missing value */ + "TCP_NODELAY=2", /* boolean must be 0/1 */ + "TCP_NODELAY=on", /* non-numeric boolean */ + "SO_RCVBUF=-1", /* negative buffer */ + "SO_SNDBUF=abc", /* non-numeric buffer */ + ""}; /* empty spec */ + for (size_t i = 0; i < sizeof(bad) / sizeof(bad[0]); i++) { + SockOptEntry* out = NULL; + int count = 0; + EXPECT_EQ_INT(config_sockopts_parse(bad[i], &out, &count), -1); + EXPECT_NULL(out); + } +} + +/* Applying a validated allowlist entry must actually set the socket option (a + * real setsockopt on a fresh TCP socket) so the config->wire path is proven. */ +static void test_sockopts_apply_sets_option() { + SockOptEntry* entries = NULL; + int count = 0; + EXPECT_EQ_INT(config_sockopts_parse("TCP_NODELAY=1,SO_REUSEADDR=1", &entries, &count), 0); + + int fd = socket(AF_INET, SOCK_STREAM, 0); + EXPECT_TRUE(fd >= 0); + for (int i = 0; i < count; i++) { + int value = entries[i].value; + int level = entries[i].id == SOCKOPT_TCP_NODELAY ? IPPROTO_TCP : SOL_SOCKET; + int name = entries[i].id == SOCKOPT_TCP_NODELAY ? TCP_NODELAY : SO_REUSEADDR; + EXPECT_EQ_INT(setsockopt(fd, level, name, &value, sizeof(value)), 0); + } + int got = 0; + socklen_t len = sizeof(got); + EXPECT_EQ_INT(getsockopt(fd, IPPROTO_TCP, TCP_NODELAY, &got, &len), 0); + EXPECT_EQ_INT(got, 1); + close(fd); + free(entries); +} + +/* server_create_ex with an explicit --address and family binds to that local + * address; the resulting socket's address family must match. */ +static void test_server_create_bind_address() { + ServerBindOptions opts; + opts.bind_address = "127.0.0.1"; + opts.family = AF_INET; + Server* s = server_create_ex(0, &opts); + EXPECT_NOT_NULL(s); + EXPECT_EQ_INT(s->address.ss_family, AF_INET); + server_delete(&s); +} + +/* An IPv6 bind is honored when the host supports it; on a host with no IPv6 a + * NULL return is acceptable (the feature degrades to unavailable, not wrong). */ +static void test_server_create_bind_ipv6() { + ServerBindOptions opts; + opts.bind_address = "::1"; + opts.family = AF_INET6; + Server* s = server_create_ex(0, &opts); + if (s) { + EXPECT_EQ_INT(s->address.ss_family, AF_INET6); + server_delete(&s); + } +} static void test_server_create_ephemeral() { Server* s = server_create(0); EXPECT_NOT_NULL(s); EXPECT_TRUE(s->file_descriptor >= 0); - EXPECT_EQ_INT(s->address.sin_family, AF_INET); + EXPECT_EQ_INT(s->address.ss_family, AF_INET); server_delete(&s); EXPECT_NULL(s); } @@ -23,8 +129,8 @@ static void test_server_delete_null() { static void test_client_create() { Client* c = client_create(); EXPECT_NOT_NULL(c); - EXPECT_TRUE(c->file_descriptor >= 0); - EXPECT_EQ_INT(c->address.sin_family, AF_INET); + EXPECT_TRUE(c->file_descriptor == -1); + EXPECT_EQ_INT(c->address.ss_family, 0); EXPECT_EQ_INT(c->ssh_child_pid, -1); EXPECT_NULL(c->ssl); EXPECT_NULL(c->ssl_ctx); @@ -99,4 +205,10 @@ void test_transport_tcp() { test_server_create_specific_port(); test_server_delete_double(); test_client_disconnect_delete(); + test_tcp_connect_family_hints(); + test_sockopts_parse_valid(); + test_sockopts_parse_rejects(); + test_sockopts_apply_sets_option(); + test_server_create_bind_address(); + test_server_create_bind_ipv6(); } diff --git a/tests/test_transport_tls.c b/tests/test_transport_tls.c index 5e21580..28bbba5 100644 --- a/tests/test_transport_tls.c +++ b/tests/test_transport_tls.c @@ -3,6 +3,7 @@ #include "test_utils.h" #include "transport_tcp.h" #include "transport_tls.h" +#include #include #include @@ -17,6 +18,12 @@ static void test_server_create_tls_without_certs() { bool ok = server_create_tls(s, NULL, NULL, NULL); EXPECT_TRUE(ok); EXPECT_NOT_NULL(s->ssl_ctx); + /* The context must disable TLS compression (CRIME) and renegotiation. */ + SSL_CTX* ctx = (SSL_CTX*)s->ssl_ctx; + EXPECT_TRUE((SSL_CTX_get_options(ctx) & SSL_OP_NO_COMPRESSION) != 0); +#ifdef SSL_OP_NO_RENEGOTIATION + EXPECT_TRUE((SSL_CTX_get_options(ctx) & SSL_OP_NO_RENEGOTIATION) != 0); +#endif server_delete(&s); EXPECT_NULL(s); } diff --git a/tests/test_xattr.c b/tests/test_xattr.c new file mode 100644 index 0000000..4997ae2 --- /dev/null +++ b/tests/test_xattr.c @@ -0,0 +1,366 @@ +#include "test_xattr.h" +#include "xattr.h" +#include "config.h" +#include "file.h" +#include "identity.h" +#include "protocol.h" +#include "test_utils.h" +#include +#include +#include +#include +#include +#include +#include +#include +#include + +static void run_recv_helper(int fd) { + int ok = 0; + FileXattrList* list = xattr_receive(fd, &ok); + if (!ok) + _exit(1); + if (!list) { + /* NULL list only on error, already handled above. */ + _exit(1); + } + if (list->count != 2) + _exit(1); + if (strcmp(list->items[0].name, "user.foo") != 0 || list->items[0].value_len != 3 || + memcmp(list->items[0].value, "bar", 3) != 0) + _exit(1); + if (strcmp(list->items[1].name, "user.empty") != 0 || list->items[1].value_len != 0) + _exit(1); + xattr_list_free(list); + _exit(0); +} + +static void test_xattr_wire_roundtrip() { + /* Round-trip a user.* list incl. an empty value. */ + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + run_recv_helper(p[0]); + } + close(p[0]); + io_set_fds(p[1], p[1]); + FileXattrList* list = xattr_list_new(); + EXPECT_NOT_NULL(list); + EXPECT_TRUE(xattr_list_append(list, "user.foo", "bar", 3)); + EXPECT_TRUE(xattr_list_append(list, "user.empty", NULL, 0)); + EXPECT_TRUE(xattr_send(p[1], list)); + xattr_list_free(list); + int status; + waitpid(pid, &status, 0); + close(p[1]); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); +} + +static void run_recv_must_fail(int fd) { + int ok = 0; + FileXattrList* list = xattr_receive(fd, &ok); + /* A NULL list with ok==0 is the expected rejection. */ + if (ok == 0 && list == NULL) + _exit(0); + xattr_list_free(list); + _exit(1); +} + +/* A receiver must reject a security.* (privileged-namespace) attribute, never + * apply it: the send side can be malicious, so only the receiver whitelist + * matters. */ +static void test_xattr_reject_privileged_namespace() { + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + run_recv_must_fail(p[0]); + } + close(p[0]); + io_set_fds(p[1], p[1]); + FileXattrList* list = xattr_list_new(); + EXPECT_NOT_NULL(list); + /* security.capability must be rejected by the receiver. */ + EXPECT_TRUE(xattr_list_append(list, "security.capability", "\x01\x00", 2)); + xattr_send(p[1], list); /* receiver rejects at the name check and exits */ + xattr_list_free(list); + int status; + waitpid(pid, &status, 0); + close(p[1]); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); +} + +/* An oversized value (beyond XATTR_VALUE_MAX) must be rejected on receive. */ +static void test_xattr_reject_oversized_value() { + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + run_recv_must_fail(p[0]); + } + close(p[0]); + io_set_fds(p[1], p[1]); + FileXattrList* list = xattr_list_new(); + EXPECT_NOT_NULL(list); + size_t huge = (size_t)XATTR_VALUE_MAX + 1; + unsigned char* blob = calloc(1, huge); + EXPECT_NOT_NULL(blob); + EXPECT_TRUE(xattr_list_append(list, "user.huge", blob, huge)); + xattr_send(p[1], list); /* send is best-effort; the receiver rejects and exits */ + free(blob); + xattr_list_free(list); + int status; + waitpid(pid, &status, 0); + close(p[1]); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); +} + +/* The list append enforces the count bound (defense in depth). */ +static void test_xattr_count_bound() { + FileXattrList* list = xattr_list_new(); + EXPECT_NOT_NULL(list); + bool all_ok = true; + for (int i = 0; i < XATTR_MAX_COUNT + 1; i++) { + char name[32]; + snprintf(name, sizeof(name), "user.k%d", i); + if (!xattr_list_append(list, name, "v", 1)) + all_ok = false; + } + EXPECT_FALSE(all_ok); + EXPECT_EQ_INT(list->count, XATTR_MAX_COUNT); + xattr_list_free(list); +} + +/* The captured list on a plain file reflects only whitelisted namespaces + * (Linux only; skipped when the filesystem has no xattr support). */ +static void test_xattr_capture_and_appliable() { + EXPECT_FALSE(xattr_name_appliable(NULL)); + EXPECT_FALSE(xattr_name_appliable("")); + EXPECT_FALSE(xattr_name_appliable("security.selinux")); + EXPECT_FALSE(xattr_name_appliable("trusted.blob")); + EXPECT_TRUE(xattr_name_appliable("user.foo")); + /* The reserved fake-super key is receiver-only and never forwarded/applied. */ + EXPECT_FALSE(xattr_name_appliable("user.fastsync.stat")); + EXPECT_TRUE(xattr_name_appliable("system.posix_acl_access")); + EXPECT_TRUE(xattr_name_appliable("system.posix_acl_default")); +} + +/* MINOR-2: a --link-dest / -H copy fallback (linkat refused) must still apply + * the per-file xattrs and --fake-super stat. A DIRECTORY basis forces linkat + * to fail with EPERM, exercising the byte-copy fallback deterministically. + * Guarded on filesystem xattr support. */ +static void test_link_copy_fallback_preserves_xattrs() { + const char* dest = "test_link_xattr_dest.txt"; + const char* basis_dir = "test_link_xattr_basis_dir"; + unlink(dest); + rmdir(basis_dir); + EXPECT_EQ_INT(mkdir(basis_dir, 0700), 0); + + /* Probe xattr support on the cwd filesystem using the destination file. */ + int probe = open(dest, O_WRONLY | O_CREAT | O_TRUNC, 0600); + bool has_xattr = probe >= 0 && setxattr(dest, "user.fastsync.xprobe", "p", 1, 0) == 0; + if (probe >= 0) + close(probe); + if (!has_xattr) { + removexattr(dest, "user.fastsync.xprobe"); + unlink(dest); + rmdir(basis_dir); + return; /* skip silently when the filesystem has no xattr support */ + } + removexattr(dest, "user.fastsync.xprobe"); + + FileXattrList* xattrs = xattr_list_new(); + EXPECT_NOT_NULL(xattrs); + EXPECT_TRUE(xattr_list_append(xattrs, "user.fallback", "kept", 4)); + + FileMetadata m; + memset(&m, 0, sizeof(m)); + m.mode = 0640; + m.uid = 1001; + m.gid = 1002; + m.mtime_sec = 1234567890; + m.mtime_nsec = 0; + m.atime_valid = false; + m.crtime_valid = false; + + bool ok = file_to_disk_secure_link_attrs(dest, basis_dir, "payload", 7, false, &m, false, false, + xattrs, true, NULL); + xattr_list_free(xattrs); + EXPECT_TRUE(ok); + + /* Content landed (the copy fallback wrote the caller's bytes). */ + int fd = open(dest, O_RDONLY); + EXPECT_TRUE(fd >= 0); + char buf[16]; + ssize_t n = read(fd, buf, sizeof(buf)); + close(fd); + EXPECT_EQ_INT((int)strlen("payload"), (int)n); + if (n == 7) + EXPECT_TRUE(memcmp(buf, "payload", 7) == 0); + /* Per-file xattr applied on the copy. */ + char vbuf[16]; + ssize_t vlen = getxattr(dest, "user.fallback", vbuf, sizeof(vbuf)); + EXPECT_EQ_INT(4, (int)vlen); + if (vlen == 4) + EXPECT_TRUE(memcmp(vbuf, "kept", 4) == 0); + /* fake-super stat parked by the receiver. */ + EXPECT_TRUE((int)getxattr(dest, FAKESUPER_XATTR, NULL, 0) > 0); + + unlink(dest); + rmdir(basis_dir); +} + +/* --fake-super replay: fake_super_store_fd records the source stat into the + * reserved xattr, and fake_super_restore_fd re-applies mode/mtime (and owner, + * when the process may) fd-relative. Restore must also be a safe no-op with no + * xattr present. Guarded on filesystem xattr support. */ +static void test_fake_super_restore() { + const char* path = "test_fake_super_restore.txt"; + unlink(path); + int fd = open(path, O_WRONLY | O_CREAT | O_TRUNC, 0600); + if (fd < 0) + return; + bool has_xattr = setxattr(path, "user.fastsync.xprobe", "p", 1, 0) == 0; + if (has_xattr) + removexattr(path, "user.fastsync.xprobe"); + if (!has_xattr) { + close(fd); + unlink(path); + return; /* skip silently when the filesystem has no xattr support */ + } + + /* No xattr present yet: restore is a silent no-op (returns false, no crash). */ + EXPECT_FALSE(fake_super_restore_fd(fd)); + + fake_super_store_fd(fd, 1001, 1002, 0751, 1700000000, 123456789); + EXPECT_TRUE(fake_super_restore_fd(fd)); + struct stat st; + EXPECT_EQ_INT(fstat(fd, &st), 0); + EXPECT_EQ_INT((int)(st.st_mode & 07777), 0751); + + /* Mode sanitization: the normal metadata path never grants group/other write + bits, and fake-super replay must not re-add them (a recorded 0666 restores + as 0644, never as world-writable). */ + fake_super_store_fd(fd, 1001, 1002, 0666, 1700000000, 0); + EXPECT_TRUE(fake_super_restore_fd(fd)); + EXPECT_EQ_INT(fstat(fd, &st), 0); + EXPECT_EQ_INT((int)(st.st_mode & 0777), 0644); + + /* Restore with a malformed record must skip without failing. */ + time_t before = st.st_mtime; + int wfd = open(path, O_RDONLY); + if (wfd >= 0) { + EXPECT_EQ_INT((int)fsetxattr(wfd, FAKESUPER_XATTR, "not-a-valid-record", 19, 0), 0); + close(wfd); + } + EXPECT_FALSE(fake_super_restore_fd(fd)); + fstat(fd, &st); + EXPECT_EQ_INT((int)st.st_mtime, (int)before); + + close(fd); + unlink(path); +} + +/* --fake-super owner replay must honor the super gate and copy-as authority: + --no-super suppresses the recorded-source-owner chown even for root, and an + active --copy-as keeps its forced owner (the recorded source owner must never + override it). Root-gated: only root can observe a chown actually landing. */ +static void test_fake_super_owner_gate() { + if (geteuid() != 0) + return; /* non-root cannot observe ownership changes; skip silently */ + const char* path = "test_fake_super_owner_gate.txt"; + unlink(path); + int fd = open(path, O_WRONLY | O_CREAT | O_TRUNC, 0600); + if (fd < 0) + return; + bool has_xattr = setxattr(path, "user.fastsync.xprobe", "p", 1, 0) == 0; + if (has_xattr) + removexattr(path, "user.fastsync.xprobe"); + if (!has_xattr) { + close(fd); + unlink(path); + return; /* filesystem without xattr support */ + } + if (fchown(fd, 0, 0) != 0) { + close(fd); + unlink(path); + return; + } + fake_super_store_fd(fd, 12345, 12346, 0755, 1700000000, 0); + + Config* c = config_create(); + EXPECT_NOT_NULL(c); + + /* An explicit ownership policy is required before fake-super replay may + chown; --fake-super alone only records the source owner (A2). */ + c->numeric_ids = true; + + /* --no-super: the owner leg is skipped even as root. */ + c->super_mode = SUPER_MODE_OFF; + EXPECT_TRUE(identity_set_active(c)); + EXPECT_TRUE(fake_super_restore_fd(fd)); + struct stat st; + EXPECT_EQ_INT(fstat(fd, &st), 0); + EXPECT_EQ_INT((int)st.st_uid, 0); + EXPECT_EQ_INT((int)st.st_gid, 0); + + /* AUTO with an identity policy: the recorded source owner is applied. */ + c->super_mode = SUPER_MODE_AUTO; + EXPECT_TRUE(identity_set_active(c)); + EXPECT_TRUE(fake_super_restore_fd(fd)); + EXPECT_EQ_INT(fstat(fd, &st), 0); + EXPECT_EQ_INT((int)st.st_uid, 12345); + EXPECT_EQ_INT((int)st.st_gid, 12346); + + /* --super / --fake-super with NO explicit identity flag must NOT apply a + client-chosen owner: super_mode alone never enables ownership. */ + EXPECT_EQ_INT(fchown(fd, 0, 0), 0); + c->numeric_ids = false; + c->super_mode = SUPER_MODE_ON; + EXPECT_TRUE(identity_set_active(c)); + EXPECT_TRUE(fake_super_restore_fd(fd)); + EXPECT_EQ_INT(fstat(fd, &st), 0); + EXPECT_EQ_INT((int)st.st_uid, 0); + EXPECT_EQ_INT((int)st.st_gid, 0); + + /* Active --copy-as is authoritative: the recorded source owner must not + override it, even with AUTO/ON. */ + c->copy_as_set = true; + c->copy_as_uid = 777; + c->copy_as_gid = 778; + EXPECT_TRUE(identity_set_active(c)); + EXPECT_TRUE(fake_super_restore_fd(fd)); + EXPECT_EQ_INT(fstat(fd, &st), 0); + EXPECT_EQ_INT((int)st.st_uid, 0); + EXPECT_EQ_INT((int)st.st_gid, 0); + + identity_clear_active(); + config_delete(c); + close(fd); + unlink(path); +} + +void test_xattr() { + test_xattr_wire_roundtrip(); + test_xattr_reject_privileged_namespace(); + test_xattr_reject_oversized_value(); + test_xattr_count_bound(); + test_xattr_capture_and_appliable(); + test_link_copy_fallback_preserves_xattrs(); + test_fake_super_restore(); + test_fake_super_owner_gate(); +} \ No newline at end of file diff --git a/tests/test_xattr.h b/tests/test_xattr.h new file mode 100644 index 0000000..8252143 --- /dev/null +++ b/tests/test_xattr.h @@ -0,0 +1,6 @@ +#ifndef TEST_XATTR_H +#define TEST_XATTR_H + +void test_xattr(void); + +#endif \ No newline at end of file