From 4ac37c4d8af5b111b723c8ab3bb713b4033f0909 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sat, 12 Sep 2026 20:42:47 +0200 Subject: [PATCH 001/155] refactor(dir-times): extract dir_times_should_capture predicate Deduplicate the repeated directory-time capture gate (`config->use_metadata && !config->omit_dir_times`) used by the sender-side (multiprocessing.c) and receiver-side (receiver.c) sinks into a single predicate declared next to the DirTimeList machinery in file_receive.h and defined in file_receive.c. Behavior preserved: identical short-circuit condition and semantics, no signature or protocol changes. --- src/server/receiver.c | 2 +- src/shared/file_receive.c | 4 ++++ src/shared/file_receive.h | 6 ++++++ src/shared/multiprocessing.c | 3 ++- 4 files changed, 13 insertions(+), 2 deletions(-) diff --git a/src/server/receiver.c b/src/server/receiver.c index 936fc1f..4b84831 100644 --- a/src/server/receiver.c +++ b/src/server/receiver.c @@ -353,7 +353,7 @@ static bool receiver_save_file(File* file, void* context_pointer) { metadata now and apply it at the end. -O/--omit-dir-times is honored by dir_time_list_apply's caller (see receiver_send_success_frame). */ if (result != FILE_SAVE_ERROR && file->is_dir && file->metadata && - context->config->use_metadata && !context->config->omit_dir_times && + dir_times_should_capture(context->config) && !dir_time_list_add(&context->dir_times, file->path, file->metadata)) { file_destroy(file); return false; diff --git a/src/shared/file_receive.c b/src/shared/file_receive.c index d29bf77..ffb72df 100644 --- a/src/shared/file_receive.c +++ b/src/shared/file_receive.c @@ -2236,6 +2236,10 @@ File* file_receive(const Config* config, int file_descriptor) { /* ---- P7 Wave D: deferred directory times ---- */ +bool dir_times_should_capture(const Config* config) { + return config->use_metadata && !config->omit_dir_times; +} + void dir_time_list_init(DirTimeList* list) { if (!list) return; diff --git a/src/shared/file_receive.h b/src/shared/file_receive.h index 555a342..37dc15a 100644 --- a/src/shared/file_receive.h +++ b/src/shared/file_receive.h @@ -30,6 +30,12 @@ typedef struct { size_t capacity; } DirTimeList; +/* Capture gate shared by the sender-side and receiver-side sinks: directory + * metadata is accumulated only when --times/--metadata is in effect and + * -O/--omit-dir-times does not suppress it. Kept here, next to the accumulator + * it guards, so both call sites express the same condition. */ +bool dir_times_should_capture(const Config* config); + void dir_time_list_init(DirTimeList* list); void dir_time_list_free(DirTimeList* list); /* Deep-copy one directory's path + metadata into the list. Returns false on diff --git a/src/shared/multiprocessing.c b/src/shared/multiprocessing.c index 57af92a..b93c9f0 100644 --- a/src/shared/multiprocessing.c +++ b/src/shared/multiprocessing.c @@ -6,6 +6,7 @@ #include "config.h" #include "data.h" #include "file.h" +#include "file_receive.h" #include "log.h" #include "protocol.h" #include "queue.h" @@ -331,7 +332,7 @@ int write_thread(void* pipeline_context) { write would clobber them); accumulate the metadata here and let the caller apply it once every writer has drained. */ if (result != FILE_SAVE_ERROR && file->is_dir && file->metadata && - context->config->use_metadata && !context->config->omit_dir_times && + dir_times_should_capture(context->config) && !dir_time_list_add(&context->dir_times, file->path, file->metadata)) { file_destroy(file); pipeline_context_receiver_note_bytes_released(context, file_bytes); From 1fd462cca8b0f4843abce8156b72e490828fcf2f Mon Sep 17 00:00:00 2001 From: TapTap Date: Sat, 12 Sep 2026 20:43:24 +0200 Subject: [PATCH 002/155] refactor(config): replace SUPER_MODE_* macros with SuperMode enum Type Config.super_mode as SuperMode (a proper C enum) instead of a bare int. The wire boundary still carries the mode as an int: send casts the enum explicitly and receive reads a temporary int, validates the AUTO..OFF range, then casts. Emitted bytes and accepted values are unchanged. ModuleGateContext.super_mode_override keeps its -1 sentinel as int with an explicit cast at the apply site. Behavior preserved. --- src/server/server.c | 2 +- src/shared/config.c | 4 ++-- src/shared/config.h | 18 ++++++++---------- src/shared/identity.c | 4 ++-- src/shared/identity.h | 2 +- tests/test_config.c | 2 +- 6 files changed, 15 insertions(+), 17 deletions(-) diff --git a/src/server/server.c b/src/server/server.c index dce8fee..066b817 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -458,7 +458,7 @@ void handler(int file_descriptor) { * device-node creation) sees SUPER_MODE_OFF. The gate never mutated the * received config. */ if (gate_ctx.super_mode_override != -1) - config->super_mode = gate_ctx.super_mode_override; + config->super_mode = (SuperMode)gate_ctx.super_mode_override; protocol_set_8_bit_output(config->eight_bit_output); if (!authorized_root) { log_message(LOG_LEVEL_ERROR, "No server-side destination root configured"); diff --git a/src/shared/config.c b/src/shared/config.c index 61ea67c..4bd72c9 100644 --- a/src/shared/config.c +++ b/src/shared/config.c @@ -1293,14 +1293,14 @@ static bool receive_iconv_spec(int fd, Config* c) { * validated to the SUPER_MODE_AUTO..SUPER_MODE_OFF range (also re-checked by * validate_received_config). */ static bool send_privilege_options(int fd, const Config* c) { - return send_int(fd, c->super_mode); + return send_int(fd, (int)c->super_mode); } static bool receive_privilege_options(int fd, Config* c) { int mode; if (!receive_int(fd, &mode) || mode < SUPER_MODE_AUTO || mode > SUPER_MODE_OFF) return false; - c->super_mode = mode; + c->super_mode = (SuperMode)mode; return true; } diff --git a/src/shared/config.h b/src/shared/config.h index da3770e..66fd5eb 100644 --- a/src/shared/config.h +++ b/src/shared/config.h @@ -68,6 +68,13 @@ typedef struct { int value; /* 0/1 for booleans, byte count for SO_RCVBUF/SO_SNDBUF */ } SockOptEntry; +/* --super / --no-super tri-state (Config->super_mode). AUTO (default) and ON + * both permit a confined super-user attempt (AUTO preserves FastSync's + * historical best-effort behavior; an unprivileged attempt is refused by the + * kernel and skipped per entry); OFF forbids the attempt even for root. See + * privilege_super_mode_permitted() in identity.h. */ +typedef enum SuperMode { SUPER_MODE_AUTO = 0, SUPER_MODE_ON = 1, SUPER_MODE_OFF = 2 } SuperMode; + typedef struct Config { char* version; char* send_directory; @@ -416,7 +423,7 @@ typedef struct Config { * as a trailing int so the receiver can enforce the policy. See * privilege_super_permitted() and identity_ownership_requested() in * identity.h. */ - int super_mode; + SuperMode super_mode; // Receiver-side runtime staging registry for --delay-updates. Never sent // over the wire and never set on the sender side. @@ -644,15 +651,6 @@ typedef struct Config { #define IDENTITY_CURRENT (-1) #define MAX_IDENTITY_MAP 128 -/* --super / --no-super tri-state (Config->super_mode). AUTO (default) and ON - * both permit a confined super-user attempt (AUTO preserves FastSync's - * historical best-effort behavior; an unprivileged attempt is refused by the - * kernel and skipped per entry); OFF forbids the attempt even for root. See - * privilege_super_mode_permitted() in identity.h. */ -#define SUPER_MODE_AUTO 0 -#define SUPER_MODE_ON 1 -#define SUPER_MODE_OFF 2 - Config* config_create(void); void config_delete(Config* config); diff --git a/src/shared/identity.c b/src/shared/identity.c index 5a39f0e..a43c84b 100644 --- a/src/shared/identity.c +++ b/src/shared/identity.c @@ -30,7 +30,7 @@ typedef struct { /* --super / --no-super tri-state (SUPER_MODE_AUTO when unset). Snapshotted * per connection so privilege_super_permitted() can gate super-user * activities without a Config argument. */ - int super_mode; + SuperMode super_mode; /* --copy-as=USER[:GROUP]: snapshotted so the ownership resolver can force the * target ids without a Config argument. */ bool copy_as_set; @@ -128,7 +128,7 @@ bool privilege_super_permitted(void) { return privilege_super_mode_permitted(g_identity.super_mode); } -bool privilege_super_mode_permitted(int mode) { +bool privilege_super_mode_permitted(SuperMode mode) { /* AUTO and ON both attempt the confined operation; OFF forbids it even for a * root receiver. AUTO is the historical FastSync behavior (always attempt * and let the kernel refuse an unprivileged call, which the caller skips), so diff --git a/src/shared/identity.h b/src/shared/identity.h index 592c5e7..e01c674 100644 --- a/src/shared/identity.h +++ b/src/shared/identity.h @@ -128,6 +128,6 @@ bool identity_wire_valid(const Config* config); * best-effort behavior where an unprivileged attempt is refused by the kernel * and skipped. Neither EVER elevates privileges. */ bool privilege_super_permitted(void); -bool privilege_super_mode_permitted(int mode); +bool privilege_super_mode_permitted(SuperMode mode); #endif \ No newline at end of file diff --git a/tests/test_config.c b/tests/test_config.c index f70d99d..4ffc49f 100644 --- a/tests/test_config.c +++ b/tests/test_config.c @@ -1672,7 +1672,7 @@ static void test_config_receive_rejects_invalid_iconv_spec() { static void test_config_super_mode_wire_roundtrip() { if (is_running_under_valgrind()) return; - int modes[] = {SUPER_MODE_AUTO, SUPER_MODE_ON, SUPER_MODE_OFF}; + SuperMode modes[] = {SUPER_MODE_AUTO, SUPER_MODE_ON, SUPER_MODE_OFF}; for (size_t i = 0; i < sizeof(modes) / sizeof(modes[0]); i++) { int p[2]; EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); From 921472b8b3d99a1bcc3db467b0ed2660f669f4ab Mon Sep 17 00:00:00 2001 From: TapTap Date: Sat, 12 Sep 2026 20:55:04 +0200 Subject: [PATCH 003/155] refactor(client-cli): split parse_args into focused option handlers Break the ~700-line parse_args god function into cohesive static helpers grouped by concern: output controls, pre-negation, range/time options, the OPTION_TABLE dispatcher, flag/meta handlers, IO/network options, filter and logging options, checksum/socket options, remote/basis/identity options, positional handling, and a final lowering step. A file-local CliParseCtx carries the config, cursor, positional buffers and the mutable parse flags, so each handler stays focused. The dispatcher calls the handlers in the original recognition order and preserves the exact return contract (0/1/negative), error messages, log levels and control flow. Behavior preserved; no functional changes. --- src/client/client_cli.c | 1620 ++++++++++++++++++++++++--------------- 1 file changed, 991 insertions(+), 629 deletions(-) diff --git a/src/client/client_cli.c b/src/client/client_cli.c index 9c10610..d1676da 100644 --- a/src/client/client_cli.c +++ b/src/client/client_cli.c @@ -819,18 +819,25 @@ static int apply_table_option(Config* config, const OptionEntry* entry, const ch return -1; } -/* Parse CLI arguments into config. Returns 0 on success, -1 on error, 1 for help/clean-exit. */ -int parse_args(Config* config, int argc, char* argv[], int* positional_args, - int* positional_count) { - bool verbose = false; - /* Explicit --no-delta / --no-incremental seen on the command line: the user - switched part of the delta machinery off, so the --fuzzy implication must - not silently turn it back on. */ - bool no_delta = false; - bool no_incremental = false; - protocol_set_8_bit_output(config->eight_bit_output); +/* Shared state for the parse_args helper functions. Keeping the cursor and the + * mutable parse flags here avoids threading a long parameter list through every + * option handler while preserving the original single-pass control flow. */ +typedef struct { + Config* config; + int argc; + char** argv; + int* positional_args; + int* positional_count; + int i; /* index of the argument currently being examined */ + int exit_code; /* nonzero when a matched handler wants parse_args to return */ + bool verbose; /* "-v"/"--verbose" seen (drives the final log level) */ + bool no_delta; /* explicit "--no-delta" seen */ + bool no_incremental; /* explicit "--no-incremental" seen */ +} CliParseCtx; - /* Apply output controls before processing other options so their order is irrelevant. */ +/* Apply output controls before processing other options so their order is + * irrelevant. Returns 0 on success, -1 on error. */ +static int cli_apply_output_controls(Config* config, int argc, char* argv[]) { for (int i = 1; i < argc; i++) { if (strcmp(argv[i], "-v") == 0 || strcmp(argv[i], "--verbose") == 0) { set_log_level(LOG_LEVEL_DEBUG); @@ -842,626 +849,924 @@ int parse_args(Config* config, int argc, char* argv[], int* positional_args, return -1; } } + return 0; +} - for (int i = 1; i < argc; i++) { - if (strcmp(argv[i], "-P") == 0) { - config->partial = true; - config->show_progress = true; - continue; - } - /* "--no-implied-dirs" is a real rsync option name, not a negation of - * "--implied-dirs", so it must be handled before the generic --no-* - * negation branch. */ - if (strcmp(argv[i], "--no-implied-dirs") == 0) { - config->no_implied_dirs = true; - continue; - } - /* "--no-motd" is a real rsync option name (client-side daemon MOTD display - * suppression), not a negation of a "--motd" flag, so it is handled before - * the generic --no-* negation branch. */ - if (strcmp(argv[i], "--no-motd") == 0) { - config->no_motd = true; - continue; - } - /* "--super" / "--no-super" are real rsync option names controlling the - * receiver's super-user activity policy (ownership, device nodes), not a - * Boolean pair for the generic --no-* negation branch: both map onto the - * Config->super_mode tri-state. Handle them explicitly (exact match only, - * so a malformed "--super=x" still falls through to the unknown-option - * error) before the generic negation branch would mis-reject "--no-super". */ - if (strcmp(argv[i], "--super") == 0) { - config->super_mode = SUPER_MODE_ON; - continue; - } - if (strcmp(argv[i], "--no-super") == 0) { - config->super_mode = SUPER_MODE_OFF; - continue; - } - if (strncmp(argv[i], "--no-", strlen("--no-")) == 0) { - if (strcmp(argv[i], "--no-delta") == 0) - no_delta = true; - else if (strcmp(argv[i], "--no-incremental") == 0) - no_incremental = true; - if (apply_negation(config, argv[i]) != 0) - return -1; - continue; - } - const char* modify_window_prefix = "--modify-window="; - if (strncmp(argv[i], modify_window_prefix, strlen(modify_window_prefix)) == 0) { - if (set_nonneg_int_option(&config->modify_window, argv[i] + strlen(modify_window_prefix), - "--modify-window") != 0) - return -1; - continue; - } - if (strncmp(argv[i], "-@", 2) == 0 && argv[i][2] != '\0') { - if (set_nonneg_int_option(&config->modify_window, argv[i] + 2, "-@") != 0) - return -1; - continue; - } - /* --stop-after/--stop-at are client-only sender-side stop deadlines. They - * are parsed by stop_condition (so the unit tests exercise the same validate - * that production uses) and never serialized into the config frame. */ - if (strncmp(argv[i], "--stop-after=", 13) == 0) { - if (!stop_parse_after_minutes(argv[i] + 13, &config->stop_after_mins)) { - log_message(LOG_LEVEL_ERROR, "--stop-after must be a positive number of minutes"); - return -1; - } - continue; - } - if (strcmp(argv[i], "--stop-after") == 0) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for --stop-after"); - return -1; - } - if (!stop_parse_after_minutes(argv[++i], &config->stop_after_mins)) { - log_message(LOG_LEVEL_ERROR, "--stop-after must be a positive number of minutes"); - return -1; - } - continue; - } - if (strncmp(argv[i], "--stop-at=", 10) == 0) { - if (!stop_parse_at_time(argv[i] + 10, time(NULL), &config->stop_at)) { - log_message(LOG_LEVEL_ERROR, "--stop-at must be HH:MM[:SS] or now+N[smhd]"); - return -1; - } - config->stop_at_set = true; - continue; - } - if (strcmp(argv[i], "--stop-at") == 0) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for --stop-at"); - return -1; - } - if (!stop_parse_at_time(argv[++i], time(NULL), &config->stop_at)) { - log_message(LOG_LEVEL_ERROR, "--stop-at must be HH:MM[:SS] or now+N[smhd]"); - return -1; - } - config->stop_at_set = true; - continue; - } - const char* threads_prefix = "--compress-threads="; - if (strncmp(argv[i], threads_prefix, strlen(threads_prefix)) == 0) { - if (set_compression_threads_option(&config->compression_threads, - argv[i] + strlen(threads_prefix)) != 0) - return -1; - continue; - } - if (strncmp(argv[i], "--max-alloc=", 12) == 0 || strcmp(argv[i], "--max-alloc") == 0) { - const char* value = strcmp(argv[i], "--max-alloc") == 0 ? "" : argv[i] + 12; - if (*value == '\0') { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for --max-alloc"); - return -1; - } - value = argv[++i]; - } - if (parse_size_arg(value, &config->max_alloc) != 0) { - log_message(LOG_LEVEL_ERROR, - "--max-alloc must be a positive size (B, K, M, G, T, P, or E)"); - return -1; - } - continue; - } - - const OptionEntry* entry = find_table_option(argv[i]); - const char* inline_value = NULL; - if (!entry) - entry = find_table_option_with_equals(argv[i], &inline_value); - if (entry) { - const char* value = NULL; - if (entry->kind != OPT_FLAG) { - value = inline_value; - if (!value && i + 1 < argc) - value = argv[++i]; - if (!value) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", entry->name); - return -1; - } - if (strcmp(entry->name, "--compress-choice") == 0) { - if (set_compression_choice(config, value) != 0) - return -1; - } else { - if (apply_table_option(config, entry, value) != 0) - return -1; - if (strcmp(entry->name, "--compress-level") == 0 && - (config->compression_level < 1 || config->compression_level > 22)) { - log_message(LOG_LEVEL_ERROR, "--compress-level must be between 1 and 22"); - return -1; - } - if (entry->offset == offsetof(Config, chmod_spec)) { - mode_t ignored; - if (!chmod_apply(0, config->chmod_spec, &ignored)) { - log_message(LOG_LEVEL_ERROR, "--chmod has invalid permission changes"); - return -1; - } - config->use_metadata = true; - } - } - } else if (apply_table_option(config, entry, NULL) != 0) { - return -1; - } - if (entry->offset == offsetof(Config, eight_bit_output)) - protocol_set_8_bit_output(true); - /* A delete-timing flag selects when --delete removes extras, so it - implies --delete exactly like the rsync options do. */ - if (entry->offset == offsetof(Config, delete_before) || - entry->offset == offsetof(Config, delete_during) || - entry->offset == offsetof(Config, delete_delay) || - entry->offset == offsetof(Config, delete_after)) - config->use_delete = true; - /* --delete-missing-args implies --ignore-missing-args (missing entries - are skipped for deletion instead of failing the run). The implication - is order-independent because it is applied over the final parsed - config. */ - if (entry->offset == offsetof(Config, delete_missing_args)) - config->ignore_missing_args = true; - /* -U/--atimes and -N/--crtimes carry their times inside the metadata - payload, which is only transmitted when use_metadata is set, so either - one implies metadata transmission. This is FastSync's broad -M bundle - (mode/mtime travel too); it does NOT enable ownership application, - which stays opt-in via the identity flags. */ - if (entry->offset == offsetof(Config, preserve_atimes) || - entry->offset == offsetof(Config, preserve_crtimes)) - config->use_metadata = true; - if (entry->offset == offsetof(Config, preserve_xattrs) || - entry->offset == offsetof(Config, preserve_acls)) { - config->use_metadata = true; - config->use_xattrs = config->preserve_acls || config->preserve_xattrs; - } - if (entry->offset == offsetof(Config, fake_super)) - config->use_metadata = true; - continue; - } - - if (strncmp(argv[i], "--chmod=", 8) == 0) { - if (set_string_option(&config->chmod_spec, argv[i] + 8, "--chmod") != 0) - return -1; - mode_t ignored; - if (!chmod_apply(0, config->chmod_spec, &ignored)) { - log_message(LOG_LEVEL_ERROR, "--chmod has invalid permission changes"); - return -1; - } - config->use_metadata = true; - continue; - } - - if (opt_is(argv[i], "--help", NULL)) { - print_usage(); - return 1; - } else if (opt_is(argv[i], "-V", "--version")) { - printf("fastsync version %s\n", PROTOCOL_VERSION); - return 1; - } else if (opt_is(argv[i], "-D", NULL)) { - /* rsync -D == --devices --specials. -D is otherwise unassigned in - FastSync (verified: no collision), so it is free to imply both. */ - config->preserve_devices = true; - config->preserve_specials = true; - log_info_message(LOG_INFO_MISC, "Enabled preservation of device and special files (-D)"); - } else if (opt_is(argv[i], "-a", "--archive")) { - /* Real rsync archive (-rlptgoD). FastSync is always recursive and always - * preserves hard-link/other transfer semantics per its own flags, so -a - * implies links, full metadata (perms/times/group/owner as FastSync's - * broad bundle), devices and specials. Compression and multithreading - * are NOT implied (they are no longer part of archive mode). */ - config->follow_symlinks = true; - config->use_metadata = true; - config->preserve_devices = true; - config->preserve_specials = true; - log_info_message(LOG_INFO_MISC, - "Enabled archive mode (-rlptgoD: links, metadata, devices, specials)"); - } else if (opt_is(argv[i], "-p", "--perms")) { - /* rsync -p/--perms: preserve permission bits. Folded into FastSync's - * broad metadata bundle (mode/mtime travel together). */ - config->use_metadata = true; - log_info_message(LOG_INFO_MISC, "Enabled permission preservation"); - } else if (opt_is(argv[i], "--ssh-port", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (set_positive_int_option(&config->ssh_port, argv[++i], "--ssh-port") != 0) - return -1; - if (config->ssh_port > 65535) { - log_message(LOG_LEVEL_ERROR, "SSH port must be 1-65535"); - return -1; - } - } else if (strncmp(argv[i], "--ssh-port=", 11) == 0) { - if (set_positive_int_option(&config->ssh_port, argv[i] + 11, "--ssh-port") != 0) - return -1; - if (config->ssh_port > 65535) { - log_message(LOG_LEVEL_ERROR, "SSH port must be 1-65535"); - return -1; - } - } else if (opt_is(argv[i], "--exclude", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (config_add_pattern(&config->exclude_patterns, &config->exclude_count, argv[++i], - "--exclude") != 0) - return -1; - } else if (opt_is(argv[i], "--include", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (config_add_pattern(&config->include_patterns, &config->include_count, argv[++i], - "--include") != 0) - return -1; - } else if (strncmp(argv[i], "--delta-block=", 14) == 0) { - if (set_delta_block_size(config, argv[i] + 14) != 0) - return -1; - } else if (strncmp(argv[i], "--block-size=", 13) == 0) { - if (set_delta_block_size(config, argv[i] + 13) != 0) - return -1; - } else if (opt_is(argv[i], "--delta-block", "--block-size")) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (set_delta_block_size(config, argv[++i]) != 0) - return -1; - } else if (opt_is(argv[i], "--delta-max", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - unsigned long long val; - if (parse_ull_arg(argv[++i], &val, "--delta-max") != 0) - return -1; - if (val >= DELTA_MIN_FILE_SIZE) - config->delta_max_file_size = val; - else - log_message(LOG_LEVEL_WARNING, "--delta-max value %llu too small, using default", val); - } else if (opt_is(argv[i], "-z", "--compress")) { - config->use_compression = - !config->compress_choice || strcmp(config->compress_choice, "zstd") == 0; - log_info_message(LOG_INFO_MISC, "Enabled Compression"); - if (i + 1 < argc) { - char* end_ptr; - long level = strtol(argv[i + 1], &end_ptr, 10); - if (*end_ptr == '\0') { - if (level < 1 || level > 22) { - log_message(LOG_LEVEL_ERROR, "compression level must be 1-22"); - return -1; - } - config->compression_level = (int)level; - log_info_message(LOG_INFO_MISC, "Set Compression level to %ld", level); - i++; - } - } - } else if (opt_is(argv[i], "--preserve", NULL)) { - config->use_metadata = true; - log_info_message(LOG_INFO_MISC, "Enabled metadata preservation"); - } else if (opt_is(argv[i], "-E", "--executability")) { - config->use_metadata = true; - config->use_executability = true; - log_info_message(LOG_INFO_MISC, "Enabled executable permission preservation"); - } else if (opt_is(argv[i], "--sendfile", NULL)) { - config->use_sendfile = true; - log_info_message(LOG_INFO_MISC, "Enabled sendfile"); - } else if (opt_is(argv[i], "-j", "--threads")) { - config->use_multithreading = true; - log_info_message(LOG_INFO_MISC, "Enabled Multithreading"); - } else if (opt_is(argv[i], "--chunk-serialization", NULL)) { - config->use_chunk_serialization = true; - log_info_message(LOG_INFO_MISC, "Enabled Chunk Serialization"); - } else if (opt_is(argv[i], "--server-port", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (!parse_positive_int(argv[++i], &config->server_port)) { - char* escaped = output_escape(argv[i], false); - log_message(LOG_LEVEL_ERROR, "invalid --server-port value: %s", - escaped ? escaped : ""); - free(escaped); - return -1; - } - if (config->server_port > 65535) { - log_message(LOG_LEVEL_ERROR, "server port must be 1-65535"); - return -1; - } - } else if (opt_is(argv[i], "--bwlimit", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - unsigned long long kbps; - if (parse_ull_arg(argv[++i], &kbps, "--bwlimit") != 0) - return -1; - if (kbps == 0) { - log_message(LOG_LEVEL_ERROR, "--bwlimit must be a positive integer"); - return -1; - } - if (kbps > ULLONG_MAX / 1024) { - log_message(LOG_LEVEL_ERROR, "--bwlimit value too large"); - return -1; - } - io_set_bwlimit(kbps * 1024); - log_info_message(LOG_INFO_MISC, "Set bandwidth limit to %llu KB/s", kbps); - } else if (opt_is(argv[i], "--chunk-size", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - unsigned long long val; - if (parse_ull_arg(argv[++i], &val, "--chunk-size") != 0) - return -1; - if (val == 0) { - log_message(LOG_LEVEL_ERROR, "--chunk-size must be a positive integer"); - return -1; - } - config->chunk_size = val; - } else if (opt_is(argv[i], "--log-file", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (config->log_file) { - fclose(config->log_file); - config->log_file = NULL; - log_set_file(NULL); - } - FILE* lf = fopen(argv[++i], "a"); - if (!lf) { - char* escaped = output_escape(argv[i], false); - log_message(LOG_LEVEL_ERROR, "could not open log file '%s': %s", - escaped ? escaped : "", strerror(errno)); - free(escaped); - return -1; - } - config->log_file = lf; - log_set_file(lf); - } else if (strncmp(argv[i], "--stderr=", 9) == 0) { - if (set_stderr_mode(argv[i] + 9) != 0) - return -1; - } else if (opt_is(argv[i], "--stderr", NULL)) { - if (i + 1 >= argc || set_stderr_mode(argv[++i]) != 0) - return -1; - } else if (opt_is(argv[i], "--exclude-from", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (read_patterns_from_file(argv[++i], &config->exclude_patterns, &config->exclude_count) != - 0) - return -1; - } else if (opt_is(argv[i], "--include-from", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (read_patterns_from_file(argv[++i], &config->include_patterns, &config->include_count) != - 0) - return -1; - } else if (strncmp(argv[i], "--filter=", 9) == 0) { - if (config_add_filter(config, argv[i] + 9) != 0) - return -1; - } else if (strncmp(argv[i], "-f=", 3) == 0) { - if (config_add_filter(config, argv[i] + 3) != 0) - return -1; - } else if (opt_is(argv[i], "--filter", "-f")) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (config_add_filter(config, argv[++i]) != 0) - return -1; - } else if (strncmp(argv[i], "--files-from=", 13) == 0) { - if (set_string_option(&config->files_from, argv[i] + 13, "--files-from") != 0) - return -1; - } else if (opt_is(argv[i], "--files-from", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (set_string_option(&config->files_from, argv[++i], "--files-from") != 0) - return -1; - } else if (opt_is(argv[i], "-v", "--verbose")) { - verbose = true; - set_log_level(LOG_LEVEL_DEBUG); - } else if (opt_is(argv[i], "-q", "--quiet")) { - config->quiet = true; - } else if (strncmp(argv[i], "--debug=", 8) == 0) { - int debug_ret = parse_debug_flags(argv[i] + 8, config); - if (debug_ret != 0) - return debug_ret; - } else if (opt_is(argv[i], "--debug", NULL)) { - if (i + 1 >= argc) - return parse_debug_flags(NULL, config); - int debug_ret = parse_debug_flags(argv[++i], config); - if (debug_ret != 0) - return debug_ret; - } else if (strncmp(argv[i], "--info=", 7) == 0) { - if (parse_info_flags(argv[i] + 7, config) != 0) - return -1; - } else if (opt_is(argv[i], "--info", NULL)) { - if (i + 1 >= argc || parse_info_flags(argv[++i], config) != 0) - return -1; - } else if (strncmp(argv[i], "--skip-compress=", 16) == 0) { - if (parse_skip_compress(config, argv[i] + 16) != 0) - return -1; - } else if (opt_is(argv[i], "--skip-compress", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (parse_skip_compress(config, argv[++i]) != 0) - return -1; - } else if (opt_is(argv[i], "--compress-threads", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (set_compression_threads_option(&config->compression_threads, argv[++i]) != 0) - return -1; - } else if (opt_is(argv[i], "--checksum-choice", "--cc")) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (set_checksum_choice(config, argv[++i]) != 0) - return -1; - } else if (strncmp(argv[i], "--checksum-choice=", 18) == 0) { - if (set_checksum_choice(config, argv[i] + 18) != 0) - return -1; - } else if (strncmp(argv[i], "--cc=", 5) == 0) { - if (set_checksum_choice(config, argv[i] + 5) != 0) - return -1; - } else if (strncmp(argv[i], "--checksum-seed=", 16) == 0) { - if (set_checksum_seed(config, argv[i] + 16) != 0) - return -1; - } else if (opt_is(argv[i], "--checksum-seed", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for --checksum-seed"); - return -1; - } - if (set_checksum_seed(config, argv[++i]) != 0) - return -1; - } else if (strncmp(argv[i], "--sockopts=", 11) == 0) { - if (set_sockopts_option(config, argv[i] + 11) != 0) - return -1; - } else if (opt_is(argv[i], "--sockopts", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for --sockopts"); - return -1; - } - if (set_sockopts_option(config, argv[++i]) != 0) - return -1; - } else if (strncmp(argv[i], "--remote-option=", 16) == 0) { - if (config_add_remote_option(config, argv[i] + 16, "--remote-option") != 0) - return -1; - } else if (strncmp(argv[i], "-M=", 3) == 0) { - if (config_add_remote_option(config, argv[i] + 3, "-M") != 0) - return -1; - } else if (opt_is(argv[i], "--remote-option", "-M")) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for --remote-option"); - return -1; - } - if (config_add_remote_option(config, argv[++i], "--remote-option") != 0) - return -1; - } else if (strncmp(argv[i], "--compare-dest=", 15) == 0) { - if (set_basis_dest_option(config, BASIS_DEST_COMPARE, argv[i] + 15, "--compare-dest") != 0) - return -1; - } else if (opt_is(argv[i], "--compare-dest", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (set_basis_dest_option(config, BASIS_DEST_COMPARE, argv[++i], "--compare-dest") != 0) - return -1; - } else if (strncmp(argv[i], "--copy-dest=", 12) == 0) { - if (set_basis_dest_option(config, BASIS_DEST_COPY, argv[i] + 12, "--copy-dest") != 0) - return -1; - } else if (opt_is(argv[i], "--copy-dest", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (set_basis_dest_option(config, BASIS_DEST_COPY, argv[++i], "--copy-dest") != 0) - return -1; - } else if (strncmp(argv[i], "--link-dest=", 12) == 0) { - if (set_basis_dest_option(config, BASIS_DEST_LINK, argv[i] + 12, "--link-dest") != 0) - return -1; - } else if (opt_is(argv[i], "--link-dest", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (set_basis_dest_option(config, BASIS_DEST_LINK, argv[++i], "--link-dest") != 0) - return -1; - } else if (strncmp(argv[i], "--usermap=", 10) == 0) { - if (identity_parse_map(config, argv[i] + 10, false) != 0) - return -1; - config->use_metadata = true; - } else if (opt_is(argv[i], "--usermap", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (identity_parse_map(config, argv[++i], false) != 0) - return -1; - config->use_metadata = true; - } else if (strncmp(argv[i], "--groupmap=", 11) == 0) { - if (identity_parse_map(config, argv[i] + 11, true) != 0) - return -1; - config->use_metadata = true; - } else if (opt_is(argv[i], "--groupmap", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (identity_parse_map(config, argv[++i], true) != 0) - return -1; - config->use_metadata = true; - } else if (strncmp(argv[i], "--chown=", 8) == 0) { - if (identity_parse_chown(config, argv[i] + 8) != 0) - return -1; - config->use_metadata = true; - } else if (opt_is(argv[i], "--chown", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (identity_parse_chown(config, argv[++i]) != 0) - return -1; - config->use_metadata = true; - } else if (strncmp(argv[i], "--copy-as=", 10) == 0) { - if (identity_parse_copy_as(config, argv[i] + 10) != 0) - return -1; - } else if (opt_is(argv[i], "--copy-as", NULL)) { - if (i + 1 >= argc) { - log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); - return -1; - } - if (identity_parse_copy_as(config, argv[++i]) != 0) - return -1; - } else if (strncmp(argv[i], "--outbuf=", 9) == 0) { - if (set_outbuf_option(config, argv[i] + 9) != 0) - return -1; - } else if (opt_is(argv[i], "--outbuf", NULL)) { - if (i + 1 >= argc || set_outbuf_option(config, argv[++i]) != 0) - return -1; - } else if (argv[i][0] == '-') { - char* escaped = output_escape(argv[i], false); - fprintf(stderr, "Unknown option: %s\n", escaped ? escaped : ""); - free(escaped); - print_usage(); - return -1; - } else { - if (*positional_count < 2) - positional_args[(*positional_count)++] = i; - else { - char* escaped = output_escape(argv[i], false); - fprintf(stderr, "Unexpected argument: %s\n", escaped ? escaped : ""); - free(escaped); - print_usage(); - return -1; - } - } +/* Options handled before the generic --no-* negation branch: -P and the real + * rsync option names that merely start with "--no-" (--no-implied-dirs, + * --no-motd, --no-super), plus the generic negation itself. Returns true when + * the argument was consumed. */ +static bool cli_handle_pre_negation(CliParseCtx* ctx) { + Config* config = ctx->config; + const char* arg = ctx->argv[ctx->i]; + if (strcmp(arg, "-P") == 0) { + config->partial = true; + config->show_progress = true; + return true; } + /* "--no-implied-dirs" is a real rsync option name, not a negation of + * "--implied-dirs", so it must be handled before the generic --no-* + * negation branch. */ + if (strcmp(arg, "--no-implied-dirs") == 0) { + config->no_implied_dirs = true; + return true; + } + /* "--no-motd" is a real rsync option name (client-side daemon MOTD display + * suppression), not a negation of a "--motd" flag, so it is handled before + * the generic --no-* negation branch. */ + if (strcmp(arg, "--no-motd") == 0) { + config->no_motd = true; + return true; + } + /* "--super" / "--no-super" are real rsync option names controlling the + * receiver's super-user activity policy (ownership, device nodes), not a + * Boolean pair for the generic --no-* negation branch: both map onto the + * Config->super_mode tri-state. Handle them explicitly (exact match only, + * so a malformed "--super=x" still falls through to the unknown-option + * error) before the generic negation branch would mis-reject "--no-super". */ + if (strcmp(arg, "--super") == 0) { + config->super_mode = SUPER_MODE_ON; + return true; + } + if (strcmp(arg, "--no-super") == 0) { + config->super_mode = SUPER_MODE_OFF; + return true; + } + if (strncmp(arg, "--no-", strlen("--no-")) == 0) { + if (strcmp(arg, "--no-delta") == 0) + ctx->no_delta = true; + else if (strcmp(arg, "--no-incremental") == 0) + ctx->no_incremental = true; + if (apply_negation(config, arg) != 0) { + ctx->exit_code = -1; + return true; + } + return true; + } + return false; +} + +/* Numeric/range/time options with dedicated prefixes: --modify-window, -@, + * --stop-after, --stop-at, --compress-threads and --max-alloc. Returns true + * when the argument was consumed. */ +static bool cli_handle_range_time_options(CliParseCtx* ctx) { + Config* config = ctx->config; + const char* arg = ctx->argv[ctx->i]; + const char* modify_window_prefix = "--modify-window="; + if (strncmp(arg, modify_window_prefix, strlen(modify_window_prefix)) == 0) { + if (set_nonneg_int_option(&config->modify_window, arg + strlen(modify_window_prefix), + "--modify-window") != 0) + ctx->exit_code = -1; + return true; + } + if (strncmp(arg, "-@", 2) == 0 && arg[2] != '\0') { + if (set_nonneg_int_option(&config->modify_window, arg + 2, "-@") != 0) + ctx->exit_code = -1; + return true; + } + /* --stop-after/--stop-at are client-only sender-side stop deadlines. They + * are parsed by stop_condition (so the unit tests exercise the same validate + * that production uses) and never serialized into the config frame. */ + if (strncmp(arg, "--stop-after=", 13) == 0) { + if (!stop_parse_after_minutes(arg + 13, &config->stop_after_mins)) { + log_message(LOG_LEVEL_ERROR, "--stop-after must be a positive number of minutes"); + ctx->exit_code = -1; + } + return true; + } + if (strcmp(arg, "--stop-after") == 0) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for --stop-after"); + ctx->exit_code = -1; + return true; + } + if (!stop_parse_after_minutes(ctx->argv[++ctx->i], &config->stop_after_mins)) { + log_message(LOG_LEVEL_ERROR, "--stop-after must be a positive number of minutes"); + ctx->exit_code = -1; + } + return true; + } + if (strncmp(arg, "--stop-at=", 10) == 0) { + if (!stop_parse_at_time(arg + 10, time(NULL), &config->stop_at)) { + log_message(LOG_LEVEL_ERROR, "--stop-at must be HH:MM[:SS] or now+N[smhd]"); + ctx->exit_code = -1; + return true; + } + config->stop_at_set = true; + return true; + } + if (strcmp(arg, "--stop-at") == 0) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for --stop-at"); + ctx->exit_code = -1; + return true; + } + if (!stop_parse_at_time(ctx->argv[++ctx->i], time(NULL), &config->stop_at)) { + log_message(LOG_LEVEL_ERROR, "--stop-at must be HH:MM[:SS] or now+N[smhd]"); + ctx->exit_code = -1; + return true; + } + config->stop_at_set = true; + return true; + } + const char* threads_prefix = "--compress-threads="; + if (strncmp(arg, threads_prefix, strlen(threads_prefix)) == 0) { + if (set_compression_threads_option(&config->compression_threads, + arg + strlen(threads_prefix)) != 0) + ctx->exit_code = -1; + return true; + } + if (strncmp(arg, "--max-alloc=", 12) == 0 || strcmp(arg, "--max-alloc") == 0) { + const char* value = strcmp(arg, "--max-alloc") == 0 ? "" : arg + 12; + if (*value == '\0') { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for --max-alloc"); + ctx->exit_code = -1; + return true; + } + value = ctx->argv[++ctx->i]; + } + if (parse_size_arg(value, &config->max_alloc) != 0) { + log_message(LOG_LEVEL_ERROR, "--max-alloc must be a positive size (B, K, M, G, T, P, or E)"); + ctx->exit_code = -1; + } + return true; + } + return false; +} + +/* Options that map directly onto a Config field through OPTION_TABLE, plus the + * derived implications those options trigger. Returns true when an entry + * matched. */ +static bool cli_handle_table_option(CliParseCtx* ctx) { + Config* config = ctx->config; + const char* arg = ctx->argv[ctx->i]; + const OptionEntry* entry = find_table_option(arg); + const char* inline_value = NULL; + if (!entry) + entry = find_table_option_with_equals(arg, &inline_value); + if (!entry) + return false; + const char* value = NULL; + if (entry->kind != OPT_FLAG) { + value = inline_value; + if (!value && ctx->i + 1 < ctx->argc) + value = ctx->argv[++ctx->i]; + if (!value) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", entry->name); + ctx->exit_code = -1; + return true; + } + if (strcmp(entry->name, "--compress-choice") == 0) { + if (set_compression_choice(config, value) != 0) { + ctx->exit_code = -1; + return true; + } + } else { + if (apply_table_option(config, entry, value) != 0) { + ctx->exit_code = -1; + return true; + } + if (strcmp(entry->name, "--compress-level") == 0 && + (config->compression_level < 1 || config->compression_level > 22)) { + log_message(LOG_LEVEL_ERROR, "--compress-level must be between 1 and 22"); + ctx->exit_code = -1; + return true; + } + if (entry->offset == offsetof(Config, chmod_spec)) { + mode_t ignored; + if (!chmod_apply(0, config->chmod_spec, &ignored)) { + log_message(LOG_LEVEL_ERROR, "--chmod has invalid permission changes"); + ctx->exit_code = -1; + return true; + } + config->use_metadata = true; + } + } + } else if (apply_table_option(config, entry, NULL) != 0) { + ctx->exit_code = -1; + return true; + } + if (entry->offset == offsetof(Config, eight_bit_output)) + protocol_set_8_bit_output(true); + /* A delete-timing flag selects when --delete removes extras, so it + implies --delete exactly like the rsync options do. */ + if (entry->offset == offsetof(Config, delete_before) || + entry->offset == offsetof(Config, delete_during) || + entry->offset == offsetof(Config, delete_delay) || + entry->offset == offsetof(Config, delete_after)) + config->use_delete = true; + /* --delete-missing-args implies --ignore-missing-args (missing entries + are skipped for deletion instead of failing the run). The implication + is order-independent because it is applied over the final parsed + config. */ + if (entry->offset == offsetof(Config, delete_missing_args)) + config->ignore_missing_args = true; + /* -U/--atimes and -N/--crtimes carry their times inside the metadata + payload, which is only transmitted when use_metadata is set, so either + one implies metadata transmission. This is FastSync's broad -M bundle + (mode/mtime travel too); it does NOT enable ownership application, + which stays opt-in via the identity flags. */ + if (entry->offset == offsetof(Config, preserve_atimes) || + entry->offset == offsetof(Config, preserve_crtimes)) + config->use_metadata = true; + if (entry->offset == offsetof(Config, preserve_xattrs) || + entry->offset == offsetof(Config, preserve_acls)) { + config->use_metadata = true; + config->use_xattrs = config->preserve_acls || config->preserve_xattrs; + } + if (entry->offset == offsetof(Config, fake_super)) + config->use_metadata = true; + return true; +} + +/* The inline "--chmod=SPEC" form (kept as its own handler because it bypasses + * the table's OPT_STRING storage). Returns true when the argument was + * consumed. */ +static bool cli_handle_inline_chmod(CliParseCtx* ctx) { + Config* config = ctx->config; + const char* arg = ctx->argv[ctx->i]; + if (strncmp(arg, "--chmod=", 8) != 0) + return false; + if (set_string_option(&config->chmod_spec, arg + 8, "--chmod") != 0) { + ctx->exit_code = -1; + return true; + } + mode_t ignored; + if (!chmod_apply(0, config->chmod_spec, &ignored)) { + log_message(LOG_LEVEL_ERROR, "--chmod has invalid permission changes"); + ctx->exit_code = -1; + return true; + } + config->use_metadata = true; + return true; +} + +/* Help/version and the short archive-style flags. Returns true when the + * argument was consumed. */ +static bool cli_handle_meta_flags(CliParseCtx* ctx) { + Config* config = ctx->config; + const char* arg = ctx->argv[ctx->i]; + if (opt_is(arg, "--help", NULL)) { + print_usage(); + ctx->exit_code = 1; + return true; + } + if (opt_is(arg, "-V", "--version")) { + printf("fastsync version %s\n", PROTOCOL_VERSION); + ctx->exit_code = 1; + return true; + } + if (opt_is(arg, "-D", NULL)) { + /* rsync -D == --devices --specials. -D is otherwise unassigned in + FastSync (verified: no collision), so it is free to imply both. */ + config->preserve_devices = true; + config->preserve_specials = true; + log_info_message(LOG_INFO_MISC, "Enabled preservation of device and special files (-D)"); + return true; + } + if (opt_is(arg, "-a", "--archive")) { + /* Real rsync archive (-rlptgoD). FastSync is always recursive and always + * preserves hard-link/other transfer semantics per its own flags, so -a + * implies links, full metadata (perms/times/group/owner as FastSync's + * broad bundle), devices and specials. Compression and multithreading + * are NOT implied (they are no longer part of archive mode). */ + config->follow_symlinks = true; + config->use_metadata = true; + config->preserve_devices = true; + config->preserve_specials = true; + log_info_message(LOG_INFO_MISC, + "Enabled archive mode (-rlptgoD: links, metadata, devices, specials)"); + return true; + } + if (opt_is(arg, "-p", "--perms")) { + /* rsync -p/--perms: preserve permission bits. Folded into FastSync's + * broad metadata bundle (mode/mtime travel together). */ + config->use_metadata = true; + log_info_message(LOG_INFO_MISC, "Enabled permission preservation"); + return true; + } + return false; +} + +/* SSH port and pattern/block-size options. Returns true when the argument was + * consumed. */ +static bool cli_handle_ssh_and_pattern_options(CliParseCtx* ctx) { + Config* config = ctx->config; + const char* arg = ctx->argv[ctx->i]; + if (opt_is(arg, "--ssh-port", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (set_positive_int_option(&config->ssh_port, ctx->argv[++ctx->i], "--ssh-port") != 0) { + ctx->exit_code = -1; + return true; + } + if (config->ssh_port > 65535) { + log_message(LOG_LEVEL_ERROR, "SSH port must be 1-65535"); + ctx->exit_code = -1; + } + return true; + } + if (strncmp(arg, "--ssh-port=", 11) == 0) { + if (set_positive_int_option(&config->ssh_port, arg + 11, "--ssh-port") != 0) { + ctx->exit_code = -1; + return true; + } + if (config->ssh_port > 65535) { + log_message(LOG_LEVEL_ERROR, "SSH port must be 1-65535"); + ctx->exit_code = -1; + } + return true; + } + if (opt_is(arg, "--exclude", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (config_add_pattern(&config->exclude_patterns, &config->exclude_count, ctx->argv[++ctx->i], + "--exclude") != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--include", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (config_add_pattern(&config->include_patterns, &config->include_count, ctx->argv[++ctx->i], + "--include") != 0) + ctx->exit_code = -1; + return true; + } + if (strncmp(arg, "--delta-block=", 14) == 0) { + if (set_delta_block_size(config, arg + 14) != 0) + ctx->exit_code = -1; + return true; + } + if (strncmp(arg, "--block-size=", 13) == 0) { + if (set_delta_block_size(config, arg + 13) != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--delta-block", "--block-size")) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (set_delta_block_size(config, ctx->argv[++ctx->i]) != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--delta-max", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + unsigned long long val; + if (parse_ull_arg(ctx->argv[++ctx->i], &val, "--delta-max") != 0) { + ctx->exit_code = -1; + return true; + } + if (val >= DELTA_MIN_FILE_SIZE) + config->delta_max_file_size = val; + else + log_message(LOG_LEVEL_WARNING, "--delta-max value %llu too small, using default", val); + return true; + } + return false; +} + +/* Transfer-behavior flags that only toggle a Config field (plus their info + * log lines). Returns true when the argument was consumed. */ +static bool cli_handle_transfer_flags(CliParseCtx* ctx) { + Config* config = ctx->config; + const char* arg = ctx->argv[ctx->i]; + if (opt_is(arg, "-z", "--compress")) { + config->use_compression = + !config->compress_choice || strcmp(config->compress_choice, "zstd") == 0; + log_info_message(LOG_INFO_MISC, "Enabled Compression"); + if (ctx->i + 1 < ctx->argc) { + char* end_ptr; + long level = strtol(ctx->argv[ctx->i + 1], &end_ptr, 10); + if (*end_ptr == '\0') { + if (level < 1 || level > 22) { + log_message(LOG_LEVEL_ERROR, "compression level must be 1-22"); + ctx->exit_code = -1; + return true; + } + config->compression_level = (int)level; + log_info_message(LOG_INFO_MISC, "Set Compression level to %ld", level); + ctx->i++; + } + } + return true; + } + if (opt_is(arg, "--preserve", NULL)) { + config->use_metadata = true; + log_info_message(LOG_INFO_MISC, "Enabled metadata preservation"); + return true; + } + if (opt_is(arg, "-E", "--executability")) { + config->use_metadata = true; + config->use_executability = true; + log_info_message(LOG_INFO_MISC, "Enabled executable permission preservation"); + return true; + } + if (opt_is(arg, "--sendfile", NULL)) { + config->use_sendfile = true; + log_info_message(LOG_INFO_MISC, "Enabled sendfile"); + return true; + } + if (opt_is(arg, "-j", "--threads")) { + config->use_multithreading = true; + log_info_message(LOG_INFO_MISC, "Enabled Multithreading"); + return true; + } + if (opt_is(arg, "--chunk-serialization", NULL)) { + config->use_chunk_serialization = true; + log_info_message(LOG_INFO_MISC, "Enabled Chunk Serialization"); + return true; + } + return false; +} + +/* Network/IO options: --server-port, --bwlimit, --chunk-size, --log-file and + * --stderr. Returns true when the argument was consumed. */ +static bool cli_handle_io_options(CliParseCtx* ctx) { + Config* config = ctx->config; + const char* arg = ctx->argv[ctx->i]; + if (opt_is(arg, "--server-port", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (!parse_positive_int(ctx->argv[++ctx->i], &config->server_port)) { + char* escaped = output_escape(ctx->argv[ctx->i], false); + log_message(LOG_LEVEL_ERROR, "invalid --server-port value: %s", + escaped ? escaped : ""); + free(escaped); + ctx->exit_code = -1; + return true; + } + if (config->server_port > 65535) { + log_message(LOG_LEVEL_ERROR, "server port must be 1-65535"); + ctx->exit_code = -1; + } + return true; + } + if (opt_is(arg, "--bwlimit", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + unsigned long long kbps; + if (parse_ull_arg(ctx->argv[++ctx->i], &kbps, "--bwlimit") != 0) { + ctx->exit_code = -1; + return true; + } + if (kbps == 0) { + log_message(LOG_LEVEL_ERROR, "--bwlimit must be a positive integer"); + ctx->exit_code = -1; + return true; + } + if (kbps > ULLONG_MAX / 1024) { + log_message(LOG_LEVEL_ERROR, "--bwlimit value too large"); + ctx->exit_code = -1; + return true; + } + io_set_bwlimit(kbps * 1024); + log_info_message(LOG_INFO_MISC, "Set bandwidth limit to %llu KB/s", kbps); + return true; + } + if (opt_is(arg, "--chunk-size", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + unsigned long long val; + if (parse_ull_arg(ctx->argv[++ctx->i], &val, "--chunk-size") != 0) { + ctx->exit_code = -1; + return true; + } + if (val == 0) { + log_message(LOG_LEVEL_ERROR, "--chunk-size must be a positive integer"); + ctx->exit_code = -1; + return true; + } + config->chunk_size = val; + return true; + } + if (opt_is(arg, "--log-file", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (config->log_file) { + fclose(config->log_file); + config->log_file = NULL; + log_set_file(NULL); + } + FILE* lf = fopen(ctx->argv[++ctx->i], "a"); + if (!lf) { + char* escaped = output_escape(ctx->argv[ctx->i], false); + log_message(LOG_LEVEL_ERROR, "could not open log file '%s': %s", + escaped ? escaped : "", strerror(errno)); + free(escaped); + ctx->exit_code = -1; + return true; + } + config->log_file = lf; + log_set_file(lf); + return true; + } + if (strncmp(arg, "--stderr=", 9) == 0) { + if (set_stderr_mode(arg + 9) != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--stderr", NULL)) { + if (ctx->i + 1 >= ctx->argc || set_stderr_mode(ctx->argv[++ctx->i]) != 0) + ctx->exit_code = -1; + return true; + } + return false; +} + +/* Filter / files-from options. Returns true when the argument was consumed. */ +static bool cli_handle_filter_options(CliParseCtx* ctx) { + Config* config = ctx->config; + const char* arg = ctx->argv[ctx->i]; + if (opt_is(arg, "--exclude-from", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (read_patterns_from_file(ctx->argv[++ctx->i], &config->exclude_patterns, + &config->exclude_count) != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--include-from", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (read_patterns_from_file(ctx->argv[++ctx->i], &config->include_patterns, + &config->include_count) != 0) + ctx->exit_code = -1; + return true; + } + if (strncmp(arg, "--filter=", 9) == 0) { + if (config_add_filter(config, arg + 9) != 0) + ctx->exit_code = -1; + return true; + } + if (strncmp(arg, "-f=", 3) == 0) { + if (config_add_filter(config, arg + 3) != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--filter", "-f")) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (config_add_filter(config, ctx->argv[++ctx->i]) != 0) + ctx->exit_code = -1; + return true; + } + if (strncmp(arg, "--files-from=", 13) == 0) { + if (set_string_option(&config->files_from, arg + 13, "--files-from") != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--files-from", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (set_string_option(&config->files_from, ctx->argv[++ctx->i], "--files-from") != 0) + ctx->exit_code = -1; + return true; + } + return false; +} + +/* Logging/verbosity options: -v/--verbose, -q/--quiet, --debug, --info and + * --skip-compress. Returns true when the argument was consumed. */ +static bool cli_handle_logging_options(CliParseCtx* ctx) { + Config* config = ctx->config; + const char* arg = ctx->argv[ctx->i]; + if (opt_is(arg, "-v", "--verbose")) { + ctx->verbose = true; + set_log_level(LOG_LEVEL_DEBUG); + return true; + } + if (opt_is(arg, "-q", "--quiet")) { + config->quiet = true; + return true; + } + if (strncmp(arg, "--debug=", 8) == 0) { + int debug_ret = parse_debug_flags(arg + 8, config); + if (debug_ret != 0) + ctx->exit_code = debug_ret; + return true; + } + if (opt_is(arg, "--debug", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + ctx->exit_code = parse_debug_flags(NULL, config); + return true; + } + int debug_ret = parse_debug_flags(ctx->argv[++ctx->i], config); + if (debug_ret != 0) + ctx->exit_code = debug_ret; + return true; + } + if (strncmp(arg, "--info=", 7) == 0) { + if (parse_info_flags(arg + 7, config) != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--info", NULL)) { + if (ctx->i + 1 >= ctx->argc || parse_info_flags(ctx->argv[++ctx->i], config) != 0) + ctx->exit_code = -1; + return true; + } + if (strncmp(arg, "--skip-compress=", 16) == 0) { + if (parse_skip_compress(config, arg + 16) != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--skip-compress", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (parse_skip_compress(config, ctx->argv[++ctx->i]) != 0) + ctx->exit_code = -1; + return true; + } + return false; +} + +/* Checksum/socket options: --compress-threads, --checksum-choice, --cc, + * --checksum-seed and --sockopts. Returns true when the argument was + * consumed. */ +static bool cli_handle_checksum_options(CliParseCtx* ctx) { + Config* config = ctx->config; + const char* arg = ctx->argv[ctx->i]; + if (opt_is(arg, "--compress-threads", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (set_compression_threads_option(&config->compression_threads, ctx->argv[++ctx->i]) != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--checksum-choice", "--cc")) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (set_checksum_choice(config, ctx->argv[++ctx->i]) != 0) + ctx->exit_code = -1; + return true; + } + if (strncmp(arg, "--checksum-choice=", 18) == 0) { + if (set_checksum_choice(config, arg + 18) != 0) + ctx->exit_code = -1; + return true; + } + if (strncmp(arg, "--cc=", 5) == 0) { + if (set_checksum_choice(config, arg + 5) != 0) + ctx->exit_code = -1; + return true; + } + if (strncmp(arg, "--checksum-seed=", 16) == 0) { + if (set_checksum_seed(config, arg + 16) != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--checksum-seed", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for --checksum-seed"); + ctx->exit_code = -1; + return true; + } + if (set_checksum_seed(config, ctx->argv[++ctx->i]) != 0) + ctx->exit_code = -1; + return true; + } + if (strncmp(arg, "--sockopts=", 11) == 0) { + if (set_sockopts_option(config, arg + 11) != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--sockopts", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for --sockopts"); + ctx->exit_code = -1; + return true; + } + if (set_sockopts_option(config, ctx->argv[++ctx->i]) != 0) + ctx->exit_code = -1; + return true; + } + return false; +} + +/* Remote-option, basis-directory and identity-mapping options. Returns true + * when the argument was consumed. */ +static bool cli_handle_remote_basis_options(CliParseCtx* ctx) { + Config* config = ctx->config; + const char* arg = ctx->argv[ctx->i]; + if (strncmp(arg, "--remote-option=", 16) == 0) { + if (config_add_remote_option(config, arg + 16, "--remote-option") != 0) + ctx->exit_code = -1; + return true; + } + if (strncmp(arg, "-M=", 3) == 0) { + if (config_add_remote_option(config, arg + 3, "-M") != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--remote-option", "-M")) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for --remote-option"); + ctx->exit_code = -1; + return true; + } + if (config_add_remote_option(config, ctx->argv[++ctx->i], "--remote-option") != 0) + ctx->exit_code = -1; + return true; + } + if (strncmp(arg, "--compare-dest=", 15) == 0) { + if (set_basis_dest_option(config, BASIS_DEST_COMPARE, arg + 15, "--compare-dest") != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--compare-dest", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (set_basis_dest_option(config, BASIS_DEST_COMPARE, ctx->argv[++ctx->i], "--compare-dest") != + 0) + ctx->exit_code = -1; + return true; + } + if (strncmp(arg, "--copy-dest=", 12) == 0) { + if (set_basis_dest_option(config, BASIS_DEST_COPY, arg + 12, "--copy-dest") != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--copy-dest", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (set_basis_dest_option(config, BASIS_DEST_COPY, ctx->argv[++ctx->i], "--copy-dest") != 0) + ctx->exit_code = -1; + return true; + } + if (strncmp(arg, "--link-dest=", 12) == 0) { + if (set_basis_dest_option(config, BASIS_DEST_LINK, arg + 12, "--link-dest") != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--link-dest", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (set_basis_dest_option(config, BASIS_DEST_LINK, ctx->argv[++ctx->i], "--link-dest") != 0) + ctx->exit_code = -1; + return true; + } + if (strncmp(arg, "--usermap=", 10) == 0) { + if (identity_parse_map(config, arg + 10, false) != 0) { + ctx->exit_code = -1; + return true; + } + config->use_metadata = true; + return true; + } + if (opt_is(arg, "--usermap", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (identity_parse_map(config, ctx->argv[++ctx->i], false) != 0) { + ctx->exit_code = -1; + return true; + } + config->use_metadata = true; + return true; + } + if (strncmp(arg, "--groupmap=", 11) == 0) { + if (identity_parse_map(config, arg + 11, true) != 0) { + ctx->exit_code = -1; + return true; + } + config->use_metadata = true; + return true; + } + if (opt_is(arg, "--groupmap", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (identity_parse_map(config, ctx->argv[++ctx->i], true) != 0) { + ctx->exit_code = -1; + return true; + } + config->use_metadata = true; + return true; + } + if (strncmp(arg, "--chown=", 8) == 0) { + if (identity_parse_chown(config, arg + 8) != 0) { + ctx->exit_code = -1; + return true; + } + config->use_metadata = true; + return true; + } + if (opt_is(arg, "--chown", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (identity_parse_chown(config, ctx->argv[++ctx->i]) != 0) { + ctx->exit_code = -1; + return true; + } + config->use_metadata = true; + return true; + } + if (strncmp(arg, "--copy-as=", 10) == 0) { + if (identity_parse_copy_as(config, arg + 10) != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--copy-as", NULL)) { + if (ctx->i + 1 >= ctx->argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); + ctx->exit_code = -1; + return true; + } + if (identity_parse_copy_as(config, ctx->argv[++ctx->i]) != 0) + ctx->exit_code = -1; + return true; + } + return false; +} + +/* --outbuf. Returns true when the argument was consumed. */ +static bool cli_handle_outbuf_option(CliParseCtx* ctx) { + Config* config = ctx->config; + const char* arg = ctx->argv[ctx->i]; + if (strncmp(arg, "--outbuf=", 9) == 0) { + if (set_outbuf_option(config, arg + 9) != 0) + ctx->exit_code = -1; + return true; + } + if (opt_is(arg, "--outbuf", NULL)) { + if (ctx->i + 1 >= ctx->argc || set_outbuf_option(config, ctx->argv[++ctx->i]) != 0) + ctx->exit_code = -1; + return true; + } + return false; +} + +/* Post-parse lowering: derive implied options over the final parsed config and + * load --files-from once every argument has been seen. Returns 0 on success, + * -1 on error. */ +static int cli_finalize_config(Config* config, bool verbose, bool no_delta, bool no_incremental) { set_log_level(config->quiet ? LOG_LEVEL_ERROR : (verbose ? LOG_LEVEL_DEBUG : LOG_LEVEL_WARNING)); if (config->compress_choice) config->use_compression = strcmp(config->compress_choice, "zstd") == 0; @@ -1538,6 +1843,63 @@ int parse_args(Config* config, int argc, char* argv[], int* positional_args, return 0; } +/* Parse CLI arguments into config. Returns 0 on success, -1 on error, 1 for help/clean-exit. */ +int parse_args(Config* config, int argc, char* argv[], int* positional_args, + int* positional_count) { + protocol_set_8_bit_output(config->eight_bit_output); + + if (cli_apply_output_controls(config, argc, argv) != 0) + return -1; + + CliParseCtx ctx = { + .config = config, + .argc = argc, + .argv = argv, + .positional_args = positional_args, + .positional_count = positional_count, + .i = 1, + .exit_code = 0, + .verbose = false, + .no_delta = false, + .no_incremental = false, + }; + + for (ctx.i = 1; ctx.i < argc; ctx.i++) { + ctx.exit_code = 0; + bool handled = cli_handle_pre_negation(&ctx) || cli_handle_range_time_options(&ctx) || + cli_handle_table_option(&ctx) || cli_handle_inline_chmod(&ctx) || + cli_handle_meta_flags(&ctx) || cli_handle_ssh_and_pattern_options(&ctx) || + cli_handle_transfer_flags(&ctx) || cli_handle_io_options(&ctx) || + cli_handle_filter_options(&ctx) || cli_handle_logging_options(&ctx) || + cli_handle_checksum_options(&ctx) || cli_handle_remote_basis_options(&ctx) || + cli_handle_outbuf_option(&ctx); + if (handled) { + if (ctx.exit_code != 0) + return ctx.exit_code; + continue; + } + + if (argv[ctx.i][0] == '-') { + char* escaped = output_escape(argv[ctx.i], false); + fprintf(stderr, "Unknown option: %s\n", escaped ? escaped : ""); + free(escaped); + print_usage(); + return -1; + } + if (*positional_count < 2) + positional_args[(*positional_count)++] = ctx.i; + else { + char* escaped = output_escape(argv[ctx.i], false); + fprintf(stderr, "Unexpected argument: %s\n", escaped ? escaped : ""); + free(escaped); + print_usage(); + return -1; + } + } + + return cli_finalize_config(config, ctx.verbose, ctx.no_delta, ctx.no_incremental); +} + static int read_patterns_from_file(const char* filepath, char*** patterns, int* count) { FILE* fp = fopen(filepath, "r"); if (!fp) { From 84b7cb0de3620f614955669db8397bf0616c1997 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sat, 12 Sep 2026 20:56:33 +0200 Subject: [PATCH 004/155] refactor(server): split server_module_gate into ordered helper stages --- src/server/server.c | 273 ++++++++++++++++++++++++++------------------ 1 file changed, 165 insertions(+), 108 deletions(-) diff --git a/src/server/server.c b/src/server/server.c index 066b817..77b1e8b 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -251,6 +251,155 @@ static bool configure_authorization(const char* root) { return true; } +/* Discriminates the outcome of the A7 auth gate so the dispatcher can map it + * back to the config_receive_with_validate contract: accepted (including + * "module needs no auth"), a config-level refusal carrying an error string, or + * a handshake that already wrote its own terminal status frame. */ +typedef enum { + MODULE_AUTH_ACCEPTED = 0, + MODULE_AUTH_REFUSED, + MODULE_AUTH_TERMINATED, +} ModuleAuthResult; + +/* Looks up the daemon module selected by the client's config frame and rejects + * a `read only` one (every FastSync network transfer writes; there is no + * read-only wire operation yet). Returns the module, or NULL with *error set + * to the caller-facing rejection message. */ +static const DaemonModule* module_gate_lookup_module(const Config* config, const char** error) { + const DaemonModule* module = daemon_conf_find_module(g_daemon_conf, config->module); + if (module == NULL) { + char* escaped_module = output_escape(config->module, config->eight_bit_output); + log_message(LOG_LEVEL_ERROR, "unknown daemon module '%s' requested", + escaped_module ? escaped_module : ""); + free(escaped_module); + *error = "requested daemon module does not exist"; + return NULL; + } + if (module->read_only) { + log_message(LOG_LEVEL_ERROR, "daemon module '%s' is read only; refusing write transfer", + config->module); + *error = "requested daemon module is read only"; + return NULL; + } + return module; +} + +/* Per-module client-chosen ownership / super-user policy (P7 Wave E hardening): + * a daemon module refuses EVERY ownership-affecting request (--numeric-ids, + * --chown, --usermap/--groupmap, --fake-super, --copy-as, explicit --super) + * unless the operator opted THIS module in with `client owner = yes`. + * Otherwise any client could force arbitrary ownership inside the module root. + * The ownership check is evaluated against the ORIGINAL config so an explicit + * --super is refused even when an operator --no-super veto already forced the + * effective copy to OFF (the veto must not silently convert a refusal into an + * accept); when no ownership flag is present, super-user DEVICE activities are + * forced off for this connection instead. Returns an error string on refusal, + * NULL on acceptance. */ +static const char* module_gate_check_ownership(const Config* config, const DaemonModule* module, + ModuleGateContext* gate_ctx) { + if (module->client_owner) + return NULL; + /* Ownership: refuse the whole transfer up front (a clear failure). */ + if (identity_ownership_requested(config)) { + log_message(LOG_LEVEL_ERROR, + "daemon module '%s' refuses client-chosen ownership/super-user activities " + "(no `client owner = yes` opt-in); refusing", + config->module); + return "client-chosen ownership is not permitted by this daemon module"; + } + /* Super-user DEVICE activities (char/block mknod and --write-devices) are + permitted under the default AUTO mode, so without this override a root + daemon would still let a non-opted module create arbitrary device nodes + and write raw devices. Force them off for this connection: those entries + are skipped (never mknod'ed) while an ordinary `-a` push still succeeds + without device nodes, matching the operator's least-privilege choice. + The operator-level --no-super veto is already folded into this. */ + if (gate_ctx) + gate_ctx->super_mode_override = SUPER_MODE_OFF; + return NULL; +} + +/* A7 auth gate: runs the SCRAM challenge/response for an auth-required module + * BEFORE the module root is installed and before any data moves. Returns + * MODULE_AUTH_ACCEPTED when the module needs no auth or the handshake succeeds, + * MODULE_AUTH_REFUSED with *error set on a config-level rejection, or + * MODULE_AUTH_TERMINATED when the handshake already wrote a terminal status. */ +static ModuleAuthResult module_gate_authenticate(const Config* config, const DaemonModule* module, + ModuleGateContext* gate_ctx, const char** error) { + if (module->auth_user_count == 0) + return MODULE_AUTH_ACCEPTED; + /* Fail closed: no store -> refuse (server misconfiguration, STATUS_ERROR). */ + if (g_credentials == NULL) { + log_message(LOG_LEVEL_ERROR, + "daemon module '%s' requires authentication but no credential store is " + "configured (--password-file/--early-input); refusing", + config->module); + *error = "requested daemon module requires authentication and no credential " + "store is configured"; + return MODULE_AUTH_REFUSED; + } + /* Transport policy (A7-3/S1): an auth-required module only accepts + * credentials over (a) an encrypted, verified TLS connection whose client + * certificate matches --client-cn, or (b) an actual PLAINTEXT connection + * from a loopback peer that the operator explicitly opted into with + * --allow-unauthenticated. A remote plaintext peer, an un-flagged loopback + * plaintext peer, and a loopback TLS peer whose certificate does not match + * --client-cn are all refused HERE, before the challenge is sent, so an + * unverified client never receives a nonce: the loopback allowance requires + * !gate_ctx->ssl, so --tls + --allow-unauthenticated can never be used to + * bypass the client-CN check. The operator flag never permits REMOTE + * plaintext auth: remote peers still require verified TLS regardless. */ + bool tls_ok = gate_ctx && gate_ctx->ssl && SSL_get_verify_result(gate_ctx->ssl) == X509_V_OK && + tls_client_identity_allowed(gate_ctx->ssl); + bool local_ok = allow_unauthenticated && gate_ctx && !gate_ctx->ssl && gate_ctx->fd >= 0 && + utils_fd_peer_is_local(gate_ctx->fd); + if (!tls_ok && !local_ok) { + log_message(LOG_LEVEL_ERROR, + "daemon module '%s' requires authentication over an encrypted, verified TLS " + "connection (or an opted-in loopback plaintext transport); refusing", + config->module); + *error = "daemon module requires authentication over an encrypted, verified TLS " + "connection"; + return MODULE_AUTH_REFUSED; + } + /* Belt-and-braces: the transport policy above already guarantees a context + * with a usable socket (verified TLS implies a live SSL object and loopback + * allowance requires gate_ctx->fd >= 0), so this is unreachable today; keep + * the guard so the handshake can never be driven over an invalid fd. */ + if (!gate_ctx || gate_ctx->fd < 0) { + log_message(LOG_LEVEL_ERROR, "daemon module '%s': no auth transport available", config->module); + *error = "authentication failed for the requested daemon module"; + return MODULE_AUTH_REFUSED; + } + /* The handshake writes exactly one terminal status on failure and signals so + * via MODULE_AUTH_TERMINATED; the username may be logged (never the password + * or any derived proof). */ + if (!server_auth_handshake(gate_ctx->fd, config, module)) { + char* escaped_user = + config->auth_user ? output_escape(config->auth_user, config->eight_bit_output) : NULL; + log_message(LOG_LEVEL_ERROR, "daemon module '%s': authentication failed for user '%s'", + config->module, escaped_user ? escaped_user : "(none)"); + free(escaped_user); + return MODULE_AUTH_TERMINATED; + } + char* escaped_user = output_escape(config->auth_user, config->eight_bit_output); + log_message(LOG_LEVEL_INFO, "daemon module '%s': user '%s' authenticated", config->module, + escaped_user ? escaped_user : ""); + free(escaped_user); + return MODULE_AUTH_ACCEPTED; +} + +/* Installs the module's configured path as the connection's authorized root. + * Returns an error string when the root is unusable, NULL on success. */ +static const char* module_gate_install_root(const Config* config, const DaemonModule* module) { + if (!configure_authorization(module->path)) { + log_message(LOG_LEVEL_ERROR, "daemon module '%s' path '%s' is not usable", config->module, + module->path ? module->path : "(null)"); + return "requested daemon module root is not usable"; + } + return NULL; +} + /* Config-frame gate (runs inside config_receive_with_validate, BEFORE the * STATUS_OK ack, so a rejected connection is refused at the config handshake * and no file data is ever exchanged). @@ -324,115 +473,23 @@ static const char* server_module_gate(const Config* config, void* context) { return "daemon connection did not select a module (expected a " "host::module/path destination)"; - const DaemonModule* module = daemon_conf_find_module(g_daemon_conf, config->module); - if (module == NULL) { - char* escaped_module = output_escape(config->module, config->eight_bit_output); - log_message(LOG_LEVEL_ERROR, "unknown daemon module '%s' requested", - escaped_module ? escaped_module : ""); - free(escaped_module); - return "requested daemon module does not exist"; + const char* error = NULL; + const DaemonModule* module = module_gate_lookup_module(config, &error); + if (!module) + return error; + error = module_gate_check_ownership(config, module, gate_ctx); + if (error) + return error; + switch (module_gate_authenticate(config, module, gate_ctx, &error)) { + case MODULE_AUTH_REFUSED: + return error; + case MODULE_AUTH_TERMINATED: + return CONFIG_VALIDATE_ALREADY_TERMINATED; + case MODULE_AUTH_ACCEPTED: + break; } - if (module->read_only) { - log_message(LOG_LEVEL_ERROR, "daemon module '%s' is read only; refusing write transfer", - config->module); - return "requested daemon module is read only"; - } - /* Client-chosen ownership / super-user policy (P7 Wave E hardening): a daemon - module refuses EVERY ownership-affecting request (--numeric-ids, --chown, - --usermap/--groupmap, --fake-super, --copy-as, explicit --super) unless the - operator opted THIS module in with `client owner = yes`. Otherwise any - client could force arbitrary ownership inside the module root. The - standalone/SSH server has a single operator-authorized root and keeps - honoring these. */ - if (!module->client_owner) { - /* Ownership: refuse the whole transfer up front (a clear failure). - Evaluated against the ORIGINAL config so an explicit --super is refused - even when an operator --no-super veto already forced the effective copy - to OFF (the veto must not silently convert a refusal into an accept). */ - if (identity_ownership_requested(config)) { - log_message(LOG_LEVEL_ERROR, - "daemon module '%s' refuses client-chosen ownership/super-user activities " - "(no `client owner = yes` opt-in); refusing", - config->module); - return "client-chosen ownership is not permitted by this daemon module"; - } - /* Super-user DEVICE activities (char/block mknod and --write-devices) are - permitted under the default AUTO mode, so without this override a root - daemon would still let a non-opted module create arbitrary device nodes - and write raw devices. Force them off for this connection: those entries - are skipped (never mknod'ed) while an ordinary `-a` push still succeeds - without device nodes, matching the operator's least-privilege choice. - The operator-level --no-super veto is already folded into this. */ - if (gate_ctx) - gate_ctx->super_mode_override = SUPER_MODE_OFF; - } - if (module->auth_user_count > 0) { - /* Auth-required module (A7, protocol 2.19.0): run the SCRAM challenge/ - * response BEFORE the module root is installed and before any data moves. - * Fail closed: no store -> refuse (server misconfiguration, STATUS_ERROR); - * a handshake that fails before the success response writes exactly one - * STATUS_AUTH_FAILED before signalling ALREADY_TERMINATED (a failure while - * writing the success signature instead just drops the broken connection). - * The username may be logged (never the password or any derived proof). */ - if (g_credentials == NULL) { - log_message(LOG_LEVEL_ERROR, - "daemon module '%s' requires authentication but no credential store is " - "configured (--password-file/--early-input); refusing", - config->module); - return "requested daemon module requires authentication and no credential " - "store is configured"; - } - /* Transport policy (A7-3/S1): an auth-required module only accepts - * credentials over (a) an encrypted, verified TLS connection whose client - * certificate matches --client-cn, or (b) an actual PLAINTEXT connection - * from a loopback peer that the operator explicitly opted into with - * --allow-unauthenticated. A remote plaintext peer, an un-flagged loopback - * plaintext peer, and a loopback TLS peer whose certificate does not match - * --client-cn are all refused HERE, before the challenge is sent, so an - * unverified client never receives a nonce: the loopback allowance requires - * !gate_ctx->ssl, so --tls + --allow-unauthenticated can never be used to - * bypass the client-CN check. The operator flag never permits REMOTE - * plaintext auth: remote peers still require verified TLS regardless. */ - bool tls_ok = gate_ctx && gate_ctx->ssl && SSL_get_verify_result(gate_ctx->ssl) == X509_V_OK && - tls_client_identity_allowed(gate_ctx->ssl); - bool local_ok = allow_unauthenticated && gate_ctx && !gate_ctx->ssl && gate_ctx->fd >= 0 && - utils_fd_peer_is_local(gate_ctx->fd); - if (!tls_ok && !local_ok) { - log_message(LOG_LEVEL_ERROR, - "daemon module '%s' requires authentication over an encrypted, verified TLS " - "connection (or an opted-in loopback plaintext transport); refusing", - config->module); - return "daemon module requires authentication over an encrypted, verified TLS " - "connection"; - } - /* Belt-and-braces: the transport policy above already guarantees a context - * with a usable socket (verified TLS implies a live SSL object and loopback - * allowance requires gate_ctx->fd >= 0), so this is unreachable today; keep - * the guard so the handshake can never be driven over an invalid fd. */ - if (!gate_ctx || gate_ctx->fd < 0) { - log_message(LOG_LEVEL_ERROR, "daemon module '%s': no auth transport available", - config->module); - return "authentication failed for the requested daemon module"; - } - if (!server_auth_handshake(gate_ctx->fd, config, module)) { - char* escaped_user = - config->auth_user ? output_escape(config->auth_user, config->eight_bit_output) : NULL; - log_message(LOG_LEVEL_ERROR, "daemon module '%s': authentication failed for user '%s'", - config->module, escaped_user ? escaped_user : "(none)"); - free(escaped_user); - return CONFIG_VALIDATE_ALREADY_TERMINATED; - } - char* escaped_user = output_escape(config->auth_user, config->eight_bit_output); - log_message(LOG_LEVEL_INFO, "daemon module '%s': user '%s' authenticated", config->module, - escaped_user ? escaped_user : ""); - free(escaped_user); - } - if (!configure_authorization(module->path)) { - log_message(LOG_LEVEL_ERROR, "daemon module '%s' path '%s' is not usable", config->module, - module->path ? module->path : "(null)"); - return "requested daemon module root is not usable"; - } - return NULL; /* accepted; authorized root is now the module's path */ + /* accepted; the authorized root is now the module's path */ + return module_gate_install_root(config, module); } void handler(int file_descriptor) { From 37037a6ee7b73c101bd196bf0a6cb6c6432e542a Mon Sep 17 00:00:00 2001 From: TapTap Date: Sat, 12 Sep 2026 21:07:14 +0200 Subject: [PATCH 005/155] refactor(client-cli): drop unused CliParseCtx positional fields (cppcheck) --- src/client/client_cli.c | 4 ---- 1 file changed, 4 deletions(-) diff --git a/src/client/client_cli.c b/src/client/client_cli.c index d1676da..400be1b 100644 --- a/src/client/client_cli.c +++ b/src/client/client_cli.c @@ -826,8 +826,6 @@ typedef struct { Config* config; int argc; char** argv; - int* positional_args; - int* positional_count; int i; /* index of the argument currently being examined */ int exit_code; /* nonzero when a matched handler wants parse_args to return */ bool verbose; /* "-v"/"--verbose" seen (drives the final log level) */ @@ -1855,8 +1853,6 @@ int parse_args(Config* config, int argc, char* argv[], int* positional_args, .config = config, .argc = argc, .argv = argv, - .positional_args = positional_args, - .positional_count = positional_count, .i = 1, .exit_code = 0, .verbose = false, From 59ce174d22a9d29a804e6023fd3e9465a120b0e4 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 00:57:09 +0200 Subject: [PATCH 006/155] fix(client-send): UAF in basis preflight and missing_args leak --- src/client/client_send.c | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/src/client/client_send.c b/src/client/client_send.c index f4db82e..16ed906 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -338,9 +338,10 @@ static bool basis_oversize_preflight(const Config* config) { return false; DirectoryScanner* scanner = directory_scanner_create_with_options(config->send_directory, &prepared.options); - prepared_scanner_destroy(&prepared); - if (!scanner) + if (!scanner) { + prepared_scanner_destroy(&prepared); return false; + } bool ok = true; Chunk* chunk; while ((chunk = directory_scanner_next(scanner)) != NULL) { @@ -364,7 +365,10 @@ static bool basis_oversize_preflight(const Config* config) { } if (directory_scanner_failed(scanner) || directory_scanner_had_io_error(scanner)) ok = false; + /* The scanner borrows prepared.options' base_filters/hardlinks pointers, so + prepared must outlive the scanner. */ directory_scanner_destroy(scanner); + prepared_scanner_destroy(&prepared); return ok; } @@ -1886,6 +1890,8 @@ int send_files(Config* config) { if (config->transport == TRANSPORT_TCP) log_message(LOG_LEVEL_ERROR, "could not connect to server%s", config->use_tls ? " via TLS" : ""); + if (missing_args) + array_list_delete(missing_args); return 1; } ProtocolSession session; From 4557924972a662d8e2440302ad3bbf1f0c131737 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 00:57:32 +0200 Subject: [PATCH 007/155] fix(receiver): cap DirTimeList growth and fix placeholder Data leaks --- src/shared/file_receive.c | 16 +++++++-- src/shared/file_receive.h | 13 ++++++- tests/test_file.c | 72 +++++++++++++++++++++++++++++++++++++++ 3 files changed, 98 insertions(+), 3 deletions(-) diff --git a/src/shared/file_receive.c b/src/shared/file_receive.c index ffb72df..f21867c 100644 --- a/src/shared/file_receive.c +++ b/src/shared/file_receive.c @@ -1696,9 +1696,9 @@ File* receive_incremental_check(int fd, const Config* config, bool* skipped) { return NULL; } - if (has_path_traversal(check_path)) { + if (check_path[0] == '\0' || has_path_traversal(check_path)) { char* escaped_path = output_escape(check_path, log_get_8_bit_output()); - log_message(LOG_LEVEL_ERROR, "Path traversal detected: %s", + log_message(LOG_LEVEL_ERROR, "Invalid received check path: %s", escaped_path ? escaped_path : ""); free(escaped_path); free(check_path); @@ -1832,6 +1832,7 @@ File* receive_incremental_check(int fd, const Config* config, bool* skipped) { existing/ignore-existing/update/backup/delay-updates policy. */ File* materialized = file_create(check_path); if (materialized && basis.content) { + data_destroy(materialized->data); materialized->data = basis.content; basis.content = NULL; materialized->metadata = file_metadata_create(NULL, &basis.st, false, false); @@ -2095,6 +2096,7 @@ File* receive_incremental_check(int fd, const Config* config, bool* skipped) { file->metadata = meta; file->xattrs = append_xattrs; append_xattrs = NULL; + data_destroy(file->data); file->data = data_create(full, full_size); if (!file->data) { /* data_create already freed full on failure */ file_destroy(file); @@ -2247,6 +2249,7 @@ void dir_time_list_init(DirTimeList* list) { list->entries = NULL; list->count = 0; list->capacity = 0; + list->bytes = 0; } void dir_time_list_free(DirTimeList* list) { @@ -2260,11 +2263,19 @@ void dir_time_list_free(DirTimeList* list) { list->entries = NULL; list->count = 0; list->capacity = 0; + list->bytes = 0; } bool dir_time_list_add(DirTimeList* list, const char* wire_path, const FileMetadata* metadata) { if (!list || !wire_path || !metadata) return true; /* nothing to remember; never a hard error */ + /* Cumulative, not per-frame: the sender may stream a tree across unbounded + STATUS_DIR_TIMES frames, so bound the TOTAL retained here. Reject before + touching the list, leaving it exactly as it was (the caller fails the + transfer, which becomes a clean protocol error). */ + size_t path_len = strlen(wire_path); + if (list->count >= MAX_DIR_TIME_ENTRIES || path_len > MAX_DIR_TIME_BYTES - list->bytes) + return false; if (list->count == list->capacity) { size_t new_capacity = list->capacity == 0 ? 16 : list->capacity * 2; if (new_capacity < list->capacity) @@ -2291,6 +2302,7 @@ bool dir_time_list_add(DirTimeList* list, const char* wire_path, const FileMetad list->paths[list->count] = copy; list->entries[list->count] = *metadata; list->count++; + list->bytes += path_len; return true; } diff --git a/src/shared/file_receive.h b/src/shared/file_receive.h index 37dc15a..ec6ccfa 100644 --- a/src/shared/file_receive.h +++ b/src/shared/file_receive.h @@ -7,6 +7,15 @@ /* Server-side file receive/save path. */ +/* Cumulative caps for the deferred directory-time accumulator. The sender may + * legitimately split a large tree across repeated STATUS_DIR_TIMES frames, so a + * per-frame bound is not enough: the receiver must bound the TOTAL it retains + * against a hostile sender. Mirror the delete-manifest limits + * (MAX_MANIFEST_ENTRIES / MAX_MANIFEST_BYTES): the entry count bounds the + * metadata array and the byte budget bounds the concatenated path strings. */ +#define MAX_DIR_TIME_ENTRIES (1024 * 1024) +#define MAX_DIR_TIME_BYTES (16ULL * 1024 * 1024) + File* file_receive(const Config* config, int file_descriptor); File* file_receive_directory(int file_descriptor, const Config* config); File* file_receive_dir_time(int file_descriptor, const Config* config); @@ -28,6 +37,7 @@ typedef struct { FileMetadata* entries; /* owned, parallel to paths */ size_t count; size_t capacity; + size_t bytes; /* cumulative strlen of every retained path */ } DirTimeList; /* Capture gate shared by the sender-side and receiver-side sinks: directory @@ -39,7 +49,8 @@ bool dir_times_should_capture(const Config* config); void dir_time_list_init(DirTimeList* list); void dir_time_list_free(DirTimeList* list); /* Deep-copy one directory's path + metadata into the list. Returns false on - * allocation failure (the caller fails the transfer). */ + * allocation failure OR when the cumulative entry/byte caps would be exceeded + * (the caller fails the transfer). */ bool dir_time_list_add(DirTimeList* list, const char* wire_path, const FileMetadata* metadata); /* Apply every accumulated directory's mtime (and atime when captured) beneath * `root_directory`, confined fd-relative. Best-effort per entry: an absent diff --git a/tests/test_file.c b/tests/test_file.c index fd715fc..0f0225d 100644 --- a/tests/test_file.c +++ b/tests/test_file.c @@ -1370,6 +1370,76 @@ static void test_dir_time_list() { rmdir(root); } +/* A hostile sender can stream unbounded STATUS_DIR_TIMES frames; the + * accumulator must bound the CUMULATIVE path bytes (not just one frame) and + * reject the add that would cross the cap, leaving the list untouched. */ +static void test_dir_time_list_cap() { + DirTimeList list; + dir_time_list_init(&list); + EXPECT_EQ_INT((int)list.bytes, 0); + FileMetadata metadata = {.mtime_sec = 1, .mtime_nsec = 0}; + + size_t path_len = MAX_STRING_SIZE - 1; + char* path = malloc(path_len + 1); + EXPECT_NOT_NULL(path); + memset(path, 'a', path_len); + path[path_len] = '\0'; + + bool rejected = false; + for (size_t i = 0; i < MAX_DIR_TIME_ENTRIES + 1 && !rejected; i++) { + size_t before_count = list.count; + size_t before_bytes = list.bytes; + if (!dir_time_list_add(&list, path, &metadata)) { + rejected = true; + /* The rejected add must not have partially mutated the list. */ + EXPECT_TRUE(list.count == before_count); + EXPECT_TRUE(list.bytes == before_bytes); + } else { + EXPECT_TRUE(list.count == before_count + 1); + EXPECT_TRUE(list.bytes == before_bytes + path_len); + } + } + EXPECT_TRUE(rejected); + EXPECT_TRUE(list.count <= MAX_DIR_TIME_ENTRIES); + EXPECT_TRUE(list.bytes <= MAX_DIR_TIME_BYTES); + + /* The retained entries are still intact and freeable after the rejection. */ + EXPECT_TRUE(list.count > 0); + EXPECT_TRUE(strcmp(list.paths[0], path) == 0); + dir_time_list_free(&list); + EXPECT_EQ_INT((int)list.bytes, 0); + free(path); +} + +/* receive_incremental_check must reject an empty check_path; every other + * receive path rejects path[0]=='\0'. Feed the check header (empty wire path + * + size/mtime/nsec) and assert the check is refused without being skipped. */ +static void test_receive_incremental_check_empty_path() { + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + cfg->checksum = false; + + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + size_t wire_len = 0; + unsigned long long check_size = 0; + long long check_mtime = 0; + long long check_mtime_nsec = 0; + EXPECT_TRUE(send_n_data(p[1], &wire_len, sizeof(wire_len))); + EXPECT_TRUE(send_n_data(p[1], &check_size, sizeof(check_size))); + EXPECT_TRUE(send_n_data(p[1], &check_mtime, sizeof(check_mtime))); + EXPECT_TRUE(send_n_data(p[1], &check_mtime_nsec, sizeof(check_mtime_nsec))); + + bool skipped = true; + File* file = receive_incremental_check(p[0], cfg, &skipped); + EXPECT_NULL(file); + EXPECT_FALSE(skipped); + + close(p[0]); + close(p[1]); + config_delete(cfg); +} + /* -K/--keep-dirlinks secure open: with an authorized root, a destination path * component that is a symlink to an IN-ROOT directory is used as that directory * (its referent is opened through a relative O_NOFOLLOW walk from the root fd, @@ -1531,6 +1601,8 @@ void test_file() { } test_file_metadata_create(); test_dir_time_list(); + test_dir_time_list_cap(); + test_receive_incremental_check_empty_path(); test_keep_dirlinks_secure_open(); test_inplace_overwrite_clears_special_mode_bits(); test_inplace_overwrite_metadata_strips_special_bits(); From f8252cf3e7c726b6345e10d3652f3d59404ce302 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 00:57:32 +0200 Subject: [PATCH 008/155] fix(protocol): bound pre-auth config string memory --- src/shared/config.c | 106 +++++++++++++++++++++++++++++++------------- src/shared/config.h | 18 ++++++++ tests/test_config.c | 85 +++++++++++++++++++++++++++++++++++ 3 files changed, 179 insertions(+), 30 deletions(-) diff --git a/src/shared/config.c b/src/shared/config.c index 4bd72c9..584c61a 100644 --- a/src/shared/config.c +++ b/src/shared/config.c @@ -202,6 +202,51 @@ static bool receive_wire_bool(int fd, bool* value) { return true; } +/* Cumulative budget for the strings retained by one received Config (see + * MAX_CONFIG_STRING_BYTES). Config strings are received once per connection + * before authentication and live for its whole lifetime, so the charge is never + * released. */ +typedef struct { + unsigned long long used; +} ConfigStringBudget; + +/* Charge `bytes` (the retained allocation: string body plus NUL) against the + * aggregate config-string budget. Returns false when the ceiling would be + * exceeded, letting the caller reject the frame with a clear error instead of + * retaining unbounded pre-auth memory. */ +static bool config_string_budget_charge(ConfigStringBudget* budget, size_t bytes) { + if ((unsigned long long)bytes > MAX_CONFIG_STRING_BYTES || + budget->used > MAX_CONFIG_STRING_BYTES - (unsigned long long)bytes) { + log_message(LOG_LEVEL_ERROR, "Config string budget exceeded (%llu + %zu > %llu bytes)", + budget->used, bytes, (unsigned long long)MAX_CONFIG_STRING_BYTES); + return false; + } + budget->used += (unsigned long long)bytes; + return true; +} + +static char* config_receive_str(int fd, ConfigStringBudget* budget) { + char* value = receive_str(fd); + if (!value) + return NULL; + if (!config_string_budget_charge(budget, strlen(value) + 1)) { + free(value); + return NULL; + } + return value; +} + +static char* config_receive_str_redacted(int fd, ConfigStringBudget* budget) { + char* value = receive_str_redacted(fd); + if (!value) + return NULL; + if (!config_string_budget_charge(budget, strlen(value) + 1)) { + free(value); + return NULL; + } + return value; +} + static bool validate_received_config(const Config* config) { return valid_wire_bool(config->save_to_disk) && valid_wire_bool(config->use_multithreading) && valid_wire_bool(config->use_chunk_serialization) && @@ -257,7 +302,7 @@ static bool validate_received_config(const Config* config) { config->delta_block_size <= DELTA_BLOCK_SIZE_MAX && config->delta_max_file_size <= DELTA_MAX_FILE_SIZE && config->modify_window >= 0 && config->max_delete >= -1 && config->skip_compress_count >= 0 && - config->skip_compress_count <= 10000 && config->max_alloc > 0 && + config->skip_compress_count <= MAX_SKIP_COMPRESS_SUFFIXES && config->max_alloc > 0 && (!config->chmod_spec || !*config->chmod_spec || chmod_apply(0, config->chmod_spec, &(mode_t){0})) && /* The received --iconv CONVERT_SPEC is untrusted input that drives @@ -819,7 +864,7 @@ static bool send_checksum_options(int fd, const Config* c) { send_n_data(fd, &c->checksum_seed, sizeof(c->checksum_seed)); } -static bool receive_core_fields(int fd, Config* c) { +static bool receive_core_fields(int fd, Config* c, ConfigStringBudget* budget) { int value; if (!receive_wire_bool(fd, &c->eight_bit_output)) return false; @@ -829,8 +874,8 @@ static bool receive_core_fields(int fd, Config* c) { if (c->max_alloc > MAX_SERVER_ALLOC) c->max_alloc = MAX_SERVER_ALLOC; protocol_session_set_max_alloc(NULL, c->max_alloc); - c->send_directory = receive_str(fd); - c->receive_root_directory = receive_str(fd); + c->send_directory = config_receive_str(fd, budget); + c->receive_root_directory = config_receive_str(fd, budget); if (!c->send_directory || !c->receive_root_directory) return false; if (!receive_wire_bool(fd, &c->save_to_disk) || !receive_wire_bool(fd, &c->use_multithreading) || @@ -863,10 +908,10 @@ static bool receive_delta_fields(int fd, Config* c) { receive_n_data(fd, &c->delta_max_file_size, sizeof(unsigned long long)); } -static bool receive_file_options(int fd, Config* c) { +static bool receive_file_options(int fd, Config* c, ConfigStringBudget* budget) { if (!receive_wire_bool(fd, &c->backup)) return false; - char* backup_dir = receive_str(fd); + char* backup_dir = config_receive_str(fd, budget); if (!backup_dir) return false; if (*backup_dir != '\0') { @@ -920,8 +965,8 @@ static bool receive_selection_options(int fd, Config* c) { return receive_wire_bool(fd, &c->delete_delay); } -static bool receive_resume_options(int fd, Config* c) { - char* temp_dir = receive_str(fd); +static bool receive_resume_options(int fd, Config* c, ConfigStringBudget* budget) { + char* temp_dir = config_receive_str(fd, budget); if (!temp_dir) return false; if (*temp_dir != '\0') { @@ -935,7 +980,7 @@ static bool receive_resume_options(int fd, Config* c) { string for "unset". Canonicalize the empty wire value back to NULL so receivers observe exactly what the client configured (plain --backup, for example, must not look like --backup-dir ""). */ - char* partial_dir = receive_str(fd); + char* partial_dir = config_receive_str(fd, budget); if (!partial_dir) return false; if (*partial_dir != '\0') { @@ -943,7 +988,7 @@ static bool receive_resume_options(int fd, Config* c) { } else { free(partial_dir); } - char* suffix = receive_str(fd); + char* suffix = config_receive_str(fd, budget); if (!suffix) return false; if (*suffix != '\0') { @@ -957,20 +1002,20 @@ static bool receive_resume_options(int fd, Config* c) { return false; if (!receive_n_data(fd, &c->modify_window, sizeof(c->modify_window))) return false; - c->compress_choice = receive_str(fd); + c->compress_choice = config_receive_str(fd, budget); if (!c->compress_choice) return false; - c->chmod_spec = receive_str(fd); + c->chmod_spec = config_receive_str(fd, budget); if (!c->chmod_spec || !receive_wire_bool(fd, &c->skip_compress_set) || !receive_int(fd, &c->skip_compress_count) || c->skip_compress_count < 0 || - c->skip_compress_count > 10000) + c->skip_compress_count > MAX_SKIP_COMPRESS_SUFFIXES) return false; if (c->skip_compress_count > 0) { c->skip_compress_suffixes = calloc((size_t)c->skip_compress_count, sizeof(char*)); if (!c->skip_compress_suffixes) return false; for (int i = 0; i < c->skip_compress_count; i++) { - c->skip_compress_suffixes[i] = receive_str(fd); + c->skip_compress_suffixes[i] = config_receive_str(fd, budget); if (!c->skip_compress_suffixes[i]) return false; } @@ -978,7 +1023,7 @@ static bool receive_resume_options(int fd, Config* c) { return true; } -static bool receive_basis_options(int fd, Config* c) { +static bool receive_basis_options(int fd, Config* c, ConfigStringBudget* budget) { int count; if (!receive_int(fd, &count)) return false; @@ -988,7 +1033,7 @@ static bool receive_basis_options(int fd, Config* c) { int type; if (!receive_int(fd, &type) || type <= BASIS_DEST_NONE || type > BASIS_DEST_LINK) return false; - char* path = receive_str(fd); + char* path = config_receive_str(fd, budget); if (!path) return false; /* config_basis_append validates and canonicalizes the path; a rejected @@ -1119,8 +1164,8 @@ static bool send_daemon_module(int fd, const Config* c) { return send_str(fd, c->module ? c->module : ""); } -static bool receive_daemon_module(int fd, Config* c) { - char* module = receive_str(fd); +static bool receive_daemon_module(int fd, Config* c, ConfigStringBudget* budget) { + char* module = config_receive_str(fd, budget); if (!module) return false; /* Guard against a hostile client flooding the log with an over-long module @@ -1155,14 +1200,14 @@ static bool send_daemon_auth(int fd, const Config* c) { return send_str_redacted(fd, c->auth_user); } -static bool receive_daemon_auth(int fd, Config* c) { +static bool receive_daemon_auth(int fd, Config* c, ConfigStringBudget* budget) { int present; if (!receive_int(fd, &present) || !valid_wire_bool(present)) return false; if (!present) return true; /* Redacted receive: never log the incoming username body. */ - char* user = receive_str_redacted(fd); + char* user = config_receive_str_redacted(fd, budget); if (!user) return false; if (!credentials_username_valid(user)) { @@ -1272,8 +1317,8 @@ static bool send_iconv_spec(int fd, const Config* c) { return send_str(fd, c->iconv_spec ? c->iconv_spec : ""); } -static bool receive_iconv_spec(int fd, Config* c) { - char* spec = receive_str(fd); +static bool receive_iconv_spec(int fd, Config* c, ConfigStringBudget* budget) { + char* spec = config_receive_str(fd, budget); if (!spec) return false; if (*spec == '\0') { @@ -1376,8 +1421,9 @@ Config* config_receive_with_validate(int file_descriptor, ConfigValidateFunc val Config* config = config_create(); if (!config) return NULL; + ConfigStringBudget budget = {0}; free(config->version); - config->version = receive_str(file_descriptor); + config->version = config_receive_str(file_descriptor, &budget); if (!config->version) goto error; if (strcmp(config->version, PROTOCOL_VERSION) != 0) { @@ -1388,21 +1434,21 @@ Config* config_receive_with_validate(int file_descriptor, ConfigValidateFunc val send_status(file_descriptor, STATUS_ERROR); goto error; } - if (!receive_core_fields(file_descriptor, config) || + if (!receive_core_fields(file_descriptor, config, &budget) || !receive_delta_fields(file_descriptor, config) || - !receive_file_options(file_descriptor, config) || + !receive_file_options(file_descriptor, config, &budget) || !receive_selection_options(file_descriptor, config) || - !receive_resume_options(file_descriptor, config) || - !receive_basis_options(file_descriptor, config) || + !receive_resume_options(file_descriptor, config, &budget) || + !receive_basis_options(file_descriptor, config, &budget) || !receive_fuzzy_option(file_descriptor, config) || !receive_checksum_options(file_descriptor, config) || !receive_identity_options(file_descriptor, config) || !receive_metadata_times_options(file_descriptor, config) || !receive_symlink_trust_options(file_descriptor, config) || !receive_phase4_xattr_options(file_descriptor, config) || - !receive_daemon_module(file_descriptor, config) || - !receive_daemon_auth(file_descriptor, config) || - !receive_iconv_spec(file_descriptor, config) || + !receive_daemon_module(file_descriptor, config, &budget) || + !receive_daemon_auth(file_descriptor, config, &budget) || + !receive_iconv_spec(file_descriptor, config, &budget) || !receive_privilege_options(file_descriptor, config) || !receive_copy_as_options(file_descriptor, config)) goto error; diff --git a/src/shared/config.h b/src/shared/config.h index 66fd5eb..d582be2 100644 --- a/src/shared/config.h +++ b/src/shared/config.h @@ -643,6 +643,24 @@ typedef struct Config { /* Upper bound on total basis-dir entries (rsync caps --link-dest at 20). */ #define MAX_BASIS_DIRS 64 +/* Upper bound on the number of --skip-compress suffixes accepted from the wire. + * Each suffix is an independent wire string (up to MAX_STRING_SIZE = 64 KiB), so + * without this a hostile pre-auth client could otherwise retain + * skip_count * MAX_STRING_SIZE bytes on the server before authentication; 256 + * covers any realistic suffix list while keeping the worst case small. */ +#define MAX_SKIP_COMPRESS_SUFFIXES 256 + +/* Aggregate ceiling on the bytes retained by ALL strings in one received config + * frame (version, send/receive roots, backup/temp/partial/suffix, compression + * choice, chmod spec, skip-compress suffixes, basis paths, module, auth user, + * iconv spec, ...). The config frame is parsed BEFORE authentication and every + * one of these strings lives for the whole connection, so this cumulative + * (never released) budget bounds the pre-auth memory a single connection can + * pin. MAX_SKIP_COMPRESS_SUFFIXES / MAX_BASIS_DIRS bound the individual + * repeatable counts; this budget bounds their product and any single oversized + * field. */ +#define MAX_CONFIG_STRING_BYTES (1ULL * 1024 * 1024) + /* Identity-mapping sentinels and bounds (see identity.h for semantics). * IDENTITY_MATCH_ANY is a usermap/groupmap FROM '*' (matches any id); * IDENTITY_CURRENT is a chown / map TO '*' (resolve to the receiver's current diff --git a/tests/test_config.c b/tests/test_config.c index 4ffc49f..003d768 100644 --- a/tests/test_config.c +++ b/tests/test_config.c @@ -6,6 +6,7 @@ #include "queue.h" #include "test_utils.h" #include "utils.h" +#include #include #include #include @@ -1846,6 +1847,89 @@ static void test_config_receive_rejects_copy_as_without_metadata() { config_delete(c); } +/* Like roundtrip_config_ok, but the parent is the RECEIVER so the frame can be + rejected MID-way, before the sender finishes writing it. The sender child + ignores SIGPIPE so the receiver closing early cannot kill it; the parent + waits for the child to exit after observing the rejection. */ +static bool roundtrip_config_rejected(const Config* send_cfg) { + int p[2]; + if (socketpair(AF_UNIX, SOCK_STREAM, 0, p) != 0) + return false; + pid_t pid = fork(); + if (pid == 0) { + (void)signal(SIGPIPE, SIG_IGN); + close(p[0]); + io_set_fds(p[1], p[1]); + config_send(p[1], send_cfg); + close(p[1]); + _exit(0); + } + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool rejected = recv == NULL; + config_delete(recv); + close(p[0]); + int status; + waitpid(pid, &status, 0); + return rejected; +} + +/* Build a Config with `count` --skip-compress suffixes, each `suffix_len` bytes + long, for the pre-auth config-string budget tests. */ +static Config* make_skip_compress_config(int count, size_t suffix_len) { + Config* c = config_create(); + if (!c) + return NULL; + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->skip_compress_set = true; + c->skip_compress_count = count; + c->skip_compress_suffixes = calloc((size_t)count, sizeof(char*)); + if (!c->skip_compress_suffixes) { + config_delete(c); + return NULL; + } + char* suffix = malloc(suffix_len + 1); + if (!suffix) { + config_delete(c); + return NULL; + } + memset(suffix, 'x', suffix_len); + suffix[suffix_len] = '\0'; + for (int i = 0; i < count; i++) + c->skip_compress_suffixes[i] = str_dup(suffix); + free(suffix); + return c; +} + +/* Pre-auth memory bound: one connection must not retain unbounded config + strings. An over-limit --skip-compress count is refused, and even an + in-range count cannot exceed the aggregate per-connection string budget. */ +static void test_config_receive_rejects_oversized_string_budget() { + if (is_running_under_valgrind()) + return; + + /* Exactly MAX_SKIP_COMPRESS_SUFFIXES tiny suffixes are accepted. */ + Config* ok = make_skip_compress_config(MAX_SKIP_COMPRESS_SUFFIXES, 1); + EXPECT_NOT_NULL(ok); + EXPECT_TRUE(roundtrip_config_ok(ok)); + config_delete(ok); + + /* One suffix over the count cap is rejected before any suffix is read. */ + Config* over_count = make_skip_compress_config(MAX_SKIP_COMPRESS_SUFFIXES + 1, 1); + EXPECT_NOT_NULL(over_count); + EXPECT_TRUE(roundtrip_config_rejected(over_count)); + config_delete(over_count); + + /* In-range count, but the strings together exceed MAX_CONFIG_STRING_BYTES + (64 suffixes * ~64 KiB > 1 MiB), so the aggregate budget rejects it. */ + Config* over_bytes = make_skip_compress_config(64, MAX_STRING_SIZE - 1); + EXPECT_NOT_NULL(over_bytes); + EXPECT_TRUE(roundtrip_config_rejected(over_bytes)); + config_delete(over_bytes); +} + /* identity_copy_as_refused() is the pure, pre-snapshot refusal predicate: a --copy-as is refused when the receiver is not root OR the effective super mode is OFF (an operator veto), and never when --copy-as is unset. */ @@ -2009,6 +2093,7 @@ void test_config() { test_config_copy_as_wire_roundtrip(); test_config_receive_rejects_negative_copy_as(); test_config_receive_rejects_copy_as_without_metadata(); + test_config_receive_rejects_oversized_string_budget(); test_config_receive_with_validate_rejects(); } test_identity_copy_as_refused(); From a90e234eb3d7c7d4b8c423e7d5defe4f6b1e026d Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 00:59:21 +0200 Subject: [PATCH 009/155] harden: overflow guards, auth-user validation, TLS1.3 policy, build hardening --- .gitea/workflows/ci.yaml | 12 +++++----- CMakeLists.txt | 42 +++++++++++++++++++++++++++++++++- src/shared/array_list.c | 3 +++ src/shared/daemon_conf.c | 7 ++++++ src/shared/file_list.c | 4 +++- src/shared/transport_ssh.c | 2 +- src/shared/transport_tls.c | 12 ++++++++++ tests/test_array_list.c | 20 +++++++++++++++- tests/test_daemon_conf.c | 47 ++++++++++++++++++++++++++++++++++++++ 9 files changed, 139 insertions(+), 10 deletions(-) diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index 6fb81bb..bc08d57 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -12,7 +12,7 @@ jobs: container: gitea.tap-tap.win/taptap/fastsync-ci:v10 steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: clang-format check run: find src/ tests/ -name '*.c' -o -name '*.h' | xargs clang-format --dry-run --Werror @@ -30,7 +30,7 @@ jobs: needs: lint steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: Configure run: cmake -B build -S . -DSTRICT_WARNINGS=ON @@ -59,7 +59,7 @@ jobs: sanitizer: [address, undefined] steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: Configure run: cmake -B build-${{ matrix.sanitizer }} -S . -DSANITIZER=${{ matrix.sanitizer }} @@ -77,7 +77,7 @@ jobs: if: github.event_name == 'push' steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: Configure (clang + fuzz) run: CC=clang CXX=clang++ cmake -B build-fuzz -S . -DENABLE_FUZZ=ON @@ -99,7 +99,7 @@ jobs: if: github.event_name == 'push' steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: Configure run: cmake -B build -S . -DENABLE_COVERAGE=ON @@ -123,7 +123,7 @@ jobs: if: github.event_name == 'push' steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: Configure run: cmake -B build -S . -DSTRICT_WARNINGS=ON diff --git a/CMakeLists.txt b/CMakeLists.txt index 3ec3b24..ff92672 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -38,11 +38,24 @@ if(ENABLE_COVERAGE) add_link_options(--coverage) endif() +# --- Build hardening option --- +# Production hardening is applied to the shipping server/client binaries only, +# and only when no sanitizer or coverage instrumentation is active: sanitizers +# carry their own instrumentation, and _FORTIFY_SOURCE requires an optimising +# build (never the -O0 used for coverage). +option(ENABLE_HARDENING "Enable compiler/linker hardening for production targets" ON) +set(HARDENING_ACTIVE OFF) +if(ENABLE_HARDENING AND SANITIZER STREQUAL "none" AND NOT ENABLE_COVERAGE) + set(HARDENING_ACTIVE ON) +endif() + include(FetchContent) FetchContent_Declare( xxhash GIT_REPOSITORY https://github.com/Cyan4973/xxHash - GIT_TAG v0.8.3 + # v0.8.3 is a lightweight tag pointing at this exact commit (no ^{} peel + # entry); pin the commit SHA instead of the mutable tag. + GIT_TAG e626a72bc2321cd320e953a0ccf1584cad60f363 # v0.8.3 SOURCE_SUBDIR cmake_unofficial ) FetchContent_MakeAvailable(xxhash) @@ -73,6 +86,33 @@ add_executable(client ${CLIENT_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_ target_include_directories(client PRIVATE src/shared src/server src/client) target_link_libraries(client PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash) +# --- Production hardening --- +# Each compile flag is probed so a compiler/architecture that lacks it still +# configures cleanly. _FORTIFY_SOURCE is guarded separately because it only +# works in an optimising build. xxHash is a static archive built by +# FetchContent, so it must be position-independent for the -pie link. +if(HARDENING_ACTIVE) + set_target_properties(xxhash PROPERTIES POSITION_INDEPENDENT_CODE ON) + include(CheckCCompilerFlag) + foreach(flag -fstack-protector-strong -fstack-clash-protection -fPIE) + string(MAKE_C_IDENTIFIER "HARDEN_${flag}" _harden_var) + check_c_compiler_flag("${flag}" ${_harden_var}) + endforeach() + check_c_compiler_flag("-D_FORTIFY_SOURCE=2" HARDEN_FORTIFY_SOURCE) + foreach(target server client) + foreach(flag -fstack-protector-strong -fstack-clash-protection -fPIE) + string(MAKE_C_IDENTIFIER "HARDEN_${flag}" _harden_var) + if(${_harden_var}) + target_compile_options(${target} PRIVATE ${flag}) + endif() + endforeach() + if(HARDEN_FORTIFY_SOURCE) + target_compile_options(${target} PRIVATE -D_FORTIFY_SOURCE=2) + endif() + target_link_options(${target} PRIVATE -pie -Wl,-z,relro -Wl,-z,now -Wl,-z,noexecstack) + endforeach() +endif() + # --- Testing --- enable_testing() diff --git a/src/shared/array_list.c b/src/shared/array_list.c index 95799b0..7813e6d 100644 --- a/src/shared/array_list.c +++ b/src/shared/array_list.c @@ -1,6 +1,7 @@ #include "log.h" #include "array_list.h" #include "protocol.h" +#include #include #include #include @@ -39,6 +40,8 @@ void array_list_delete(ArrayList* array_list) { static bool array_list_extend(ArrayList* array_list) { if (array_list == NULL) return false; + if (array_list->capacity > INT_MAX / 2) + return false; int new_capacity = array_list->capacity * 2; if (new_capacity == 0) new_capacity = INITIAL_ARRAY_SIZE; diff --git a/src/shared/daemon_conf.c b/src/shared/daemon_conf.c index e806549..7a1c83d 100644 --- a/src/shared/daemon_conf.c +++ b/src/shared/daemon_conf.c @@ -1,4 +1,5 @@ #include "daemon_conf.h" +#include "credentials.h" #include "utils.h" #include #include @@ -192,6 +193,12 @@ static bool apply_module_key(DaemonModule* module, char* key, char* value, char* const char* user = trim_ws(token); if (*user == '\0') continue; + if (!credentials_username_valid(user)) { + set_error(err, err_size, "module '%s': invalid 'auth users' entry '%s'", module->name, + user); + free(list); + return false; + } char** grown = realloc(module->auth_users, (size_t)(module->auth_user_count + 1) * sizeof(char*)); if (!grown) { diff --git a/src/shared/file_list.c b/src/shared/file_list.c index 8e40c78..528c8d4 100644 --- a/src/shared/file_list.c +++ b/src/shared/file_list.c @@ -2,6 +2,7 @@ #include "log.h" #include "utils.h" #include +#include #include #include #include @@ -51,7 +52,8 @@ static int normalize_entry(const char* raw, size_t len, bool strip_line_endings, if (len == 0) return 0; if (raw[0] == '/') { - snprintf(err, err_size, "absolute path entries are not allowed: '%.*s'", (int)len, raw); + int print_len = len > (size_t)INT_MAX ? INT_MAX : (int)len; + snprintf(err, err_size, "absolute path entries are not allowed: '%.*s'", print_len, raw); return -1; } /* Reject NUL bytes inside a token defensively (NUL-delimited mode splits on diff --git a/src/shared/transport_ssh.c b/src/shared/transport_ssh.c index b5010ac..97eb1ea 100644 --- a/src/shared/transport_ssh.c +++ b/src/shared/transport_ssh.c @@ -128,7 +128,7 @@ char* ssh_build_remote_command(const char* server_path, bool old_args, char* con q++; len++; } - if (len > SIZE_MAX - q * 3 || len + q * 3 + 3 > SIZE_MAX - command_len) + if (q > (SIZE_MAX - len) / 3 || len + q * 3 + 3 > SIZE_MAX - command_len) return NULL; command_len += len + q * 3 + 3; } diff --git a/src/shared/transport_tls.c b/src/shared/transport_tls.c index 85bb5a1..81aa483 100644 --- a/src/shared/transport_tls.c +++ b/src/shared/transport_tls.c @@ -65,6 +65,18 @@ static SSL_CTX* create_ssl_ctx(bool is_server, const char* cert, const char* key SSL_CTX_free(ctx); return NULL; } + /* TLS 1.3 ciphersuites are configured separately from the TLS 1.2 and below + * cipher list above. Pin the three AEAD suites OpenSSL offers, dropping + * TLS_AES_128_CCM_SHA256 and the CCM_8 variant, and fail closed if the + * library rejects the policy. SSL_CTX_set_ciphersuites needs OpenSSL 1.1.1; + * earlier versions have no TLS 1.3, so the call is compile-guarded. */ +#if OPENSSL_VERSION_NUMBER >= 0x10101000L + if (SSL_CTX_set_ciphersuites( + ctx, "TLS_AES_256_GCM_SHA384:TLS_CHACHA20_POLY1305_SHA256:TLS_AES_128_GCM_SHA256") != 1) { + SSL_CTX_free(ctx); + return NULL; + } +#endif if (cert && key) { struct stat key_stat; diff --git a/tests/test_array_list.c b/tests/test_array_list.c index 964458d..a917dae 100644 --- a/tests/test_array_list.c +++ b/tests/test_array_list.c @@ -1,6 +1,7 @@ #include "test_array_list.h" #include "array_list.h" #include "test_utils.h" +#include #include static int destroyer_calls = 0; @@ -9,7 +10,7 @@ static void test_destroyer(void* item) { free(item); } -void test_array_list() { +static void test_array_list_basic() { ArrayList* list = array_list_create(free); EXPECT_NOT_NULL(list); EXPECT_EQ_INT(list->size, 0); @@ -54,3 +55,20 @@ void test_array_list() { array_list_delete(list); EXPECT_EQ_INT(destroyer_calls, 106); } + +/* A capacity that would overflow `capacity * 2` must be refused instead of + * wrapping into signed-overflow UB; array_list_add surfaces the failure. */ +static void test_array_list_extend_overflow_guard() { + ArrayList* list = array_list_create(NULL); + EXPECT_NOT_NULL(list); + list->capacity = INT_MAX / 2 + 1; + list->size = list->capacity; + EXPECT_FALSE(array_list_add(list, NULL)); + list->size = 0; + array_list_delete(list); +} + +void test_array_list() { + test_array_list_basic(); + test_array_list_extend_overflow_guard(); +} diff --git a/tests/test_daemon_conf.c b/tests/test_daemon_conf.c index 9580f8b..1f54a2e 100644 --- a/tests/test_daemon_conf.c +++ b/tests/test_daemon_conf.c @@ -1,4 +1,5 @@ #include "test_daemon_conf.h" +#include "credentials.h" #include "daemon_conf.h" #include "test_utils.h" #include @@ -320,6 +321,51 @@ static void test_daemon_conf_dparam_override() { daemon_conf_free(conf); } +/* Each `auth users` entry is validated with the same username rule as the + * credential store, so invisible whitespace/control characters can never make + * an exact strcmp match ambiguous. */ +static void test_daemon_conf_auth_users_validated() { + char* path; + char err[256]; + const DaemonConf* conf; + + EXPECT_EQ_INT(write_conf("[m]\npath = /x\nauth users = alice, bad user\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "invalid 'auth users' entry") != NULL); + + EXPECT_EQ_INT(write_conf("[m]\npath = /x\nauth users = good\tbad\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "invalid 'auth users' entry") != NULL); + + /* An over-long name exceeds CREDENTIAL_MAX_USER_LEN and is rejected. */ + { + char body[CREDENTIAL_MAX_USER_LEN + 128]; + int n = snprintf(body, sizeof(body), "[m]\npath = /x\nauth users = "); + memset(body + n, 'a', CREDENTIAL_MAX_USER_LEN + 1); + body[n + CREDENTIAL_MAX_USER_LEN + 1] = '\n'; + body[n + CREDENTIAL_MAX_USER_LEN + 2] = '\0'; + EXPECT_EQ_INT(write_conf(body, &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "invalid 'auth users' entry") != NULL); + } + + /* Empty entries between commas are skipped, not treated as invalid. */ + EXPECT_EQ_INT(write_conf("[m]\npath = /x\nauth users = alice,, bob\n", &path), 0); + DaemonConf* ok_conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(ok_conf); + EXPECT_EQ_INT(ok_conf->modules[0].auth_user_count, 2); + EXPECT_EQ_STR(ok_conf->modules[0].auth_users[0], "alice"); + EXPECT_EQ_STR(ok_conf->modules[0].auth_users[1], "bob"); + daemon_conf_free(ok_conf); +} + static void test_daemon_module_name_valid() { EXPECT_TRUE(daemon_module_name_valid("backup")); EXPECT_TRUE(daemon_module_name_valid("Backup_2")); @@ -351,5 +397,6 @@ void test_daemon_conf() { test_daemon_conf_missing_file_rejected(); test_daemon_conf_find_module(); test_daemon_conf_dparam_override(); + test_daemon_conf_auth_users_validated(); test_daemon_module_name_valid(); } \ No newline at end of file From ea2f76cd7a2144a633e2f4723a9a5e9776b13b2b Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 01:18:54 +0200 Subject: [PATCH 010/155] fix(receiver): charge per-entry DirTimeList cost; cap client --skip-compress --- src/client/client_cli.c | 5 +++++ src/shared/file_receive.c | 8 ++++++-- tests/test_file.c | 2 +- 3 files changed, 12 insertions(+), 3 deletions(-) diff --git a/src/client/client_cli.c b/src/client/client_cli.c index 400be1b..6600ec0 100644 --- a/src/client/client_cli.c +++ b/src/client/client_cli.c @@ -517,6 +517,11 @@ static int parse_skip_compress(Config* config, const char* value) { token[--len] = '\0'; if (len == 0) continue; + if (config->skip_compress_count >= MAX_SKIP_COMPRESS_SUFFIXES) { + fprintf(stderr, "--skip-compress supports at most %d suffixes\n", MAX_SKIP_COMPRESS_SUFFIXES); + free(list); + return -1; + } if (config_add_pattern(&config->skip_compress_suffixes, &config->skip_compress_count, token, "--skip-compress") != 0) { free(list); diff --git a/src/shared/file_receive.c b/src/shared/file_receive.c index f21867c..477fa68 100644 --- a/src/shared/file_receive.c +++ b/src/shared/file_receive.c @@ -2274,7 +2274,11 @@ bool dir_time_list_add(DirTimeList* list, const char* wire_path, const FileMetad touching the list, leaving it exactly as it was (the caller fails the transfer, which becomes a clean protocol error). */ size_t path_len = strlen(wire_path); - if (list->count >= MAX_DIR_TIME_ENTRIES || path_len > MAX_DIR_TIME_BYTES - list->bytes) + /* Charge the whole per-entry cost (path copy + pointer slot + metadata + struct), not just the path, so the array growth is bounded by the same + cumulative budget. */ + size_t entry_cost = path_len + sizeof(FileMetadata) + sizeof(char*); + if (list->count >= MAX_DIR_TIME_ENTRIES || entry_cost > MAX_DIR_TIME_BYTES - list->bytes) return false; if (list->count == list->capacity) { size_t new_capacity = list->capacity == 0 ? 16 : list->capacity * 2; @@ -2302,7 +2306,7 @@ bool dir_time_list_add(DirTimeList* list, const char* wire_path, const FileMetad list->paths[list->count] = copy; list->entries[list->count] = *metadata; list->count++; - list->bytes += path_len; + list->bytes += entry_cost; return true; } diff --git a/tests/test_file.c b/tests/test_file.c index 0f0225d..928f544 100644 --- a/tests/test_file.c +++ b/tests/test_file.c @@ -1396,7 +1396,7 @@ static void test_dir_time_list_cap() { EXPECT_TRUE(list.bytes == before_bytes); } else { EXPECT_TRUE(list.count == before_count + 1); - EXPECT_TRUE(list.bytes == before_bytes + path_len); + EXPECT_TRUE(list.bytes == before_bytes + path_len + sizeof(FileMetadata) + sizeof(char*)); } } EXPECT_TRUE(rejected); From b7fbb5628960515d6111e8937557d2317d4d5e27 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 01:25:29 +0200 Subject: [PATCH 011/155] test(file): silence cppcheck constVariablePointer in empty-path test --- tests/test_file.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_file.c b/tests/test_file.c index 928f544..4d53b70 100644 --- a/tests/test_file.c +++ b/tests/test_file.c @@ -1431,7 +1431,7 @@ static void test_receive_incremental_check_empty_path() { EXPECT_TRUE(send_n_data(p[1], &check_mtime_nsec, sizeof(check_mtime_nsec))); bool skipped = true; - File* file = receive_incremental_check(p[0], cfg, &skipped); + const File* file = receive_incremental_check(p[0], cfg, &skipped); EXPECT_NULL(file); EXPECT_FALSE(skipped); From 8147ff7b503ebba57a814efe814b1e822308e805 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 01:36:51 +0200 Subject: [PATCH 012/155] fix(scanner): free chunk_data on chunk-create failure --- src/client/scanner.c | 8 +++++-- tests/test_scanner.c | 53 ++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 59 insertions(+), 2 deletions(-) diff --git a/src/client/scanner.c b/src/client/scanner.c index 4911ccf..6fd3407 100644 --- a/src/client/scanner.c +++ b/src/client/scanner.c @@ -587,12 +587,16 @@ void directory_scanner_destroy(DirectoryScanner* scanner) { static Chunk* chunk_data_to_chunk(ArrayList* chunk_data) { void** chunk_items = array_list_to_array(chunk_data); - if (!chunk_items) + if (!chunk_items) { + array_list_delete(chunk_data); return NULL; + } Chunk* chunk = chunk_create((File**)chunk_items, chunk_data->size); free(chunk_items); - if (!chunk) + if (!chunk) { + array_list_delete(chunk_data); return NULL; + } chunk_data->item_destroyer = NULL; array_list_delete(chunk_data); return chunk; diff --git a/tests/test_scanner.c b/tests/test_scanner.c index 06e74ad..3ba35ed 100644 --- a/tests/test_scanner.c +++ b/tests/test_scanner.c @@ -1317,6 +1317,58 @@ static void test_scanner_captures_directory_times() { rmdir(root); } +/* Ownership guard for chunk_data_to_chunk(): a returned Chunk owns its File + * objects, so destroying the chunk must free them exactly once and the scanner + * must never free them again. chunk_size = 1 forces the mid-directory + * conversion branch (chunk_data_size > chunk_size) for every file, and the + * chunk is destroyed immediately, catching a double free / use-after-free under + * ASan if ownership transfer regressed. + * + * The failure path (array_list_to_array() or chunk_create() returning NULL) is + * not reachable from a unit test: both allocate through protocol_alloc(), and + * each allocation they perform is no larger than the array_list allocations + * that already succeeded while building the list (array_list_to_array() copies + * exactly `size` pointers, which never exceeds the capacity just grown, and + * sizeof(Chunk) is far below the initial 100-entry item array). Binding a + * small --max-alloc session therefore always fails *before* this function, not + * inside it, so fault injection cannot isolate these paths. */ +static void test_scanner_chunk_ownership() { + const char* dir = "test_scan_ownership"; + const char* file1 = "test_scan_ownership/a.txt"; + const char* file2 = "test_scan_ownership/b.txt"; + const char* file3 = "test_scan_ownership/c.txt"; + + EXPECT_EQ_INT(mkdir(dir, 0755), 0); + create_test_file(file1, "aaaa"); + create_test_file(file2, "bbbb"); + create_test_file(file3, "cccc"); + + ScannerOptions options = {0}; + options.chunk_size = 1; + DirectoryScanner* scanner = directory_scanner_create_with_options(dir, &options); + EXPECT_NOT_NULL(scanner); + + int chunks = 0; + int files = 0; + Chunk* chunk; + while ((chunk = directory_scanner_next(scanner)) != NULL) { + chunks++; + files += chunk->element_count; + EXPECT_EQ_INT(chunk->element_count, 1); + chunk_destroy(chunk); + EXPECT_FALSE(directory_scanner_failed(scanner)); + } + EXPECT_EQ_INT(files, 3); + EXPECT_EQ_INT(chunks, 3); + EXPECT_FALSE(directory_scanner_failed(scanner)); + + directory_scanner_destroy(scanner); + unlink(file1); + unlink(file2); + unlink(file3); + rmdir(dir); +} + void test_scanner() { test_scanner_single_file(); test_scanner_multiple_files(); @@ -1353,4 +1405,5 @@ void test_scanner() { test_dirs_files_from(); test_files_from_relative_send_path(); test_scanner_captures_directory_times(); + test_scanner_chunk_ownership(); } From c8f5d80fcbed8131689e2422911731d2f7aee615 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 01:44:59 +0200 Subject: [PATCH 013/155] fix(log): serialize message emission; clear log_fp before close; use logger --- src/shared/config.c | 15 +++--- src/shared/log.c | 31 ++++++++++++ tests/test_log.c | 116 ++++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 156 insertions(+), 6 deletions(-) diff --git a/src/shared/config.c b/src/shared/config.c index 584c61a..4816e06 100644 --- a/src/shared/config.c +++ b/src/shared/config.c @@ -693,6 +693,9 @@ void config_delete(Config* config) { if (config == NULL) return; if (config->log_file) { + /* The logging subsystem borrows this FILE*; detach it before closing so a + * concurrent log call can never touch the freed handle. */ + log_set_file(NULL); fclose(config->log_file); config->log_file = NULL; } @@ -1428,8 +1431,8 @@ Config* config_receive_with_validate(int file_descriptor, ConfigValidateFunc val goto error; if (strcmp(config->version, PROTOCOL_VERSION) != 0) { char* escaped_version = output_escape(config->version, false); - fprintf(stderr, "Protocol version mismatch: client=%s, server=%s\n", - escaped_version ? escaped_version : "", PROTOCOL_VERSION); + log_message(LOG_LEVEL_ERROR, "Protocol version mismatch: client=%s, server=%s", + escaped_version ? escaped_version : "", PROTOCOL_VERSION); free(escaped_version); send_status(file_descriptor, STATUS_ERROR); goto error; @@ -1455,14 +1458,14 @@ Config* config_receive_with_validate(int file_descriptor, ConfigValidateFunc val if (config->compress_choice[0] != '\0' && strcmp(config->compress_choice, "zstd") != 0 && strcmp(config->compress_choice, "none") != 0) { char* escaped_choice = output_escape(config->compress_choice, config->eight_bit_output); - fprintf(stderr, "Unsupported compression choice: %s\n", - escaped_choice ? escaped_choice : ""); + log_message(LOG_LEVEL_ERROR, "Unsupported compression choice: %s", + escaped_choice ? escaped_choice : ""); free(escaped_choice); send_status(file_descriptor, STATUS_ERROR); goto error; } if (!validate_received_config(config)) { - fprintf(stderr, "Invalid configuration received from client\n"); + log_message(LOG_LEVEL_ERROR, "Invalid configuration received from client"); send_status(file_descriptor, STATUS_ERROR); goto error; } @@ -1476,7 +1479,7 @@ Config* config_receive_with_validate(int file_descriptor, ConfigValidateFunc val * the CONFIG_VALIDATE_ALREADY_TERMINATED sentinel, so no second status is * written. */ if (rejection != CONFIG_VALIDATE_ALREADY_TERMINATED) { - fprintf(stderr, "%s\n", rejection); + log_message(LOG_LEVEL_ERROR, "%s", rejection); send_status(file_descriptor, STATUS_ERROR); } goto error; diff --git a/src/shared/log.c b/src/shared/log.c index d013bbe..d0e1f0a 100644 --- a/src/shared/log.c +++ b/src/shared/log.c @@ -4,6 +4,7 @@ #include #include #include +#include #include static const char* log_level_strings[] = {"DEBUG", "INFO", "WARN", "ERROR"}; @@ -15,6 +16,18 @@ static FILE* log_fp = NULL; static _Thread_local bool eight_bit_output; static LogStderrMode stderr_mode = LOG_STDERR_ERRORS; +/* Serializes access to log_fp and makes each emitted line atomic: the + * timestamp prefix, formatted body, and trailing newline are written as one + * critical section so concurrent threads cannot interleave partial lines. + * Initialized lazily (matching the protocol.c bw_mutex idiom) because logging + * can happen before main() installs any synchronization. */ +static mtx_t log_mutex; +static once_flag log_mutex_once = ONCE_FLAG_INIT; + +static void log_mutex_init(void) { + mtx_init(&log_mutex, mtx_plain); +} + void set_log_level(LogLevel level) { current_log_level = level; } @@ -41,7 +54,10 @@ uint32_t get_log_info_flags(void) { } void log_set_file(FILE* fp) { + call_once(&log_mutex_once, log_mutex_init); + mtx_lock(&log_mutex); log_fp = fp; + mtx_unlock(&log_mutex); } void log_set_8_bit_output(bool enabled) { @@ -79,6 +95,9 @@ void log_message(LogLevel log_level, const char* format, ...) { if (!localtime_r(&now, &t)) return; + call_once(&log_mutex_once, log_mutex_init); + mtx_lock(&log_mutex); + FILE* dest_io = stdout; if (stderr_mode == LOG_STDERR_ALL || log_level == LOG_LEVEL_ERROR) { dest_io = stderr; @@ -94,6 +113,8 @@ void log_message(LogLevel log_level, const char* format, ...) { write_message(log_fp, log_level, t, format, args); va_end(args); } + + mtx_unlock(&log_mutex); } void log_debug_message(LogDebugFlag flag, const char* format, ...) { @@ -105,6 +126,9 @@ void log_debug_message(LogDebugFlag flag, const char* format, ...) { if (!localtime_r(&now, &t)) return; + call_once(&log_mutex_once, log_mutex_init); + mtx_lock(&log_mutex); + va_list args; va_start(args, format); write_message(stdout, LOG_LEVEL_DEBUG, t, format, args); @@ -115,6 +139,8 @@ void log_debug_message(LogDebugFlag flag, const char* format, ...) { write_message(log_fp, LOG_LEVEL_DEBUG, t, format, args); va_end(args); } + + mtx_unlock(&log_mutex); } void log_info_message(LogInfoFlag flag, const char* format, ...) { @@ -127,6 +153,9 @@ void log_info_message(LogInfoFlag flag, const char* format, ...) { if (!localtime_r(&now, &t)) return; + call_once(&log_mutex_once, log_mutex_init); + mtx_lock(&log_mutex); + va_list args; va_start(args, format); write_message(stdout, LOG_LEVEL_INFO, t, format, args); @@ -137,6 +166,8 @@ void log_info_message(LogInfoFlag flag, const char* format, ...) { write_message(log_fp, LOG_LEVEL_INFO, t, format, args); va_end(args); } + + mtx_unlock(&log_mutex); } void log_perror(const char* context) { diff --git a/tests/test_log.c b/tests/test_log.c index f9caf19..b5acedd 100644 --- a/tests/test_log.c +++ b/tests/test_log.c @@ -1,7 +1,9 @@ #include "test_log.h" #include "log.h" #include "test_utils.h" +#include #include +#include #include /* Test default log level: WARNING and ERROR should print, DEBUG and INFO should not. @@ -147,6 +149,118 @@ static void test_log_debug_enabled_matches_gate() { set_log_debug_flags(LOG_DEBUG_ALL); } +#define LOG_CONCURRENCY_THREADS 8 +#define LOG_CONCURRENCY_LINES 250 + +typedef struct { + int id; +} LogConcurrencyArg; + +static int log_concurrency_worker(void* context) { + LogConcurrencyArg* arg = context; + for (int i = 0; i < LOG_CONCURRENCY_LINES; i++) { + log_message(LOG_LEVEL_WARNING, "worker %d line %d", arg->id, i); + } + return 0; +} + +static int count_substring(const char* haystack, const char* needle) { + int count = 0; + size_t needle_length = strlen(needle); + const char* cursor = haystack; + while ((cursor = strstr(cursor, needle)) != NULL) { + count++; + cursor += needle_length; + } + return count; +} + +/* Concurrent log_message() calls from many threads must never interleave a + * single line: every emitted line has exactly one timestamp prefix and one + * body. Before write_message() was serialized, the three separate fprintf + * calls (prefix, body, newline) let lines tear. */ +static void test_log_concurrent_no_torn_lines(void) { + FILE* fp = tmpfile(); + EXPECT_NOT_NULL(fp); + + /* Mute the console mirror so the workers don't flood the test output. */ + fflush(stdout); + fflush(stderr); + int saved_stdout = dup(STDOUT_FILENO); + int saved_stderr = dup(STDERR_FILENO); + int null_fd = open("/dev/null", O_WRONLY); + EXPECT_TRUE(saved_stdout >= 0); + EXPECT_TRUE(saved_stderr >= 0); + EXPECT_TRUE(null_fd >= 0); + EXPECT_TRUE(dup2(null_fd, STDOUT_FILENO) >= 0); + EXPECT_TRUE(dup2(null_fd, STDERR_FILENO) >= 0); + close(null_fd); + + set_log_level(LOG_LEVEL_WARNING); + log_set_stderr_mode(LOG_STDERR_ERRORS); + log_set_file(fp); + + thrd_t threads[LOG_CONCURRENCY_THREADS]; + LogConcurrencyArg args[LOG_CONCURRENCY_THREADS]; + int created = 0; + for (int i = 0; i < LOG_CONCURRENCY_THREADS; i++) { + args[i].id = i; + if (thrd_create(&threads[i], log_concurrency_worker, &args[i]) != thrd_success) + break; + created++; + } + for (int i = 0; i < created; i++) { + thrd_join(threads[i], NULL); + } + + log_set_file(NULL); + fflush(fp); + + fflush(stdout); + fflush(stderr); + dup2(saved_stdout, STDOUT_FILENO); + dup2(saved_stderr, STDERR_FILENO); + close(saved_stdout); + close(saved_stderr); + + rewind(fp); + char line[512]; + int total_lines = 0; + int malformed_lines = 0; + bool saw_missing_newline = false; + while (fgets(line, sizeof(line), fp) != NULL) { + size_t length = strlen(line); + if (length == 0 || line[length - 1] != '\n') + saw_missing_newline = true; + if (strncmp(line, "20", 2) != 0 || count_substring(line, "[WARN]: worker ") != 1) + malformed_lines++; + total_lines++; + } + fclose(fp); + + EXPECT_EQ_INT(created, LOG_CONCURRENCY_THREADS); + EXPECT_FALSE(saw_missing_newline); + EXPECT_EQ_INT(malformed_lines, 0); + EXPECT_EQ_INT(total_lines, LOG_CONCURRENCY_THREADS * LOG_CONCURRENCY_LINES); + log_set_stderr_mode(LOG_STDERR_ERRORS); +} + +/* Detaching the logger from a FILE* before it is closed must leave the logging + * subsystem safe: later calls must not touch the freed handle. */ +static void test_log_set_file_null_before_fclose(void) { + FILE* fp = tmpfile(); + EXPECT_NOT_NULL(fp); + + set_log_level(LOG_LEVEL_ERROR); + log_set_file(fp); + log_message(LOG_LEVEL_ERROR, "line before detach"); + log_set_file(NULL); + fclose(fp); + + log_message(LOG_LEVEL_ERROR, "line after close"); + EXPECT_TRUE(true); +} + void test_log() { test_log_message_debug(); test_log_message_info(); @@ -159,4 +273,6 @@ void test_log() { test_log_stderr_mode_all(); test_log_message_formats(); test_log_debug_enabled_matches_gate(); + test_log_concurrent_no_torn_lines(); + test_log_set_file_null_before_fclose(); } From fecbe2c90c2925adbf896cf4c578dc42b0fcbcc1 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 01:49:27 +0200 Subject: [PATCH 014/155] fix(server): child-safe signals, single fd owner, handler cleanup epilogue --- src/server/server.c | 144 +++++++++++++++++-------------------- src/shared/transport_tcp.c | 14 +++- src/shared/transport_tls.c | 5 ++ 3 files changed, 83 insertions(+), 80 deletions(-) diff --git a/src/server/server.c b/src/server/server.c index 77b1e8b..0040158 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -23,6 +23,7 @@ #include #include #include +#include #include #include @@ -502,12 +503,16 @@ void handler(int file_descriptor) { gate_ctx.ssl = ssl; gate_ctx.fd = file_descriptor; gate_ctx.super_mode_override = -1; - Config* config = config_receive_with_validate(file_descriptor, server_module_gate, &gate_ctx); + /* All teardown state starts empty so the single `done` epilogue is safe to + * reach from any error path (including before the config frame arrives). */ + Config* config = NULL; + PipelineContextReceiver* context = NULL; + char* joined_destination = NULL; + bool charset_ready = false; + config = config_receive_with_validate(file_descriptor, server_module_gate, &gate_ctx); if (config == NULL) { log_message(LOG_LEVEL_ERROR, "Failed to receive config"); - close(file_descriptor); - protocol_session_unbind(); - return; + goto done; } /* Apply the super-mode veto the gate decided on (operator --no-super, or a * daemon module without the `client owner = yes` opt-in) exactly once, so @@ -519,23 +524,15 @@ void handler(int file_descriptor) { protocol_set_8_bit_output(config->eight_bit_output); if (!authorized_root) { log_message(LOG_LEVEL_ERROR, "No server-side destination root configured"); - config_delete(config); - close(file_descriptor); - protocol_session_unbind(); - return; + goto done; } if (!allow_unauthenticated && ssl == NULL) { log_message(LOG_LEVEL_ERROR, "Rejected unauthenticated plaintext connection"); - config_delete(config); - close(file_descriptor); - protocol_session_unbind(); - return; + goto done; } if (ssl && required_client_cn && !tls_client_identity_allowed(ssl)) { log_message(LOG_LEVEL_ERROR, "Rejected TLS client with unauthorized identity"); - config_delete(config); - close(file_descriptor); - return; + goto done; } /* Daemon mode: the module's root is the authorized root (installed by server_module_gate), and the client's destination is a MODULE-RELATIVE @@ -545,13 +542,9 @@ void handler(int file_descriptor) { if (g_daemon_conf && config->receive_root_directory && config->receive_root_directory[0] == '/') { log_message(LOG_LEVEL_ERROR, "Rejected absolute daemon destination (must be relative to the " "selected module root)"); - config_delete(config); - close(file_descriptor); - protocol_session_unbind(); - return; + goto done; } char* destination = config->receive_root_directory; - char* joined_destination = NULL; if (destination && destination[0] != '/') joined_destination = path_cat(authorized_root, destination); if (joined_destination) @@ -560,19 +553,16 @@ void handler(int file_descriptor) { !path_is_within(authorized_root, destination)) { log_message(LOG_LEVEL_ERROR, "Rejected destination outside authorized root"); free(joined_destination); - config_delete(config); - close(file_descriptor); - return; + joined_destination = NULL; + goto done; } if (joined_destination) { free(config->receive_root_directory); config->receive_root_directory = joined_destination; + joined_destination = NULL; } if (!config->receive_root_directory) { - config_delete(config); - close(file_descriptor); - protocol_session_unbind(); - return; + goto done; } config->use_delete = config->use_delete && allow_delete; /* --iconv (protocol 2.16.0): install the receiver-side wire->local conversion @@ -581,13 +571,13 @@ void handler(int file_descriptor) { any) may override the local charset; a spec the client is known to have validated cannot fail here unless the server's override names an unsupported charset. */ - if (config->iconv_spec && !charset_wire_init_receiver(config->iconv_spec, server_iconv_spec)) { - log_message(LOG_LEVEL_ERROR, - "--iconv: unsupported charset conversion requested (LOCAL[,REMOTE])"); - config_delete(config); - close(file_descriptor); - protocol_session_unbind(); - return; + if (config->iconv_spec) { + if (!charset_wire_init_receiver(config->iconv_spec, server_iconv_spec)) { + log_message(LOG_LEVEL_ERROR, + "--iconv: unsupported charset conversion requested (LOCAL[,REMOTE])"); + goto done; + } + charset_ready = true; } /* --delete-missing-args deletes destination mirrors receiver-side, so it is deletion and stays gated by the same --allow-delete server policy. When @@ -602,10 +592,7 @@ void handler(int file_descriptor) { log_message(LOG_LEVEL_ERROR, "destination root is not available: %s", escaped_root ? escaped_root : ""); free(escaped_root); - config_delete(config); - close(file_descriptor); - protocol_session_unbind(); - return; + goto done; } /* A --delay-updates transfer stages under a private 0700 directory inside the receive root. Create it up front (wiping leftovers of any previously @@ -614,11 +601,7 @@ void handler(int file_descriptor) { config->delay_context = delay_updates_context_create(config->receive_root_directory); if (!config->delay_context || !delay_updates_prepare(config->delay_context)) { log_message(LOG_LEVEL_ERROR, "Failed to initialize --delay-updates staging area"); - delay_updates_cleanup(config->delay_context); - config_delete(config); - close(file_descriptor); - protocol_session_unbind(); - return; + goto done; } } /* Preserve the negotiated identity policy for the fd-relative ownership @@ -628,10 +611,7 @@ void handler(int file_descriptor) { rather than silently applying the wrong ownership policy. */ if (!identity_set_active(config)) { log_message(LOG_LEVEL_ERROR, "Failed to activate identity policy"); - config_delete(config); - close(file_descriptor); - protocol_session_unbind(); - return; + goto done; } /* Persist the negotiated --keep-dirlinks policy once, here at config-accept, before any multithreaded receiver/writer threads are spawned, so the @@ -662,38 +642,25 @@ void handler(int file_descriptor) { if (!motd_send(file_descriptor, motd ? motd : "")) { free(motd); log_message(LOG_LEVEL_ERROR, "Failed to send daemon MOTD"); - config_delete(config); - close(file_descriptor); - protocol_session_unbind(); - identity_clear_active(); - return; + goto done; } free(motd); } if (config->use_multithreading) { Queue* q = queue_create(100, file_destroy); - if (q == NULL) { - config_delete(config); - close(file_descriptor); - protocol_session_unbind(); - identity_clear_active(); - return; - } - PipelineContextReceiver* context = - pipeline_context_receiver_create(config, q, file_descriptor, ssl); + if (q == NULL) + goto done; + context = pipeline_context_receiver_create(config, q, file_descriptor, ssl); if (context == NULL) { queue_destroy(q); - config_delete(config); - close(file_descriptor); - protocol_session_unbind(); - identity_clear_active(); - return; + goto done; } protocol_session_set_max_alloc(&context->session, config->max_alloc); atomic_store(&context->session.total_allocated_bytes, atomic_load(&session.total_allocated_bytes)); pipeline_context_receiver_set_queue_byte_limit(context, RECEIVER_QUEUE_MAX_BYTES); - thrd_t receiver, writer; + thrd_t receiver = {0}; + thrd_t writer = {0}; bool receiver_created = thrd_create(&receiver, receive_thread, context) == thrd_success; bool writer_created = false; if (receiver_created) @@ -706,17 +673,15 @@ void handler(int file_descriptor) { cnd_broadcast(&context->condition_not_full); cnd_broadcast(&context->condition_not_empty); mtx_unlock(&context->mutex); - close(file_descriptor); + /* Unblock a worker parked in socket I/O without closing the fd: the + * child owns the single close. shutdown() makes the pending I/O fail + * so thrd_join cannot hang waiting for a thread that never returns. */ + shutdown(file_descriptor, SHUT_RDWR); thrd_join(receiver, NULL); - } else { - close(file_descriptor); } if (writer_created) thrd_join(writer, NULL); - pipeline_context_receiver_destroy(context); - protocol_session_unbind(); - identity_clear_active(); - return; + goto done; } int receiver_result; int writer_result; @@ -765,16 +730,34 @@ void handler(int file_descriptor) { if (config->delay_updates && config->delay_context) delay_updates_cleanup(config->delay_context); } - pipeline_context_receiver_destroy(context); } else { if (receiver_receive_files(config, file_descriptor) != 0) log_message(LOG_LEVEL_ERROR, "Transfer failed"); - config_delete(config); } - protocol_session_unbind(); + +done: + /* Single cleanup epilogue: every error path jumps here, so the iconv + * receiver conversion is released, the identity snapshot cleared, the + * protocol session unbound and the config freed exactly once. The + * connection fd is deliberately NOT closed here -- the child functions own + * its single close (plain_child_fn / tls_child_fn), and the --stdio call + * site must leave stdin/stdout open. */ + if (charset_ready) + charset_wire_free(); + if (config && config->delay_context) + delay_updates_cleanup(config->delay_context); identity_clear_active(); - charset_wire_free(); - close(file_descriptor); + protocol_session_unbind(); + if (context != NULL) { + /* context owns both the config and the queue it was created with. */ + pipeline_context_receiver_destroy(context); + context = NULL; + config = NULL; + } else { + config_delete(config); + config = NULL; + } + free(joined_destination); } #ifndef FASTSYNC_SERVER_AS_LIB @@ -969,6 +952,9 @@ int main(int argc, char* argv[]) { return 1; } io_set_fds(STDIN_FILENO, STDOUT_FILENO); + /* handler() does not own the stdio fds: it never closes its descriptor + * argument, so STDIN/STDOUT stay open for this (single-shot) SSH session + * and are released by process exit. */ handler(STDIN_FILENO); release_authorization(); server_cli_options_free(&opts); diff --git a/src/shared/transport_tcp.c b/src/shared/transport_tcp.c index b26cb8e..0a4aeef 100644 --- a/src/shared/transport_tcp.c +++ b/src/shared/transport_tcp.c @@ -152,9 +152,18 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil log_message(LOG_LEVEL_INFO, "%s", log_fmt); pid_t pid = fork(); if (pid == 0) { + /* Connection children must not run the parent's global cleanup(): it + * frees state (credentials / daemon conf) that the child's worker + * threads may still be reading and closes fd numbers the child could + * already have reused. Reset the inherited handlers so a signal + * terminates the child directly; SIGCHLD is reset too since a child + * must never reap the parent's children. This runs before the child + * spawns any thread, so it cannot race one. */ + signal(SIGINT, SIG_DFL); + signal(SIGTERM, SIG_DFL); + signal(SIGCHLD, SIG_DFL); close(server->file_descriptor); child_fn(fd, child_ctx); - close(fd); _exit(0); } else if (pid > 0) { g_active_connections++; @@ -169,6 +178,9 @@ struct plain_ctx { static void plain_child_fn(int fd, void* ctx) { ((struct plain_ctx*)ctx)->handler(fd); + /* handler() never closes the connection fd; the child owns its single + * close here after the handler has fully torn down. */ + close(fd); } bool server_listen(Server* server, void (*handler)(int file_descriptor)) { diff --git a/src/shared/transport_tls.c b/src/shared/transport_tls.c index 81aa483..f95a7bc 100644 --- a/src/shared/transport_tls.c +++ b/src/shared/transport_tls.c @@ -192,13 +192,18 @@ static void tls_child_fn(int fd, void* arg) { SSL* ssl = wrap_fd_with_ssl(fd, ctx->ssl_ctx, true, NULL); if (!ssl) { io_set_ssl(NULL); + close(fd); return; } io_set_ssl(ssl); ctx->handler(fd); + /* Shut the TLS layer down before releasing the fd: handler() no longer + * closes it, so SSL_shutdown still has a valid socket. The child owns the + * single fd close, performed last. */ SSL_shutdown(ssl); SSL_free(ssl); io_set_ssl(NULL); + close(fd); } bool server_listen_tls(Server* server, void (*handler)(int file_descriptor)) { From ba1c7a369f71327fd84c359b45720fb0a56b8b18 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 02:13:22 +0200 Subject: [PATCH 015/155] fix(server,log): non-socket shutdown fallback, drop redundant delay cleanup, unlock logging I/O --- src/client/client_cli.c | 4 +- src/server/server.c | 21 ++++----- src/shared/log.c | 99 ++++++++++++++++++++++++----------------- 3 files changed, 71 insertions(+), 53 deletions(-) diff --git a/src/client/client_cli.c b/src/client/client_cli.c index 6600ec0..27eb93f 100644 --- a/src/client/client_cli.c +++ b/src/client/client_cli.c @@ -1376,9 +1376,11 @@ static bool cli_handle_io_options(CliParseCtx* ctx) { return true; } if (config->log_file) { + /* Detach the logger before closing: log I/O may be in flight and must + never touch a freed FILE*. */ + log_set_file(NULL); fclose(config->log_file); config->log_file = NULL; - log_set_file(NULL); } FILE* lf = fopen(ctx->argv[++ctx->i], "a"); if (!lf) { diff --git a/src/server/server.c b/src/server/server.c index 0040158..842931b 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -673,10 +673,14 @@ void handler(int file_descriptor) { cnd_broadcast(&context->condition_not_full); cnd_broadcast(&context->condition_not_empty); mtx_unlock(&context->mutex); - /* Unblock a worker parked in socket I/O without closing the fd: the - * child owns the single close. shutdown() makes the pending I/O fail - * so thrd_join cannot hang waiting for a thread that never returns. */ - shutdown(file_descriptor, SHUT_RDWR); + /* Unblock a worker parked in socket I/O without closing the fd (the + * child owns the single close). shutdown() only affects sockets; for + * the --stdio pipe the receiver's per-message poll timeout still + * bounds the join, so do nothing there rather than close a descriptor + * another thread may still be using. */ + struct stat fd_stat; + if (fstat(file_descriptor, &fd_stat) == 0 && S_ISSOCK(fd_stat.st_mode)) + shutdown(file_descriptor, SHUT_RDWR); thrd_join(receiver, NULL); } if (writer_created) @@ -725,11 +729,8 @@ void handler(int file_descriptor) { } else { send_status(file_descriptor, STATUS_ERROR); } - if (!transfer_ok) { + if (!transfer_ok) log_message(LOG_LEVEL_ERROR, "Transfer failed"); - if (config->delay_updates && config->delay_context) - delay_updates_cleanup(config->delay_context); - } } else { if (receiver_receive_files(config, file_descriptor) != 0) log_message(LOG_LEVEL_ERROR, "Transfer failed"); @@ -744,8 +745,8 @@ done: * site must leave stdin/stdout open. */ if (charset_ready) charset_wire_free(); - if (config && config->delay_context) - delay_updates_cleanup(config->delay_context); + /* The delay-updates staging tree is released by config_delete (which the + branch below always reaches), so it is cleaned exactly once. */ identity_clear_active(); protocol_session_unbind(); if (context != NULL) { diff --git a/src/shared/log.c b/src/shared/log.c index d0e1f0a..4e69a05 100644 --- a/src/shared/log.c +++ b/src/shared/log.c @@ -3,6 +3,7 @@ #include #include #include +#include #include #include #include @@ -76,13 +77,48 @@ LogStderrMode log_get_stderr_mode(void) { return stderr_mode; } -static inline void write_message(FILE* dest_io, LogLevel log_level, struct tm t, const char* format, - va_list args) { - fprintf(dest_io, "%04d-%02d-%02d %02d:%02d:%02d [%s]: ", t.tm_year + 1900, t.tm_mon + 1, - t.tm_mday, t.tm_hour, t.tm_min, t.tm_sec, log_level_strings[log_level]); +/* Format one complete log line (timestamp prefix + body + newline) into a + * freshly allocated buffer. This is pure CPU/malloc work and must happen + * OUTSIDE the log mutex: the mutex only guards the log_fp pointer, so a + * stalled stderr/stdout pipe cannot block every logging thread. Returns NULL + * on allocation/formatting failure. */ +static char* format_log_line(LogLevel log_level, const struct tm* t, const char* format, + va_list args) { + char prefix[64]; + int prefix_len = snprintf( + prefix, sizeof(prefix), "%04d-%02d-%02d %02d:%02d:%02d [%s]: ", t->tm_year + 1900, + t->tm_mon + 1, t->tm_mday, t->tm_hour, t->tm_min, t->tm_sec, log_level_strings[log_level]); + if (prefix_len < 0 || prefix_len >= (int)sizeof(prefix)) + return NULL; + va_list copy; + va_copy(copy, args); + int body_len = vsnprintf(NULL, 0, format, copy); + va_end(copy); + if (body_len < 0) + return NULL; + size_t total = (size_t)prefix_len + (size_t)body_len; + char* line = malloc(total + 2); /* body bytes + '\n' + NUL */ + if (!line) + return NULL; + memcpy(line, prefix, (size_t)prefix_len); + vsnprintf(line + prefix_len, (size_t)body_len + 1, format, args); + line[total] = '\n'; + line[total + 1] = '\0'; + return line; +} - vfprintf(dest_io, format, args); - fprintf(dest_io, "\n"); +/* Write an already-formatted line to the console and, if configured, the log + * file. Only the log_fp pointer is read under the mutex (so log_set_file / + * config_delete cannot free it while it is in use); the single console fputs + * runs unlocked but is internally atomic per stdio stream. */ +static void emit_log_line(FILE* console, const char* line) { + fputs(line, console); + call_once(&log_mutex_once, log_mutex_init); + mtx_lock(&log_mutex); + FILE* file = log_fp; + if (file) + fputs(line, file); + mtx_unlock(&log_mutex); } void log_message(LogLevel log_level, const char* format, ...) { @@ -95,9 +131,6 @@ void log_message(LogLevel log_level, const char* format, ...) { if (!localtime_r(&now, &t)) return; - call_once(&log_mutex_once, log_mutex_init); - mtx_lock(&log_mutex); - FILE* dest_io = stdout; if (stderr_mode == LOG_STDERR_ALL || log_level == LOG_LEVEL_ERROR) { dest_io = stderr; @@ -105,16 +138,12 @@ void log_message(LogLevel log_level, const char* format, ...) { va_list args; va_start(args, format); - write_message(dest_io, log_level, t, format, args); + char* line = format_log_line(log_level, &t, format, args); va_end(args); - - if (log_fp) { - va_start(args, format); - write_message(log_fp, log_level, t, format, args); - va_end(args); - } - - mtx_unlock(&log_mutex); + if (!line) + return; + emit_log_line(dest_io, line); + free(line); } void log_debug_message(LogDebugFlag flag, const char* format, ...) { @@ -126,21 +155,14 @@ void log_debug_message(LogDebugFlag flag, const char* format, ...) { if (!localtime_r(&now, &t)) return; - call_once(&log_mutex_once, log_mutex_init); - mtx_lock(&log_mutex); - va_list args; va_start(args, format); - write_message(stdout, LOG_LEVEL_DEBUG, t, format, args); + char* line = format_log_line(LOG_LEVEL_DEBUG, &t, format, args); va_end(args); - - if (log_fp) { - va_start(args, format); - write_message(log_fp, LOG_LEVEL_DEBUG, t, format, args); - va_end(args); - } - - mtx_unlock(&log_mutex); + if (!line) + return; + emit_log_line(stdout, line); + free(line); } void log_info_message(LogInfoFlag flag, const char* format, ...) { @@ -153,21 +175,14 @@ void log_info_message(LogInfoFlag flag, const char* format, ...) { if (!localtime_r(&now, &t)) return; - call_once(&log_mutex_once, log_mutex_init); - mtx_lock(&log_mutex); - va_list args; va_start(args, format); - write_message(stdout, LOG_LEVEL_INFO, t, format, args); + char* line = format_log_line(LOG_LEVEL_INFO, &t, format, args); va_end(args); - - if (log_fp) { - va_start(args, format); - write_message(log_fp, LOG_LEVEL_INFO, t, format, args); - va_end(args); - } - - mtx_unlock(&log_mutex); + if (!line) + return; + emit_log_line(stdout, line); + free(line); } void log_perror(const char* context) { From dff6609976d19ca09156c5b5f8d21eed2dec0511 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 02:36:15 +0200 Subject: [PATCH 016/155] feat(daemon): host ACL, configurable max connections, peer audit, auth-failure delay --- README.md | 26 ++++ RSYNC_COMPAT.md | 6 +- src/server/server.c | 106 ++++++++++++- src/shared/daemon_conf.c | 256 +++++++++++++++++++++++++++++++ src/shared/daemon_conf.h | 59 ++++++- src/shared/transport_tcp.c | 16 +- src/shared/transport_tcp.h | 3 + src/shared/utils.c | 56 +++++++ src/shared/utils.h | 8 + tests/integration/test_daemon.py | 19 ++- tests/test_daemon_conf.c | 122 +++++++++++++++ tests/test_shared_utils.c | 64 ++++++++ 12 files changed, 722 insertions(+), 19 deletions(-) diff --git a/README.md b/README.md index 204a19d..d84146b 100644 --- a/README.md +++ b/README.md @@ -487,6 +487,32 @@ defaults to the current directory. | | `-v`, `--verbose` | Enable debug logging. | | `--help` | Print server usage. | +### Daemon configuration + +`fastsync-server --daemon --config FILE` reads a line-based module config (an +implicit global section, then `[module]` sections). Besides `port`, `motd file`, +and `address`, the global section accepts: + +- `max connections = N` — cap on concurrent connections, default 100. The + listener enforces it; `0`, negative, and non-numeric values are parse errors. +- `auth failure delay = MS` — milliseconds to sleep after a failed + authentication, default 500. `0` disables it and the value is capped at 60000, + so online password guessing is rate-limited per connection. Successful auths + are never delayed. +- `hosts allow` / `hosts deny` — comma- and/or whitespace-separated host access + patterns. + +A `[module]` may also set `max connections` (parsed and validated but not +enforced per module — the global cap applies to the whole listener) and its own +`hosts allow`/`hosts deny`. + +Host patterns are `*` (match all), IPv4/IPv6 literals, IPv4/IPv6 CIDR +(`10.0.0.0/8`, `2001:db8::/32`), or hostname globs (`*.example.com`). A matching +`hosts deny` rejects; if any `hosts allow` entries exist, a peer matching none of +them is rejected; deny takes precedence over allow. The global list is checked +before the module list, before authentication, and the connecting peer address +(IPv4 or IPv6) appears in the connection and authentication audit log lines. + ## Architecture ### Client diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index ef5ead2..8cdeafb 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -627,7 +627,7 @@ now transmits targets (the prior behavior was broken/partial); its status moved |------|-------------------|-----------------|-------| | `--daemon` | Run as rsync daemon | ✅ Implemented | Wave A: a real persistent listener. `fastsync-server --daemon --config FILE` (plus `--no-detach` to stay foreground; without it the listener detaches to the background after binding) reads a FastSync-native module config file and serves each connection confined to the requested module's `path` root (never a client-chosen root; every client-chosen-ownership/super-user request (`--numeric-ids`/`--chown`/`--usermap`/`--groupmap`/`--fake-super`/`--copy-as`/explicit `--super`) is refused unless the module opts in with `client owner = yes`, and the operator `--no-super` veto is honored). TCP/TLS via the existing `--tls` stack; plaintext still requires `--allow-unauthenticated` (same secure default as the standalone server). Client destinations use rsync's `host::module/path` form. Wire/protocol: the config frame gained a trailing daemon-module string and `PROTOCOL_VERSION` was bumped **2.14.0 → 2.15.0** (see the Daemon Mode notes below). Daemon mode is built in FastSync's own protocol/config grammar, not rsync's SMB/daemon option encoding | | `--config=FILE` | Alternate rsyncd.conf file | ✅ Implemented | Wave A: selects the daemon config file. Default when omitted (in `--daemon` mode): `~/.config/fastsync/fastsyncd.conf` if it exists, else `/etc/fastsyncd.conf`. The grammar is FastSync-native (documented in the Daemon Mode notes below) and strictly rejects unknown keys so a typo can never silently change what a module serves; requires `--daemon` | -| `--dparam=OVERRIDE` | Override global daemon config | ✅ Implemented | Wave A: overrides one global scalar from the command line (`--dparam port=8734` and `--dparam=KEY=VALUE` both work). Limited to the global scalar keys the grammar defines (`port`, `motd file`, `address`); keys are case-insensitive and unknown keys/invalid values are rejected. Requires `--daemon` | +| `--dparam=OVERRIDE` | Override global daemon config | ✅ Implemented | Wave A: overrides one global scalar from the command line (`--dparam port=8734` and `--dparam=KEY=VALUE` both work). Limited to the global keys the grammar defines (`port`, `motd file`, `address`, `max connections`, `auth failure delay`, `hosts allow`, `hosts deny`); keys are case-insensitive and unknown keys/invalid values are rejected. Requires `--daemon` | | `--no-detach` | Don't detach from parent | ✅ Implemented | Wave A: with `--daemon`, keeps the listener in the foreground (what integration tests use). Without it the daemonizes (fork/setsid, stdio redirected to /dev/null) after the listening socket is bound. Requires `--daemon` | | `--password-file=FILE` | Read daemon password from file | ✅ Implemented | A7 daemon auth. Client: `--password-file` supplies `user:password` for a `host::module/path` destination (the username is taken from this file, so `user@host::module` stays rejected); the literal password is held client-side only for the SCRAM handshake and wiped at teardown. Server (`fastsync-server --daemon --password-file FILE`): the salted-PBKDF2 verifier store that modules with `auth users` are verified against. **Neither the password nor any replayable bearer value crosses the wire or is stored server-side** — the store holds a per-user salt plus derived keys, and the daemon proves the secret with a per-connection nonce challenge. The file must be private to its owner: both the client and server verify the exact inode they read (open-then-`fstat`, so the check cannot be raced) and refuse a `--password-file`/`--early-input` that is not owned by the current user or grants any group/other permission bit (mode 0600), mirroring the TLS private-key check. A process-substitution pipe (`--early-input <(vault ...)`) is still accepted when it satisfies those checks. See the Daemon Mode notes below for the file formats and the plaintext/TLS caveat | | `--early-input=FILE` | Use FILE for daemon early exec | ✅ Implemented | Server-only (requires `--daemon`): a second credential-store file, same new-format grammar as `--password-file`, read before the listener accepts connections (a secrets-manager / process-substitution source). Its entries layer over `--password-file`: byte-identical verifiers dedupe, a conflicting verifier for the same user is a startup error. A daemon whose modules declare `auth users` must be given at least one of the two, or it refuses to start (fail closed) | @@ -635,7 +635,9 @@ now transmits targets (the prior behavior was broken/partial); its status moved **Daemon Mode notes (Wave A protocol 2.15.0; A7 auth protocol 2.19.0; MOTD no bump):** FastSync daemon mode is supported in FastSync's own protocol/config grammar, not rsync's SMB/daemon option encoding. -- **Config grammar** (`fastsyncd.conf`): line-based; an implicit global section first, then `[module]` sections. Keys are case-insensitive, values are trimmed and may be wrapped in one layer of double quotes (`path = "/srv/my dir"`). `#` and `;` at the start of a line (after leading whitespace) are full-line comments; inline comments and `\` continuations are not supported. Lines are bounded (4096 chars). Global keys: `port` (default 873), `motd file` (the daemon sends its bounded, escaped content to a client after the module gate/auth accepts, unless the client passes `--no-motd`), `address` (optional bind address). Module keys: `path` (required; the daemon-side authorized root for that module), `read only` (yes/no/true/false/1/0, default no), `client owner` (yes/no/true/false/1/0, default no; opts the module into client-chosen ownership — see below), `auth users` (comma list). **Unknown keys and malformed lines are parse-and-reject errors** (never silently ignored), so a typo cannot change what a module serves. +- **Config grammar** (`fastsyncd.conf`): line-based; an implicit global section first, then `[module]` sections. Keys are case-insensitive, values are trimmed and may be wrapped in one layer of double quotes (`path = "/srv/my dir"`). `#` and `;` at the start of a line (after leading whitespace) are full-line comments; inline comments and `\` continuations are not supported. Lines are bounded (4096 chars). Global keys: `port` (default 873), `motd file` (the daemon sends its bounded, escaped content to a client after the module gate/auth accepts, unless the client passes `--no-motd`), `address` (optional bind address), `max connections` (positive integer cap on concurrent connections, default 100; 0/negative/garbage is a parse error), `auth failure delay` (milliseconds to sleep after a failed authentication, default 500; 0 disables, capped at 60000), `hosts allow` and `hosts deny` (comma- and/or whitespace-separated host access patterns — see the host access control note below). Module keys: `path` (required; the daemon-side authorized root for that module), `read only` (yes/no/true/false/1/0, default no), `client owner` (yes/no/true/false/1/0, default no; opts the module into client-chosen ownership — see below), `auth users` (comma list), `max connections` (optional per-module cap; parsed and stored but **not enforced** — the global cap applies to the whole listener), `hosts allow`/`hosts deny` (per-module host access lists). **Unknown keys and malformed lines are parse-and-reject errors** (never silently ignored), so a typo cannot change what a module serves. +- **Host access control (`hosts allow`/`hosts deny`):** both keys accept a comma- and/or whitespace-separated list of patterns and may appear globally and/or per module (multiple lines append). Supported patterns are `*` (match all), an IPv4 or IPv6 literal (`10.0.0.1`, `2001:db8::1`), an IPv4/IPv6 CIDR (`10.0.0.0/8`, `2001:db8::/32`), and a simple glob (`*.example.com`; globs are matched case-insensitively against the peer string, so a numeric peer never matches a hostname glob). rsync-like semantics: a matching `hosts deny` rejects; if any `hosts allow` entries exist, a peer matching none of them is rejected; deny takes precedence over allow. The daemon enforces the global list first, then the selected module's list, **before authentication** in `server_module_gate`, with an audit log line naming the peer, the module and the outcome. The numeric peer address is obtained with `getpeername`+`inet_ntop` (`utils_fd_peer_ip`, handling both address families); when it cannot be obtained a module with any ACL fails closed (refused), while an ACL-free module continues and logs at debug. A malformed pattern (e.g. an out-of-range CIDR prefix) is a parse error at load time. +- **Connection cap and auth throttle:** the global `max connections` key (default 100) is plumbed into the listener (`transport_tcp.c`), which rejects a connection once the accept-loop parent's active-child count reaches it; the IPv4/IPv6 peer is logged for every accepted connection. The optional per-module `max connections` key is parsed and validated but **not enforced** (connections are counted in the parent before the client's module is known); the daemon logs a startup warning for any module that sets it. On a failed authentication the per-connection child sleeps the global `auth failure delay` (default 500 ms, 0 disables, capped at 60000) via `nanosleep` before the connection closes, rate-limiting online guessing without delaying a success. - **Module selection & confinement:** the client requests a module with an rsync-style `host::module[/path]` destination. The module name crosses the wire as a trailing string on the config frame (bumping `PROTOCOL_VERSION` 2.14.0 → 2.15.0; the bump is required because the config-frame layout changed and the strict same-version handshake is what prevents a peer from desynchronizing on the new trailing field). The daemon looks the module up in ITS OWN config and uses the module's `path` as the authorized root through the exact same `configure_authorization` confinement the standalone server applies to `--destination-root` (`file_open_secure_parent`, `has_path_traversal`, `path_is_within`); the client never supplies the root, every client-chosen-ownership/super-user request is refused unless the module declares `client owner = yes` (the daemon's per-module opt-in, see below), and the operator `--no-super` veto forces super-user activities off for every daemon connection. The client's `/path` part is relative inside the module and is rejected if absolute or if it contains `..`. Unknown modules are refused before any data moves (the run fails cleanly at the config handshake). An absolute destination and a module request against a non-daemon server are also refused. - **`client owner` (client-chosen-ownership opt-in):** by default a daemon module refuses every request that would let the client pick an owner or ask for super-user activities — `--numeric-ids`, `--chown`, `--usermap`/`--groupmap`, `--fake-super`, `--copy-as`, and an explicit `--super` — at the config handshake (before `STATUS_OK`), because a daemon has no per-module opt-in for client-chosen ownership and any anonymous client could otherwise force arbitrary owner ids inside the module root. `client owner = yes` opts a single module in, allowing those requests within that module's root (the standalone listener and the SSH `--stdio` server always honor them for their single operator-authorized root). Without the opt-in the daemon also forces super-user **device** activity off for that connection — char/block device-node creation (`--devices`) and `--write-devices` — even under the default `AUTO` mode, so a non-opted module can never be made to `mknod` or write a raw device; those entries are skipped (not refused) so an ordinary `-a` push still succeeds without device nodes. The opt-in does **not** lift the privilege requirement: `--copy-as` still needs a root receiver, and the operator `--no-super` veto still forces super-user activities off for every connection. The daemon logs a prominent startup warning for each `client owner = yes` module so the operator's deliberate choice is visible. - **`read only` safe default:** every network transfer FastSync currently supports is a push that writes under the module root, so a `read only` module refuses the connection (clear server log "module is read only"; the client exits non-zero, nothing is transferred). A future pull/list operation can be opened up when it exists; the knob is already stored. diff --git a/src/server/server.c b/src/server/server.c index 842931b..92b65cf 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -23,8 +23,10 @@ #include #include #include +#include #include #include +#include #include static char* authorized_root; @@ -68,6 +70,11 @@ typedef struct ModuleGateContext { activity (operator --no-super, or a daemon module without the `client owner = yes` opt-in); -1 when the config's own mode stands. */ int super_mode_override; + /* Numeric peer address (INET6_ADDRSTRLEN is always enough), filled once by + * server_module_gate. has_peer_ip is false when getpeername/inet_ntop could + * not classify the peer; an ACL-configured module then fails closed. */ + bool has_peer_ip; + char peer_ip[INET6_ADDRSTRLEN]; } ModuleGateContext; /* Server half of the SCRAM challenge/response (A7 remediation, protocol @@ -320,6 +327,66 @@ static const char* module_gate_check_ownership(const Config* config, const Daemo return NULL; } +/* Online-guessing throttle: sleep the configured `auth failure delay` + * milliseconds after a failed authentication. Runs in the per-connection + * forked child, so it never blocks the accept loop or another connection. 0 + * disables it; the parser already caps it at DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS. + * Resumes after EINTR so a signal cannot cut the delay short. */ +static void daemon_auth_failure_delay(void) { + if (!g_daemon_conf || g_daemon_conf->global.auth_failure_delay_ms <= 0) + return; + int ms = g_daemon_conf->global.auth_failure_delay_ms; + struct timespec delay; + delay.tv_sec = ms / 1000; + delay.tv_nsec = (long)(ms % 1000) * 1000000L; + while (nanosleep(&delay, &delay) != 0 && errno == EINTR) + ; +} + +/* Host access control (global then per-module). A configured list makes an + * unprovable peer fail closed. Deny always takes precedence over allow, and a + * non-empty allow list rejects a peer that matches none of its entries. The + * audit line names the peer, the module and the outcome. Returns an + * error string on refusal, NULL on acceptance. */ +static const char* module_gate_check_hosts(const Config* config, const DaemonModule* module, + ModuleGateContext* gate_ctx) { + bool global_restricted = daemon_hosts_restricted( + g_daemon_conf->global.hosts_allow, g_daemon_conf->global.hosts_allow_count, + g_daemon_conf->global.hosts_deny, g_daemon_conf->global.hosts_deny_count); + bool module_restricted = daemon_hosts_restricted(module->hosts_allow, module->hosts_allow_count, + module->hosts_deny, module->hosts_deny_count); + if (!global_restricted && !module_restricted) + return NULL; + if (!gate_ctx || !gate_ctx->has_peer_ip) { + log_message(LOG_LEVEL_WARNING, + "daemon module '%s': cannot determine peer address with host ACLs configured; " + "refusing (fail closed)", + config->module); + return "cannot verify the client host against host access controls"; + } + const char* peer = gate_ctx->peer_ip; + if (global_restricted && !daemon_hosts_allowed(peer, g_daemon_conf->global.hosts_allow, + g_daemon_conf->global.hosts_allow_count, + g_daemon_conf->global.hosts_deny, + g_daemon_conf->global.hosts_deny_count)) { + log_message(LOG_LEVEL_WARNING, + "daemon module '%s': peer %s denied by global 'hosts allow'/'hosts deny'; " + "refusing", + config->module, peer); + return "client host is not permitted by this daemon"; + } + if (module_restricted && + !daemon_hosts_allowed(peer, module->hosts_allow, module->hosts_allow_count, + module->hosts_deny, module->hosts_deny_count)) { + log_message(LOG_LEVEL_WARNING, + "daemon module '%s': peer %s denied by module 'hosts allow'/'hosts deny'; " + "refusing", + config->module, peer); + return "client host is not permitted by this daemon module"; + } + return NULL; +} + /* A7 auth gate: runs the SCRAM challenge/response for an auth-required module * BEFORE the module root is installed and before any data moves. Returns * MODULE_AUTH_ACCEPTED when the module needs no auth or the handshake succeeds, @@ -376,16 +443,21 @@ static ModuleAuthResult module_gate_authenticate(const Config* config, const Dae * via MODULE_AUTH_TERMINATED; the username may be logged (never the password * or any derived proof). */ if (!server_auth_handshake(gate_ctx->fd, config, module)) { + const char* peer = gate_ctx->has_peer_ip ? gate_ctx->peer_ip : "unknown"; char* escaped_user = config->auth_user ? output_escape(config->auth_user, config->eight_bit_output) : NULL; - log_message(LOG_LEVEL_ERROR, "daemon module '%s': authentication failed for user '%s'", - config->module, escaped_user ? escaped_user : "(none)"); + log_message(LOG_LEVEL_WARNING, + "daemon module '%s': authentication failed for user '%s' from %s; refusing", + config->module, escaped_user ? escaped_user : "(none)", peer); free(escaped_user); + /* Rate-limit online guessing per connection (no delay on success). */ + daemon_auth_failure_delay(); return MODULE_AUTH_TERMINATED; } char* escaped_user = output_escape(config->auth_user, config->eight_bit_output); - log_message(LOG_LEVEL_INFO, "daemon module '%s': user '%s' authenticated", config->module, - escaped_user ? escaped_user : ""); + log_message(LOG_LEVEL_INFO, "daemon module '%s': user '%s' from %s authenticated", config->module, + escaped_user ? escaped_user : "", + gate_ctx->has_peer_ip ? gate_ctx->peer_ip : "unknown"); free(escaped_user); return MODULE_AUTH_ACCEPTED; } @@ -478,6 +550,19 @@ static const char* server_module_gate(const Config* config, void* context) { const DaemonModule* module = module_gate_lookup_module(config, &error); if (!module) return error; + /* Resolve the peer once, before any auth or ownership work, so the host ACL + * and the audit lines all use the same address. A module with ACLs fails + * closed when the peer cannot be classified; an ACL-free module continues + * (the accept loop still logged the address). */ + if (gate_ctx) { + gate_ctx->has_peer_ip = + utils_fd_peer_ip(gate_ctx->fd, gate_ctx->peer_ip, sizeof(gate_ctx->peer_ip)); + if (!gate_ctx->has_peer_ip) + log_message(LOG_LEVEL_DEBUG, "daemon module '%s': peer address unavailable", config->module); + } + error = module_gate_check_hosts(config, module, gate_ctx); + if (error) + return error; error = module_gate_check_ownership(config, module, gate_ctx); if (error) return error; @@ -503,6 +588,8 @@ void handler(int file_descriptor) { gate_ctx.ssl = ssl; gate_ctx.fd = file_descriptor; gate_ctx.super_mode_override = -1; + gate_ctx.has_peer_ip = false; + gate_ctx.peer_ip[0] = '\0'; /* All teardown state starts empty so the single `done` epilogue is safe to * reach from any error path (including before the config frame arrives). */ Config* config = NULL; @@ -785,7 +872,8 @@ static void print_server_usage(void) { printf(" --config=FILE Daemon config file (default: ~/.config/fastsync/\n"); printf(" fastsyncd.conf, else /etc/fastsyncd.conf)\n"); printf(" --dparam=KEY=VALUE Override one global config key on the command line\n"); - printf(" (port, motd file, address)\n"); + printf(" (port, motd file, address, max connections,\n"); + printf(" auth failure delay, hosts allow, hosts deny)\n"); printf(" --no-detach Stay in the foreground (default detaches to\n"); printf(" background when running --daemon)\n"); printf(" --password-file=FILE Credential store for modules that declare\n"); @@ -1000,6 +1088,12 @@ int main(int argc, char* argv[]) { "and device nodes within that module root -- pair it with `auth users` " "unless the module is intentionally open to the network", g_daemon_conf->modules[i].name); + if (g_daemon_conf->modules[i].max_connections > 0) + log_message(LOG_LEVEL_WARNING, + "daemon module '%s': per-module 'max connections' is stored but not enforced " + "per module; the global 'max connections' cap (%d) applies to the whole " + "listener", + g_daemon_conf->modules[i].name, g_daemon_conf->global.max_connections); } /* Daemon credential store (Wave B). --password-file and --early-input * feed the same store, loaded BEFORE the listener forks so every @@ -1064,6 +1158,8 @@ int main(int argc, char* argv[]) { exit_code = 1; goto out; } + if (g_daemon_conf) + server_set_max_connections(g_server, (unsigned int)g_daemon_conf->global.max_connections); if (opts.use_tls) { if (!opts.tls_cert || !opts.tls_key || !opts.tls_ca || !opts.client_cn) { fprintf(stderr, "Error: --tls requires --cert, --key, --ca, and --client-cn\n"); diff --git a/src/shared/daemon_conf.c b/src/shared/daemon_conf.c index 7a1c83d..10c0f63 100644 --- a/src/shared/daemon_conf.c +++ b/src/shared/daemon_conf.c @@ -1,9 +1,13 @@ #include "daemon_conf.h" #include "credentials.h" #include "utils.h" +#include #include #include +#include +#include #include +#include #include #include #include @@ -49,6 +53,144 @@ static bool parse_bool_value(const char* value, bool* out) { return false; } +/* Parse an IPv4/IPv6 CIDR "addr/prefix" into `bytes`/`*family`. Returns false + * for a malformed address, a missing/oversized prefix, or a prefix that does + * not fit the address family. */ +static bool parse_cidr(const char* cidr, int* prefix_out, uint8_t* bytes, int* family_out) { + const char* slash = strchr(cidr, '/'); + if (!slash) + return false; + size_t addr_len = (size_t)(slash - cidr); + if (addr_len == 0 || addr_len >= INET6_ADDRSTRLEN) + return false; + char addr[INET6_ADDRSTRLEN]; + memcpy(addr, cidr, addr_len); + addr[addr_len] = '\0'; + char* end = NULL; + long prefix = strtol(slash + 1, &end, 10); + if (end == slash + 1 || *end != '\0') + return false; + struct in_addr v4; + struct in6_addr v6; + if (inet_pton(AF_INET, addr, &v4) == 1) { + if (prefix < 0 || prefix > 32) + return false; + memcpy(bytes, &v4, sizeof(v4)); + *prefix_out = (int)prefix; + *family_out = AF_INET; + return true; + } + if (inet_pton(AF_INET6, addr, &v6) == 1) { + if (prefix < 0 || prefix > 128) + return false; + memcpy(bytes, &v6, sizeof(v6)); + *prefix_out = (int)prefix; + *family_out = AF_INET6; + return true; + } + return false; +} + +/* A host pattern is valid when it is non-empty and, when it contains a '/', its + * address/prefix halves parse as a CIDR. Literals, `*` and globs are accepted + * as-is (a glob only ever matches a peer of the same shape). */ +static bool host_pattern_valid(const char* pattern) { + if (!pattern || *pattern == '\0') + return false; + if (!strchr(pattern, '/')) + return true; + uint8_t bytes[16]; + int prefix; + int family; + return parse_cidr(pattern, &prefix, bytes, &family); +} + +/* Append every comma- and/or whitespace-separated host pattern in `value` to + * the heap-owned list. Returns false (err filled) on an invalid pattern or an + * allocation failure. */ +static bool store_host_list(char*** list, int* count, const char* value, const char* key, + const char* module_name, char* err, size_t err_size) { + char* copy = str_dup(value); + if (!copy) { + if (module_name) + set_error(err, err_size, "out of memory parsing '%s' for module '%s'", key, module_name); + else + set_error(err, err_size, "out of memory parsing '%s'", key); + return false; + } + char* save = NULL; + for (char* token = strtok_r(copy, ", \t", &save); token; token = strtok_r(NULL, ", \t", &save)) { + if (!host_pattern_valid(token)) { + if (module_name) + set_error(err, err_size, "module '%s': invalid host pattern '%s' in '%s'", module_name, + token, key); + else + set_error(err, err_size, "invalid host pattern '%s' in '%s'", token, key); + free(copy); + return false; + } + char** grown = realloc(*list, (size_t)(*count + 1) * sizeof(char*)); + if (!grown) { + if (module_name) + set_error(err, err_size, "out of memory parsing '%s' for module '%s'", key, module_name); + else + set_error(err, err_size, "out of memory parsing '%s'", key); + free(copy); + return false; + } + *list = grown; + char* dup = str_dup(token); + if (!dup) { + if (module_name) + set_error(err, err_size, "out of memory parsing '%s' for module '%s'", key, module_name); + else + set_error(err, err_size, "out of memory parsing '%s'", key); + free(copy); + return false; + } + (*list)[(*count)++] = dup; + } + free(copy); + return true; +} + +/* Parse a `max connections` value: a positive integer (0/negative/garbage are + * rejected because they would silently disable the cap or admit nothing). */ +static bool store_max_connections(int* slot, const char* value, const char* module_name, char* err, + size_t err_size) { + char* end = NULL; + errno = 0; + long n = strtol(value, &end, 10); + if (*value == '\0' || errno != 0 || *end != '\0' || n <= 0 || n > INT_MAX) { + if (module_name) + set_error(err, err_size, + "module '%s': invalid 'max connections' '%s' (must be a positive " + "integer)", + module_name, value); + else + set_error(err, err_size, "invalid 'max connections' '%s' (must be a positive integer)", + value); + return false; + } + *slot = (int)n; + return true; +} + +/* Parse an `auth failure delay` value: 0 (disabled) through the configured cap. */ +static bool store_auth_failure_delay(int* slot, const char* value, char* err, size_t err_size) { + char* end = NULL; + errno = 0; + long n = strtol(value, &end, 10); + if (*value == '\0' || errno != 0 || *end != '\0' || n < 0 || + n > DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS) { + set_error(err, err_size, "invalid 'auth failure delay' '%s' (must be 0-%d milliseconds)", value, + DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS); + return false; + } + *slot = (int)n; + return true; +} + bool daemon_module_name_valid(const char* name) { if (!name || *name == '\0') return false; @@ -69,14 +211,25 @@ DaemonConf* daemon_conf_create(void) { if (!conf) return NULL; conf->global.port = DAEMON_CONF_DEFAULT_PORT; + conf->global.max_connections = DAEMON_CONF_DEFAULT_MAX_CONNECTIONS; + conf->global.auth_failure_delay_ms = DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS; return conf; } +/* Free a heap-owned pattern list of `count` entries. */ +static void free_string_list(char** list, int count) { + for (int i = 0; i < count; i++) + free(list[i]); + free(list); +} + void daemon_conf_free(DaemonConf* conf) { if (!conf) return; free(conf->global.motd_file); free(conf->global.address); + free_string_list(conf->global.hosts_allow, conf->global.hosts_allow_count); + free_string_list(conf->global.hosts_deny, conf->global.hosts_deny_count); for (int i = 0; i < conf->module_count; i++) { DaemonModule* m = &conf->modules[i]; free(m->name); @@ -84,6 +237,8 @@ void daemon_conf_free(DaemonConf* conf) { for (int j = 0; j < m->auth_user_count; j++) free(m->auth_users[j]); free(m->auth_users); + free_string_list(m->hosts_allow, m->hosts_allow_count); + free_string_list(m->hosts_deny, m->hosts_deny_count); } free(conf->modules); free(conf); @@ -141,6 +296,16 @@ static bool apply_global_key(DaemonConf* conf, char* key, const char* value, cha } return true; } + if (key_equals(key, "max connections")) + return store_max_connections(&conf->global.max_connections, value, NULL, err, err_size); + if (key_equals(key, "auth failure delay")) + return store_auth_failure_delay(&conf->global.auth_failure_delay_ms, value, err, err_size); + if (key_equals(key, "hosts allow")) + return store_host_list(&conf->global.hosts_allow, &conf->global.hosts_allow_count, value, + "hosts allow", NULL, err, err_size); + if (key_equals(key, "hosts deny")) + return store_host_list(&conf->global.hosts_deny, &conf->global.hosts_deny_count, value, + "hosts deny", NULL, err, err_size); set_error(err, err_size, "unknown global key '%s'", key); return false; } @@ -220,6 +385,14 @@ static bool apply_module_key(DaemonModule* module, char* key, char* value, char* free(list); return true; } + if (key_equals(key, "max connections")) + return store_max_connections(&module->max_connections, value, module->name, err, err_size); + if (key_equals(key, "hosts allow")) + return store_host_list(&module->hosts_allow, &module->hosts_allow_count, value, "hosts allow", + module->name, err, err_size); + if (key_equals(key, "hosts deny")) + return store_host_list(&module->hosts_deny, &module->hosts_deny_count, value, "hosts deny", + module->name, err, err_size); set_error(err, err_size, "unknown key '%s' in module '%s'", key, module->name); return false; } @@ -459,3 +632,86 @@ int daemon_conf_apply_dparam(DaemonConf* conf, const char* assignment, char* err free(copy); return ok ? 0 : -1; } + +/* Compare the first `prefix` bits of two 16-byte address buffers. */ +static bool bit_prefix_match(const uint8_t* a, const uint8_t* b, int prefix) { + int whole = prefix / 8; + if (whole > 0 && memcmp(a, b, (size_t)whole) != 0) + return false; + int remainder = prefix % 8; + if (remainder == 0) + return true; + uint8_t mask = (uint8_t)(0xffu << (8 - remainder)); + return (a[whole] & mask) == (b[whole] & mask); +} + +/* Case-insensitive glob match used for hostname patterns. Falls back to the + * shared case-sensitive matcher when an operand is too long for the stack + * buffers. */ +static bool host_glob_match(const char* pattern, const char* str) { + char pbuf[256]; + char sbuf[256]; + size_t plen = strlen(pattern); + size_t slen = strlen(str); + if (plen >= sizeof(pbuf) || slen >= sizeof(sbuf)) + return glob_match(pattern, str); + for (size_t i = 0; i <= plen; i++) + pbuf[i] = (char)tolower((unsigned char)pattern[i]); + for (size_t i = 0; i <= slen; i++) + sbuf[i] = (char)tolower((unsigned char)str[i]); + return glob_match(pbuf, sbuf); +} + +bool daemon_host_pattern_match(const char* pattern, const char* peer_ip) { + if (!pattern || *pattern == '\0' || !peer_ip || *peer_ip == '\0') + return false; + if (strcmp(pattern, "*") == 0) + return true; + if (strchr(pattern, '/')) { + uint8_t pattern_bytes[16]; + uint8_t peer_bytes[16]; + int prefix = 0; + int family = AF_UNSPEC; + if (!parse_cidr(pattern, &prefix, pattern_bytes, &family)) + return false; + if (inet_pton(family, peer_ip, peer_bytes) != 1) + return false; + return bit_prefix_match(pattern_bytes, peer_bytes, prefix); + } + struct in_addr pattern_v4; + struct in_addr peer_v4; + if (inet_pton(AF_INET, pattern, &pattern_v4) == 1) + return inet_pton(AF_INET, peer_ip, &peer_v4) == 1 && pattern_v4.s_addr == peer_v4.s_addr; + struct in6_addr pattern_v6; + struct in6_addr peer_v6; + if (inet_pton(AF_INET6, pattern, &pattern_v6) == 1) + return inet_pton(AF_INET6, peer_ip, &peer_v6) == 1 && + memcmp(&pattern_v6, &peer_v6, sizeof(pattern_v6)) == 0; + /* Not a literal: a hostname/glob pattern. */ + return host_glob_match(pattern, peer_ip); +} + +bool daemon_hosts_allowed(const char* peer_ip, char* const* allow, int allow_count, + char* const* deny, int deny_count) { + if (!peer_ip) + return false; + for (int i = 0; i < deny_count; i++) { + if (daemon_host_pattern_match(deny[i], peer_ip)) + return false; + } + if (allow_count > 0) { + for (int i = 0; i < allow_count; i++) { + if (daemon_host_pattern_match(allow[i], peer_ip)) + return true; + } + return false; + } + return true; +} + +bool daemon_hosts_restricted(char* const* allow, int allow_count, char* const* deny, + int deny_count) { + (void)allow; + (void)deny; + return allow_count > 0 || deny_count > 0; +} diff --git a/src/shared/daemon_conf.h b/src/shared/daemon_conf.h index 0ecfe24..d82276b 100644 --- a/src/shared/daemon_conf.h +++ b/src/shared/daemon_conf.h @@ -52,14 +52,32 @@ typedef struct DaemonModule { activities. Without it the daemon refuses all of them. */ char** auth_users; /* `auth users = a,b`; Wave B credential list */ int auth_user_count; + /* `max connections = N` (optional per-module cap). 0 means "not set" + * (inherit the global cap). Parsed, stored, and validated, but NOT enforced + * per-module: connections are counted in the accept-loop parent before the + * client's module is known, so only the global cap is enforced (see + * transport_tcp.c and the Daemon Mode notes in RSYNC_COMPAT.md). */ + int max_connections; + char** hosts_allow; /* `hosts allow = a,b`; host access allow patterns */ + int hosts_allow_count; + char** hosts_deny; /* `hosts deny = a,b`; host access deny patterns */ + int hosts_deny_count; } DaemonModule; /* Global (pre-module) scalar keys. `motd file` is parsed and stored but has * no wire effect yet (MOTD display is Wave C). */ typedef struct DaemonConfGlobals { - int port; /* `port`, default DAEMON_CONF_DEFAULT_PORT (873) */ - char* motd_file; /* `motd file`, may be NULL */ - char* address; /* `address` (optional bind address), may be NULL */ + int port; /* `port`, default DAEMON_CONF_DEFAULT_PORT (873) */ + char* motd_file; /* `motd file`, may be NULL */ + char* address; /* `address` (optional bind address), may be NULL */ + int max_connections; /* `max connections`, default + DAEMON_CONF_DEFAULT_MAX_CONNECTIONS (100) */ + int auth_failure_delay_ms; /* `auth failure delay`, milliseconds; default + DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS */ + char** hosts_allow; /* `hosts allow`; global host access allow patterns */ + int hosts_allow_count; + char** hosts_deny; /* `hosts deny`; global host access deny patterns */ + int hosts_deny_count; } DaemonConfGlobals; typedef struct DaemonConf { @@ -69,6 +87,14 @@ typedef struct DaemonConf { } DaemonConf; #define DAEMON_CONF_DEFAULT_PORT 873 +/* Default global connection cap when `max connections` is absent. Matches the + * historical hardcoded listener value. */ +#define DAEMON_CONF_DEFAULT_MAX_CONNECTIONS 100 +/* Default `auth failure delay` in milliseconds (0 disables the throttle). */ +#define DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS 500 +/* Largest accepted `auth failure delay`, so a typo cannot pin a connection + * child in nanosleep for an absurd time. */ +#define DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS 60000 /* Longest accepted config line (excluding the trailing newline). Longer lines * are rejected rather than buffered unboundedly. */ #define DAEMON_CONF_MAX_LINE 4096 @@ -99,9 +125,30 @@ const DaemonModule* daemon_conf_find_module(const DaemonConf* conf, const char* bool daemon_module_name_valid(const char* name); /* Parse one --dparam=KEY=VALUE (or "--dparam KEY=VALUE") override string and - * apply it to the global scalars only. Keys are case-insensitive and limited - * to the global scalar keys defined by the grammar (port, motd file, address). - * Returns 0 on success, -1 on error (err filled). */ + * apply it to the global keys only. Keys are case-insensitive and limited to + * the global keys defined by the grammar (port, motd file, address, + * max connections, auth failure delay, hosts allow, hosts deny). Returns 0 on + * success, -1 on error (err filled). */ int daemon_conf_apply_dparam(DaemonConf* conf, const char* assignment, char* err, size_t err_size); +/* Host access-control matching (pure; no I/O). `daemon_host_pattern_match` + * matches one configured pattern against a numeric peer IP string. Supported + * patterns: `*` (match anything), an IPv4/IPv6 literal, an IPv4/IPv6 CIDR + * (`10.0.0.0/8`, `2001:db8::/32`), or a glob (`*.example.com`) evaluated with + * the same matcher as file globs; a glob only matches a peer string of the + * same shape, so a numeric peer never matches a hostname glob. */ +bool daemon_host_pattern_match(const char* pattern, const char* peer_ip); + +/* rsync-like combined decision over a deny list and an allow list: a matching + * deny rejects (deny takes precedence); otherwise, when any allow entries + * exist, a peer that matches none is rejected; with no allow entries every + * peer not denied is accepted. An empty/unset pair returns true. */ +bool daemon_hosts_allowed(const char* peer_ip, char* const* allow, int allow_count, + char* const* deny, int deny_count); + +/* True when at least one allow or deny pattern is configured (i.e. an + * unprovable peer must fail closed rather than being treated as unrestricted). */ +bool daemon_hosts_restricted(char* const* allow, int allow_count, char* const* deny, + int deny_count); + #endif diff --git a/src/shared/transport_tcp.c b/src/shared/transport_tcp.c index 0a4aeef..a3f9a07 100644 --- a/src/shared/transport_tcp.c +++ b/src/shared/transport_tcp.c @@ -115,6 +115,11 @@ Server* server_create(int port) { return server_create_ex(port, NULL); } +void server_set_max_connections(Server* server, unsigned int max_connections) { + if (server && max_connections > 0) + server->max_connections = max_connections; +} + void server_delete(Server** server) { if (server == NULL || *server == NULL) return; @@ -135,7 +140,7 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil } signal(SIGCHLD, sigchld_handler); while (1) { - struct sockaddr_in client_addr; + struct sockaddr_storage client_addr; socklen_t client_len = sizeof(client_addr); int fd = accept(server->file_descriptor, (struct sockaddr*)&client_addr, &client_len); if (fd < 0) { @@ -143,13 +148,16 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil continue; } tcp_apply_socket_timeout(fd); + char peer[128]; + if (!utils_sockaddr_to_string((const struct sockaddr*)&client_addr, peer, sizeof(peer))) + snprintf(peer, sizeof(peer), "unknown"); if ((unsigned int)g_active_connections >= server->max_connections) { - log_message(LOG_LEVEL_WARNING, "Max connections (%u) reached, rejecting", - server->max_connections); + log_message(LOG_LEVEL_WARNING, "Max connections (%u) reached, rejecting %s", + server->max_connections, peer); close(fd); continue; } - log_message(LOG_LEVEL_INFO, "%s", log_fmt); + log_message(LOG_LEVEL_INFO, "%s from %s", log_fmt, peer); pid_t pid = fork(); if (pid == 0) { /* Connection children must not run the parent's global cleanup(): it diff --git a/src/shared/transport_tcp.h b/src/shared/transport_tcp.h index b5937e6..4439003 100644 --- a/src/shared/transport_tcp.h +++ b/src/shared/transport_tcp.h @@ -45,6 +45,9 @@ typedef struct { Server* server_create_ex(int port, const ServerBindOptions* bind_opts); Server* server_create(int port); +/* Override the listener's connection cap (the global daemon `max connections` + * value). A non-positive value is ignored so the default cap stands. */ +void server_set_max_connections(Server* server, unsigned int max_connections); bool server_listen(Server* server, void (*handler)(int file_descriptor)); void server_accept_loop(Server* server, void (*child_fn)(int, void*), void* child_ctx, const char* log_fmt); diff --git a/src/shared/utils.c b/src/shared/utils.c index 16ea671..e00ec25 100644 --- a/src/shared/utils.c +++ b/src/shared/utils.c @@ -587,6 +587,62 @@ bool utils_fd_peer_is_local(int fd) { return utils_sockaddr_is_loopback((const struct sockaddr*)&peer); } +/* Numeric peer address of a connected fd. Only AF_INET/AF_INET6 peers are + formatted; every other descriptor/family (pipe, AF_UNIX socketpair, ...) or a + getpeername failure returns false with buf emptied. The caller must treat + that as "cannot tell". */ +bool utils_fd_peer_ip(int fd, char* buf, size_t len) { + if (!buf || len == 0) + return false; + buf[0] = '\0'; + if (fd < 0) + return false; + struct sockaddr_storage peer; + socklen_t peer_len = sizeof(peer); + if (getpeername(fd, (struct sockaddr*)&peer, &peer_len) != 0) + return false; + const void* src = NULL; + int family = peer.ss_family; + if (family == AF_INET) + src = &((const struct sockaddr_in*)&peer)->sin_addr; + else if (family == AF_INET6) + src = &((const struct sockaddr_in6*)&peer)->sin6_addr; + else + return false; + return inet_ntop(family, src, buf, (socklen_t)len) != NULL; +} + +/* "ip:port" / "[ip]:port" for a connected peer, used to log the connecting + address in the accept loop. Returns false for a non-INET family. */ +bool utils_sockaddr_to_string(const struct sockaddr* addr, char* buf, size_t len) { + if (!addr || !buf || len == 0) + return false; + buf[0] = '\0'; + char ip[INET6_ADDRSTRLEN]; + unsigned short port; + int written; + if (addr->sa_family == AF_INET) { + const struct sockaddr_in* v4 = (const struct sockaddr_in*)addr; + if (!inet_ntop(AF_INET, &v4->sin_addr, ip, sizeof(ip))) + return false; + port = ntohs(v4->sin_port); + written = snprintf(buf, len, "%s:%u", ip, port); + } else if (addr->sa_family == AF_INET6) { + const struct sockaddr_in6* v6 = (const struct sockaddr_in6*)addr; + if (!inet_ntop(AF_INET6, &v6->sin6_addr, ip, sizeof(ip))) + return false; + port = ntohs(v6->sin6_port); + written = snprintf(buf, len, "[%s]:%u", ip, port); + } else { + return false; + } + if (written < 0 || (size_t)written >= len) { + buf[0] = '\0'; + return false; + } + return true; +} + /* True when a client-supplied host string names a loopback destination: "localhost", any 127.0.0.0/8 literal, "::1", or "[::1]". */ bool utils_host_is_loopback(const char* host) { diff --git a/src/shared/utils.h b/src/shared/utils.h index 4d09c67..4c9144a 100644 --- a/src/shared/utils.h +++ b/src/shared/utils.h @@ -77,5 +77,13 @@ bool append_tail_length(unsigned long long old_size, unsigned long long check_si bool utils_sockaddr_is_loopback(const struct sockaddr* addr); bool utils_fd_peer_is_local(int fd); bool utils_host_is_loopback(const char* host); +/* Numeric peer address of a connected fd (INET6_ADDRSTRLEN is always enough). + * Returns false and leaves buf empty when the fd is not a connected INET socket + * or getpeername/inet_ntop fails. Used by the daemon host-access gate; a false + * return is "cannot tell" and must be treated as fail-closed when ACLs apply. */ +bool utils_fd_peer_ip(int fd, char* buf, size_t len); +/* Format a sockaddr as "ip:port" (IPv4) or "[ip]:port" (IPv6) for logging. + * Returns false (buf emptied) for a non-INET family or a formatting failure. */ +bool utils_sockaddr_to_string(const struct sockaddr* addr, char* buf, size_t len); #endif diff --git a/tests/integration/test_daemon.py b/tests/integration/test_daemon.py index 30410e8..98c963d 100644 --- a/tests/integration/test_daemon.py +++ b/tests/integration/test_daemon.py @@ -51,6 +51,7 @@ READONLY_MODULE = os.path.join(MODULE_ROOT, "readonly") AUTH_MODULE = os.path.join(MODULE_ROOT, "auth") TEAM_MODULE = os.path.join(MODULE_ROOT, "team") OWNER_MODULE = os.path.join(MODULE_ROOT, "owner") +DENIED_MODULE = os.path.join(MODULE_ROOT, "denied") CONF_FILE = os.path.join(TEST_DATA_DIR, "fastsyncd.conf") CRED_FILE = os.path.join(TEST_DATA_DIR, "fastsyncd.passwd") STARTFAIL_CONF = os.path.join(TEST_DATA_DIR, "fastsyncd_startfail.conf") @@ -191,7 +192,7 @@ def _config_port(config_path): @pytest.fixture(scope="module", autouse=True) def daemon_env(): for d in (MODULE_ROOT, FILES_MODULE, READONLY_MODULE, AUTH_MODULE, TEAM_MODULE, OWNER_MODULE, - DETACH_MODULE): + DENIED_MODULE, DETACH_MODULE): shutil.rmtree(d, ignore_errors=True) os.makedirs(d, exist_ok=True) generate_test_files(SOURCE_DIR, full=False) @@ -231,7 +232,12 @@ def daemon_env(): "[owner]\n" "path = %s\n" "client owner = yes\n" - % (config_port, FILES_MODULE, READONLY_MODULE, AUTH_MODULE, TEAM_MODULE, OWNER_MODULE)) + "\n" + "[denied]\n" + "path = %s\n" + "hosts deny = 127.0.0.1\n" + % (config_port, FILES_MODULE, READONLY_MODULE, AUTH_MODULE, TEAM_MODULE, OWNER_MODULE, + DENIED_MODULE)) # A dedicated config for the fail-closed startup check: an auth-required # module with no credential store must refuse to start. Its own free port @@ -368,6 +374,15 @@ class TestDaemonRejection: result = _push("127.0.0.1::/sub", daemon.port) assert result.returncode != 0 + @pytest.mark.ci + def test_hosts_deny_rejects_loopback(self, daemon): + """Host access control: a module with `hosts deny = 127.0.0.1` refuses a + loopback client at the config gate, before any data is exchanged.""" + before = self._tree_files() + result = _push("127.0.0.1::denied", daemon.port) + assert result.returncode != 0 + assert self._tree_files() == before, "host-denied connection wrote under the module root" + def test_dotdot_destination_rejected(self, daemon): """A '..' path expansion in the module-relative path is refused at parse time so a client cannot escape the module root while it is still on the diff --git a/tests/test_daemon_conf.c b/tests/test_daemon_conf.c index 1f54a2e..7e7577a 100644 --- a/tests/test_daemon_conf.c +++ b/tests/test_daemon_conf.c @@ -31,6 +31,10 @@ static void test_daemon_conf_create_defaults() { EXPECT_EQ_INT(conf->global.port, DAEMON_CONF_DEFAULT_PORT); EXPECT_NULL(conf->global.motd_file); EXPECT_NULL(conf->global.address); + EXPECT_EQ_INT(conf->global.max_connections, DAEMON_CONF_DEFAULT_MAX_CONNECTIONS); + EXPECT_EQ_INT(conf->global.auth_failure_delay_ms, DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS); + EXPECT_EQ_INT(conf->global.hosts_allow_count, 0); + EXPECT_EQ_INT(conf->global.hosts_deny_count, 0); EXPECT_EQ_INT(conf->module_count, 0); daemon_conf_free(conf); } @@ -310,6 +314,14 @@ static void test_daemon_conf_dparam_override() { EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "port = 9000", err, sizeof(err)), 0); EXPECT_EQ_INT(conf->global.port, 9000); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "max connections=7", err, sizeof(err)), 0); + EXPECT_EQ_INT(conf->global.max_connections, 7); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "AUTH FAILURE DELAY=1500", err, sizeof(err)), 0); + EXPECT_EQ_INT(conf->global.auth_failure_delay_ms, 1500); + EXPECT_EQ_INT( + daemon_conf_apply_dparam(conf, "hosts allow=127.0.0.1,10.0.0.0/8", err, sizeof(err)), 0); + EXPECT_EQ_INT(conf->global.hosts_allow_count, 2); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "port=notaport", err, sizeof(err)), -1); EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "bogus=1", err, sizeof(err)), -1); EXPECT_TRUE(strstr(err, "unknown global key") != NULL); @@ -366,6 +378,114 @@ static void test_daemon_conf_auth_users_validated() { daemon_conf_free(ok_conf); } +/* Wave 3 daemon hardening: configurable global/per-module connection caps, + * auth-failure throttle and host access lists parse strictly (valid values are + * stored, malformed values fail the whole load). */ +static void test_daemon_conf_limits_and_hosts_parse() { + char* path; + char err[256]; + EXPECT_EQ_INT(write_conf("max connections = 25\n" + "auth failure delay = 0\n" + "hosts allow = 10.0.0.0/8, *.example.com\n" + "hosts deny = 192.168.0.1 2001:db8::/32\n" + "\n" + "[m]\n" + "path = /x\n" + "max connections = 3\n" + "hosts allow = 127.0.0.1\n" + "hosts deny = *\n", + &path), + 0); + DaemonConf* conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_EQ_INT(conf->global.max_connections, 25); + EXPECT_EQ_INT(conf->global.auth_failure_delay_ms, 0); + EXPECT_EQ_INT(conf->global.hosts_allow_count, 2); + EXPECT_EQ_STR(conf->global.hosts_allow[0], "10.0.0.0/8"); + EXPECT_EQ_STR(conf->global.hosts_allow[1], "*.example.com"); + EXPECT_EQ_INT(conf->global.hosts_deny_count, 2); + EXPECT_EQ_STR(conf->global.hosts_deny[0], "192.168.0.1"); + EXPECT_EQ_STR(conf->global.hosts_deny[1], "2001:db8::/32"); + EXPECT_EQ_INT(conf->modules[0].max_connections, 3); + EXPECT_EQ_INT(conf->modules[0].hosts_allow_count, 1); + EXPECT_EQ_STR(conf->modules[0].hosts_allow[0], "127.0.0.1"); + EXPECT_EQ_INT(conf->modules[0].hosts_deny_count, 1); + EXPECT_EQ_STR(conf->modules[0].hosts_deny[0], "*"); + daemon_conf_free(conf); + + const char* bad_values[] = { + "max connections = 0\n", "max connections = -1\n", "max connections = abc\n", + "auth failure delay = -1\n", "auth failure delay = 70000\n", "auth failure delay = soon\n", + "hosts allow = 10.0.0.0/99\n", "hosts deny = 2001:db8::/129\n", + }; + for (size_t i = 0; i < sizeof(bad_values) / sizeof(bad_values[0]); i++) { + EXPECT_EQ_INT(write_conf(bad_values[i], &path), 0); + const DaemonConf* rejected = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(rejected); + } + + /* The same strictness applies inside a module section. */ + const char* bad_module[] = { + "[m]\npath = /x\nmax connections = 0\n", + "[m]\npath = /x\nhosts allow = 10.0.0.0/40\n", + "[m]\npath = /x\nhosts deny = 999.1.1.1/8\n", + }; + for (size_t i = 0; i < sizeof(bad_module) / sizeof(bad_module[0]); i++) { + EXPECT_EQ_INT(write_conf(bad_module[i], &path), 0); + const DaemonConf* rejected = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(rejected); + EXPECT_TRUE(strstr(err, "invalid") != NULL); + } + + /* An empty hosts list is not an error (no patterns are added). */ + EXPECT_EQ_INT(write_conf("hosts allow = \n[m]\npath = /x\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_EQ_INT(conf->global.hosts_allow_count, 0); + daemon_conf_free(conf); +} + +static void test_daemon_hosts_allowed() { + /* Pattern forms. */ + EXPECT_TRUE(daemon_host_pattern_match("*", "203.0.113.9")); + EXPECT_TRUE(daemon_host_pattern_match("10.0.0.1", "10.0.0.1")); + EXPECT_FALSE(daemon_host_pattern_match("10.0.0.1", "10.0.0.2")); + EXPECT_TRUE(daemon_host_pattern_match("10.0.0.0/8", "10.255.1.2")); + EXPECT_FALSE(daemon_host_pattern_match("10.0.0.0/8", "11.0.0.1")); + EXPECT_TRUE(daemon_host_pattern_match("2001:db8::/32", "2001:db8:1234::5")); + EXPECT_FALSE(daemon_host_pattern_match("2001:db8::/32", "2001:db9::1")); + EXPECT_TRUE(daemon_host_pattern_match("::1", "::1")); + EXPECT_FALSE(daemon_host_pattern_match("::1", "::2")); + EXPECT_TRUE(daemon_host_pattern_match("*.example.com", "host.example.com")); + EXPECT_FALSE(daemon_host_pattern_match("*.example.com", "example.org")); + EXPECT_FALSE(daemon_host_pattern_match(NULL, "10.0.0.1")); + EXPECT_FALSE(daemon_host_pattern_match("10.0.0.1", NULL)); + EXPECT_FALSE(daemon_host_pattern_match("", "10.0.0.1")); + + char* allow[] = {"10.0.0.0/8"}; + char* deny[] = {"10.0.0.1"}; + /* Deny takes precedence over a matching allow. */ + EXPECT_FALSE(daemon_hosts_allowed("10.0.0.1", allow, 1, deny, 1)); + EXPECT_TRUE(daemon_hosts_allowed("10.0.0.2", allow, 1, deny, 1)); + /* A non-empty allow list rejects a peer that matches none of its entries. */ + EXPECT_FALSE(daemon_hosts_allowed("192.168.1.1", allow, 1, NULL, 0)); + /* With only a deny list, everything not denied is accepted. */ + EXPECT_TRUE(daemon_hosts_allowed("192.168.1.1", NULL, 0, deny, 1)); + EXPECT_FALSE(daemon_hosts_allowed("10.0.0.1", NULL, 0, deny, 1)); + /* No lists at all accepts everyone. */ + EXPECT_TRUE(daemon_hosts_allowed("192.168.1.1", NULL, 0, NULL, 0)); + /* An unprovable peer (NULL) never matches an allow list. */ + EXPECT_FALSE(daemon_hosts_allowed(NULL, allow, 1, NULL, 0)); + + EXPECT_FALSE(daemon_hosts_restricted(NULL, 0, NULL, 0)); + EXPECT_TRUE(daemon_hosts_restricted(allow, 1, NULL, 0)); + EXPECT_TRUE(daemon_hosts_restricted(NULL, 0, deny, 1)); +} + static void test_daemon_module_name_valid() { EXPECT_TRUE(daemon_module_name_valid("backup")); EXPECT_TRUE(daemon_module_name_valid("Backup_2")); @@ -398,5 +518,7 @@ void test_daemon_conf() { test_daemon_conf_find_module(); test_daemon_conf_dparam_override(); test_daemon_conf_auth_users_validated(); + test_daemon_conf_limits_and_hosts_parse(); + test_daemon_hosts_allowed(); test_daemon_module_name_valid(); } \ No newline at end of file diff --git a/tests/test_shared_utils.c b/tests/test_shared_utils.c index c9da6cf..86ce24b 100644 --- a/tests/test_shared_utils.c +++ b/tests/test_shared_utils.c @@ -356,6 +356,69 @@ static void test_loopback_helpers() { close(listener); } +/* The daemon host ACL reads the numeric peer address through + * utils_fd_peer_ip. A real loopback TCP peer reports "127.0.0.1"; a pipe or an + * AF_UNIX socketpair has no INET peer and must return false with an empty + * buffer (the fail-closed "cannot tell" result). */ +static void test_fd_peer_ip() { + char ip[INET6_ADDRSTRLEN]; + EXPECT_FALSE(utils_fd_peer_ip(-1, ip, sizeof(ip))); + EXPECT_EQ_STR(ip, ""); + EXPECT_FALSE(utils_fd_peer_ip(-1, NULL, 0)); + + int pipe_fds[2]; + EXPECT_EQ_INT(pipe(pipe_fds), 0); + EXPECT_FALSE(utils_fd_peer_ip(pipe_fds[0], ip, sizeof(ip))); + EXPECT_EQ_STR(ip, ""); + close(pipe_fds[0]); + close(pipe_fds[1]); + + int pair_fds[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, pair_fds), 0); + EXPECT_FALSE(utils_fd_peer_ip(pair_fds[0], ip, sizeof(ip))); + EXPECT_EQ_STR(ip, ""); + close(pair_fds[0]); + close(pair_fds[1]); + + int listener = socket(AF_INET, SOCK_STREAM, 0); + EXPECT_TRUE(listener >= 0); + struct sockaddr_in bind_addr; + memset(&bind_addr, 0, sizeof(bind_addr)); + bind_addr.sin_family = AF_INET; + bind_addr.sin_addr.s_addr = htonl(INADDR_LOOPBACK); + bind_addr.sin_port = 0; + EXPECT_EQ_INT(bind(listener, (const struct sockaddr*)&bind_addr, sizeof(bind_addr)), 0); + EXPECT_EQ_INT(listen(listener, 1), 0); + socklen_t addr_len = sizeof(bind_addr); + EXPECT_EQ_INT(getsockname(listener, (struct sockaddr*)&bind_addr, &addr_len), 0); + int dialer = socket(AF_INET, SOCK_STREAM, 0); + EXPECT_TRUE(dialer >= 0); + EXPECT_EQ_INT(connect(dialer, (const struct sockaddr*)&bind_addr, sizeof(bind_addr)), 0); + int accepted = accept(listener, NULL, NULL); + EXPECT_TRUE(accepted >= 0); + EXPECT_TRUE(utils_fd_peer_ip(accepted, ip, sizeof(ip))); + EXPECT_EQ_STR(ip, "127.0.0.1"); + + /* utils_sockaddr_to_string includes the port for a real peer. */ + struct sockaddr_storage peer; + socklen_t peer_len = sizeof(peer); + EXPECT_EQ_INT(getpeername(accepted, (struct sockaddr*)&peer, &peer_len), 0); + char peer_string[128]; + EXPECT_TRUE( + utils_sockaddr_to_string((const struct sockaddr*)&peer, peer_string, sizeof(peer_string))); + EXPECT_TRUE(strncmp(peer_string, "127.0.0.1:", strlen("127.0.0.1:")) == 0); + close(accepted); + close(dialer); + close(listener); + + /* A non-INET family formats to "unknown" at the call site, not a bogus IP. */ + struct sockaddr sa_unix; + memset(&sa_unix, 0, sizeof(sa_unix)); + sa_unix.sa_family = AF_UNIX; + EXPECT_FALSE(utils_sockaddr_to_string(&sa_unix, peer_string, sizeof(peer_string))); + EXPECT_EQ_STR(peer_string, ""); +} + void test_shared_utils() { test_walker_removes_extras_keeps_manifest_and_protected(); test_walker_max_delete_exceeded_deletes_nothing(); @@ -363,6 +426,7 @@ void test_shared_utils() { test_walker_unlimited_deletes_all(); test_walker_hard_bound_all_or_nothing(); test_loopback_helpers(); + test_fd_peer_ip(); /* --append / --append-verify tail-resume math: a resume is eligible only for a shorter existing destination, and the tail length is then the difference. */ From fc560246c1ba4f9688682bc5329a9627ade08445 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 02:51:07 +0200 Subject: [PATCH 017/155] fix(daemon): close ACL fail-opens (v4-mapped peers, invalid patterns) and cap auth delay --- README.md | 5 ++-- RSYNC_COMPAT.md | 6 ++--- src/shared/daemon_conf.c | 52 +++++++++++++++++++++++++--------------- src/shared/daemon_conf.h | 4 +++- src/shared/utils.c | 17 +++++++++---- tests/test_daemon_conf.c | 16 +++++++++---- 6 files changed, 66 insertions(+), 34 deletions(-) diff --git a/README.md b/README.md index d84146b..2d59b66 100644 --- a/README.md +++ b/README.md @@ -506,8 +506,9 @@ A `[module]` may also set `max connections` (parsed and validated but not enforced per module — the global cap applies to the whole listener) and its own `hosts allow`/`hosts deny`. -Host patterns are `*` (match all), IPv4/IPv6 literals, IPv4/IPv6 CIDR -(`10.0.0.0/8`, `2001:db8::/32`), or hostname globs (`*.example.com`). A matching +Host patterns are `*` (match all), IPv4/IPv6 literals, or IPv4/IPv6 CIDR +(`10.0.0.0/8`, `2001:db8::/32`). Hostnames are not resolved, so hostname globs +are rejected at parse time rather than silently never matching. A matching `hosts deny` rejects; if any `hosts allow` entries exist, a peer matching none of them is rejected; deny takes precedence over allow. The global list is checked before the module list, before authentication, and the connecting peer address diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 8cdeafb..5692ad0 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -635,9 +635,9 @@ now transmits targets (the prior behavior was broken/partial); its status moved **Daemon Mode notes (Wave A protocol 2.15.0; A7 auth protocol 2.19.0; MOTD no bump):** FastSync daemon mode is supported in FastSync's own protocol/config grammar, not rsync's SMB/daemon option encoding. -- **Config grammar** (`fastsyncd.conf`): line-based; an implicit global section first, then `[module]` sections. Keys are case-insensitive, values are trimmed and may be wrapped in one layer of double quotes (`path = "/srv/my dir"`). `#` and `;` at the start of a line (after leading whitespace) are full-line comments; inline comments and `\` continuations are not supported. Lines are bounded (4096 chars). Global keys: `port` (default 873), `motd file` (the daemon sends its bounded, escaped content to a client after the module gate/auth accepts, unless the client passes `--no-motd`), `address` (optional bind address), `max connections` (positive integer cap on concurrent connections, default 100; 0/negative/garbage is a parse error), `auth failure delay` (milliseconds to sleep after a failed authentication, default 500; 0 disables, capped at 60000), `hosts allow` and `hosts deny` (comma- and/or whitespace-separated host access patterns — see the host access control note below). Module keys: `path` (required; the daemon-side authorized root for that module), `read only` (yes/no/true/false/1/0, default no), `client owner` (yes/no/true/false/1/0, default no; opts the module into client-chosen ownership — see below), `auth users` (comma list), `max connections` (optional per-module cap; parsed and stored but **not enforced** — the global cap applies to the whole listener), `hosts allow`/`hosts deny` (per-module host access lists). **Unknown keys and malformed lines are parse-and-reject errors** (never silently ignored), so a typo cannot change what a module serves. -- **Host access control (`hosts allow`/`hosts deny`):** both keys accept a comma- and/or whitespace-separated list of patterns and may appear globally and/or per module (multiple lines append). Supported patterns are `*` (match all), an IPv4 or IPv6 literal (`10.0.0.1`, `2001:db8::1`), an IPv4/IPv6 CIDR (`10.0.0.0/8`, `2001:db8::/32`), and a simple glob (`*.example.com`; globs are matched case-insensitively against the peer string, so a numeric peer never matches a hostname glob). rsync-like semantics: a matching `hosts deny` rejects; if any `hosts allow` entries exist, a peer matching none of them is rejected; deny takes precedence over allow. The daemon enforces the global list first, then the selected module's list, **before authentication** in `server_module_gate`, with an audit log line naming the peer, the module and the outcome. The numeric peer address is obtained with `getpeername`+`inet_ntop` (`utils_fd_peer_ip`, handling both address families); when it cannot be obtained a module with any ACL fails closed (refused), while an ACL-free module continues and logs at debug. A malformed pattern (e.g. an out-of-range CIDR prefix) is a parse error at load time. -- **Connection cap and auth throttle:** the global `max connections` key (default 100) is plumbed into the listener (`transport_tcp.c`), which rejects a connection once the accept-loop parent's active-child count reaches it; the IPv4/IPv6 peer is logged for every accepted connection. The optional per-module `max connections` key is parsed and validated but **not enforced** (connections are counted in the parent before the client's module is known); the daemon logs a startup warning for any module that sets it. On a failed authentication the per-connection child sleeps the global `auth failure delay` (default 500 ms, 0 disables, capped at 60000) via `nanosleep` before the connection closes, rate-limiting online guessing without delaying a success. +- **Config grammar** (`fastsyncd.conf`): line-based; an implicit global section first, then `[module]` sections. Keys are case-insensitive, values are trimmed and may be wrapped in one layer of double quotes (`path = "/srv/my dir"`). `#` and `;` at the start of a line (after leading whitespace) are full-line comments; inline comments and `\` continuations are not supported. Lines are bounded (4096 chars). Global keys: `port` (default 873), `motd file` (the daemon sends its bounded, escaped content to a client after the module gate/auth accepts, unless the client passes `--no-motd`), `address` (optional bind address), `max connections` (positive integer cap on concurrent connections, default 100; 0/negative/garbage is a parse error), `auth failure delay` (milliseconds to sleep after a failed authentication, default 500; 0 disables, capped at 5000), `hosts allow` and `hosts deny` (comma- and/or whitespace-separated host access patterns — see the host access control note below). Module keys: `path` (required; the daemon-side authorized root for that module), `read only` (yes/no/true/false/1/0, default no), `client owner` (yes/no/true/false/1/0, default no; opts the module into client-chosen ownership — see below), `auth users` (comma list), `max connections` (optional per-module cap; parsed and stored but **not enforced** — the global cap applies to the whole listener), `hosts allow`/`hosts deny` (per-module host access lists). **Unknown keys and malformed lines are parse-and-reject errors** (never silently ignored), so a typo cannot change what a module serves. +- **Host access control (`hosts allow`/`hosts deny`):** both keys accept a comma- and/or whitespace-separated list of patterns and may appear globally and/or per module (multiple config-file lines append; a `--dparam` override replaces). Supported patterns are `*` (match all), an IPv4 or IPv6 literal (`10.0.0.1`, `2001:db8::1`), and an IPv4/IPv6 CIDR (`10.0.0.0/8`, `2001:db8::/32`). Hostname patterns are **not** supported: because the peer is always a numeric address and no reverse DNS is performed, a hostname/glob pattern would silently never match, so it is rejected at load time (fail-closed) instead of being accepted as a dead rule. An IPv4 peer on a dual-stack IPv6 listener is normalized from its `::ffff:a.b.c.d` form so IPv4 patterns match it. rsync-like semantics: a matching `hosts deny` rejects; if any `hosts allow` entries exist, a peer matching none of them is rejected; deny takes precedence over allow. The daemon enforces the global list first, then the selected module's list, **before authentication** in `server_module_gate`, with an audit log line naming the peer, the module and the outcome. The numeric peer address is obtained with `getpeername`+`inet_ntop` (`utils_fd_peer_ip`, handling both address families); when it cannot be obtained a module with any ACL fails closed (refused), while an ACL-free module continues and logs at debug. A malformed pattern (e.g. an out-of-range CIDR prefix) is a parse error at load time. +- **Connection cap and auth throttle:** the global `max connections` key (default 100) is plumbed into the listener (`transport_tcp.c`), which rejects a connection once the accept-loop parent's active-child count reaches it; the IPv4/IPv6 peer is logged for every accepted connection. The optional per-module `max connections` key is parsed and validated but **not enforced** (connections are counted in the parent before the client's module is known); the daemon logs a startup warning for any module that sets it. On a failed authentication the per-connection child sleeps the global `auth failure delay` (default 500 ms, 0 disables, capped at 5000) via `nanosleep` before the connection closes, rate-limiting online guessing without delaying a success. - **Module selection & confinement:** the client requests a module with an rsync-style `host::module[/path]` destination. The module name crosses the wire as a trailing string on the config frame (bumping `PROTOCOL_VERSION` 2.14.0 → 2.15.0; the bump is required because the config-frame layout changed and the strict same-version handshake is what prevents a peer from desynchronizing on the new trailing field). The daemon looks the module up in ITS OWN config and uses the module's `path` as the authorized root through the exact same `configure_authorization` confinement the standalone server applies to `--destination-root` (`file_open_secure_parent`, `has_path_traversal`, `path_is_within`); the client never supplies the root, every client-chosen-ownership/super-user request is refused unless the module declares `client owner = yes` (the daemon's per-module opt-in, see below), and the operator `--no-super` veto forces super-user activities off for every daemon connection. The client's `/path` part is relative inside the module and is rejected if absolute or if it contains `..`. Unknown modules are refused before any data moves (the run fails cleanly at the config handshake). An absolute destination and a module request against a non-daemon server are also refused. - **`client owner` (client-chosen-ownership opt-in):** by default a daemon module refuses every request that would let the client pick an owner or ask for super-user activities — `--numeric-ids`, `--chown`, `--usermap`/`--groupmap`, `--fake-super`, `--copy-as`, and an explicit `--super` — at the config handshake (before `STATUS_OK`), because a daemon has no per-module opt-in for client-chosen ownership and any anonymous client could otherwise force arbitrary owner ids inside the module root. `client owner = yes` opts a single module in, allowing those requests within that module's root (the standalone listener and the SSH `--stdio` server always honor them for their single operator-authorized root). Without the opt-in the daemon also forces super-user **device** activity off for that connection — char/block device-node creation (`--devices`) and `--write-devices` — even under the default `AUTO` mode, so a non-opted module can never be made to `mknod` or write a raw device; those entries are skipped (not refused) so an ordinary `-a` push still succeeds without device nodes. The opt-in does **not** lift the privilege requirement: `--copy-as` still needs a root receiver, and the operator `--no-super` veto still forces super-user activities off for every connection. The daemon logs a prominent startup warning for each `client owner = yes` module so the operator's deliberate choice is visible. - **`read only` safe default:** every network transfer FastSync currently supports is a push that writes under the module root, so a `read only` module refuses the connection (clear server log "module is read only"; the client exits non-zero, nothing is transferred). A future pull/list operation can be opened up when it exists; the knob is already stored. diff --git a/src/shared/daemon_conf.c b/src/shared/daemon_conf.c index 10c0f63..01a759a 100644 --- a/src/shared/daemon_conf.c +++ b/src/shared/daemon_conf.c @@ -91,25 +91,39 @@ static bool parse_cidr(const char* cidr, int* prefix_out, uint8_t* bytes, int* f return false; } -/* A host pattern is valid when it is non-empty and, when it contains a '/', its - * address/prefix halves parse as a CIDR. Literals, `*` and globs are accepted - * as-is (a glob only ever matches a peer of the same shape). */ +/* A host pattern is valid when it is `*`, a valid IPv4/IPv6 literal, or a valid + * CIDR. Peer addresses reaching the matcher are always numeric, so hostname + * globs are rejected at parse time: accepting one would create a deny rule that + * silently never matches (fail-open). */ static bool host_pattern_valid(const char* pattern) { if (!pattern || *pattern == '\0') return false; - if (!strchr(pattern, '/')) + if (strcmp(pattern, "*") == 0) return true; - uint8_t bytes[16]; - int prefix; - int family; - return parse_cidr(pattern, &prefix, bytes, &family); + if (strchr(pattern, '/')) { + uint8_t bytes[16]; + int prefix; + int family; + return parse_cidr(pattern, &prefix, bytes, &family); + } + struct in_addr v4; + struct in6_addr v6; + return inet_pton(AF_INET, pattern, &v4) == 1 || inet_pton(AF_INET6, pattern, &v6) == 1; } /* Append every comma- and/or whitespace-separated host pattern in `value` to - * the heap-owned list. Returns false (err filled) on an invalid pattern or an - * allocation failure. */ + * the heap-owned list (or replace the list when `replace` is set, which --dparam + * uses so an override can narrow access rather than only widen it). Returns + * false (err filled) on an invalid pattern or an allocation failure. */ static bool store_host_list(char*** list, int* count, const char* value, const char* key, - const char* module_name, char* err, size_t err_size) { + const char* module_name, bool replace, char* err, size_t err_size) { + if (replace) { + for (int i = 0; i < *count; i++) + free((*list)[i]); + free(*list); + *list = NULL; + *count = 0; + } char* copy = str_dup(value); if (!copy) { if (module_name) @@ -278,8 +292,8 @@ static bool store_port(int* slot, const char* value, char* err, size_t err_size) /* Apply a global scalar key/value. Keys are case-insensitive. Returns false * (err filled) on an unknown key or an invalid value. */ -static bool apply_global_key(DaemonConf* conf, char* key, const char* value, char* err, - size_t err_size) { +static bool apply_global_key(DaemonConf* conf, char* key, const char* value, bool replace_hosts, + char* err, size_t err_size) { if (key_equals(key, "port")) return store_port(&conf->global.port, value, err, err_size); if (key_equals(key, "motd file")) { @@ -302,10 +316,10 @@ static bool apply_global_key(DaemonConf* conf, char* key, const char* value, cha return store_auth_failure_delay(&conf->global.auth_failure_delay_ms, value, err, err_size); if (key_equals(key, "hosts allow")) return store_host_list(&conf->global.hosts_allow, &conf->global.hosts_allow_count, value, - "hosts allow", NULL, err, err_size); + "hosts allow", NULL, replace_hosts, err, err_size); if (key_equals(key, "hosts deny")) return store_host_list(&conf->global.hosts_deny, &conf->global.hosts_deny_count, value, - "hosts deny", NULL, err, err_size); + "hosts deny", NULL, replace_hosts, err, err_size); set_error(err, err_size, "unknown global key '%s'", key); return false; } @@ -389,10 +403,10 @@ static bool apply_module_key(DaemonModule* module, char* key, char* value, char* return store_max_connections(&module->max_connections, value, module->name, err, err_size); if (key_equals(key, "hosts allow")) return store_host_list(&module->hosts_allow, &module->hosts_allow_count, value, "hosts allow", - module->name, err, err_size); + false, module->name, err, err_size); if (key_equals(key, "hosts deny")) return store_host_list(&module->hosts_deny, &module->hosts_deny_count, value, "hosts deny", - module->name, err, err_size); + false, module->name, err, err_size); set_error(err, err_size, "unknown key '%s' in module '%s'", key, module->name); return false; } @@ -573,7 +587,7 @@ DaemonConf* daemon_conf_load(const char* path, char* err, size_t err_size) { break; } } else { - if (!apply_global_key(conf, key, value, err, err_size)) { + if (!apply_global_key(conf, key, value, false, err, err_size)) { ok = false; break; } @@ -628,7 +642,7 @@ int daemon_conf_apply_dparam(DaemonConf* conf, const char* assignment, char* err set_error(err, err_size, "--dparam '%s' has an empty value", assignment); return -1; } - bool ok = apply_global_key(conf, key, value, err, err_size); + bool ok = apply_global_key(conf, key, value, true, err, err_size); free(copy); return ok ? 0 : -1; } diff --git a/src/shared/daemon_conf.h b/src/shared/daemon_conf.h index d82276b..3b790a7 100644 --- a/src/shared/daemon_conf.h +++ b/src/shared/daemon_conf.h @@ -94,7 +94,9 @@ typedef struct DaemonConf { #define DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS 500 /* Largest accepted `auth failure delay`, so a typo cannot pin a connection * child in nanosleep for an absurd time. */ -#define DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS 60000 +/* Bounded well below the socket I/O timeout so a failed-auth child cannot hold + * a connection slot for long enough to amplify connection-cap exhaustion. */ +#define DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS 5000 /* Longest accepted config line (excluding the trailing newline). Longer lines * are rejected rather than buffered unboundedly. */ #define DAEMON_CONF_MAX_LINE 4096 diff --git a/src/shared/utils.c b/src/shared/utils.c index e00ec25..c34d4a6 100644 --- a/src/shared/utils.c +++ b/src/shared/utils.c @@ -603,12 +603,21 @@ bool utils_fd_peer_ip(int fd, char* buf, size_t len) { return false; const void* src = NULL; int family = peer.ss_family; - if (family == AF_INET) + if (family == AF_INET) { src = &((const struct sockaddr_in*)&peer)->sin_addr; - else if (family == AF_INET6) - src = &((const struct sockaddr_in6*)&peer)->sin6_addr; - else + } else if (family == AF_INET6) { + const struct sockaddr_in6* peer6 = (const struct sockaddr_in6*)&peer; + /* A dual-stack IPv6 listener reports IPv4 peers as ::ffff:a.b.c.d. Emit + * the IPv4 form so IPv4 ACL patterns (and logs) see the real address. */ + if (IN6_IS_ADDR_V4MAPPED(&peer6->sin6_addr)) { + struct in_addr v4; + memcpy(&v4, &peer6->sin6_addr.s6_addr[12], sizeof(v4)); + return inet_ntop(AF_INET, &v4, buf, (socklen_t)len) != NULL; + } + src = &peer6->sin6_addr; + } else { return false; + } return inet_ntop(family, src, buf, (socklen_t)len) != NULL; } diff --git a/tests/test_daemon_conf.c b/tests/test_daemon_conf.c index 7e7577a..e0020c9 100644 --- a/tests/test_daemon_conf.c +++ b/tests/test_daemon_conf.c @@ -321,6 +321,10 @@ static void test_daemon_conf_dparam_override() { EXPECT_EQ_INT( daemon_conf_apply_dparam(conf, "hosts allow=127.0.0.1,10.0.0.0/8", err, sizeof(err)), 0); EXPECT_EQ_INT(conf->global.hosts_allow_count, 2); + /* A later --dparam replaces the list (an override must be able to narrow). */ + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "hosts allow=127.0.0.1", err, sizeof(err)), 0); + EXPECT_EQ_INT(conf->global.hosts_allow_count, 1); + EXPECT_EQ_STR(conf->global.hosts_allow[0], "127.0.0.1"); EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "port=notaport", err, sizeof(err)), -1); EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "bogus=1", err, sizeof(err)), -1); @@ -386,7 +390,7 @@ static void test_daemon_conf_limits_and_hosts_parse() { char err[256]; EXPECT_EQ_INT(write_conf("max connections = 25\n" "auth failure delay = 0\n" - "hosts allow = 10.0.0.0/8, *.example.com\n" + "hosts allow = 10.0.0.0/8, 192.168.1.0/24\n" "hosts deny = 192.168.0.1 2001:db8::/32\n" "\n" "[m]\n" @@ -403,7 +407,7 @@ static void test_daemon_conf_limits_and_hosts_parse() { EXPECT_EQ_INT(conf->global.auth_failure_delay_ms, 0); EXPECT_EQ_INT(conf->global.hosts_allow_count, 2); EXPECT_EQ_STR(conf->global.hosts_allow[0], "10.0.0.0/8"); - EXPECT_EQ_STR(conf->global.hosts_allow[1], "*.example.com"); + EXPECT_EQ_STR(conf->global.hosts_allow[1], "192.168.1.0/24"); EXPECT_EQ_INT(conf->global.hosts_deny_count, 2); EXPECT_EQ_STR(conf->global.hosts_deny[0], "192.168.0.1"); EXPECT_EQ_STR(conf->global.hosts_deny[1], "2001:db8::/32"); @@ -415,9 +419,11 @@ static void test_daemon_conf_limits_and_hosts_parse() { daemon_conf_free(conf); const char* bad_values[] = { - "max connections = 0\n", "max connections = -1\n", "max connections = abc\n", - "auth failure delay = -1\n", "auth failure delay = 70000\n", "auth failure delay = soon\n", - "hosts allow = 10.0.0.0/99\n", "hosts deny = 2001:db8::/129\n", + "max connections = 0\n", "max connections = -1\n", + "max connections = abc\n", "auth failure delay = -1\n", + "auth failure delay = 70000\n", "auth failure delay = soon\n", + "hosts allow = 10.0.0.0/99\n", "hosts deny = 2001:db8::/129\n", + "hosts allow = *.example.com\n", "hosts deny = not-an-ip\n", }; for (size_t i = 0; i < sizeof(bad_values) / sizeof(bad_values[0]); i++) { EXPECT_EQ_INT(write_conf(bad_values[i], &path), 0); From b16349b81e7ed0146d6ad3e5feed8038d2fcdcfe Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 03:29:01 +0200 Subject: [PATCH 018/155] fix(protocol): honor --timeout for protocol I/O; bound idle/session time --- README.md | 13 +++++- src/client/client_send.c | 2 + src/server/receiver.c | 80 +++++++++++++++++++++++++++++++ src/server/receiver.h | 20 ++++++++ src/server/server.c | 6 +++ src/shared/config.c | 6 ++- src/shared/config.h | 5 ++ src/shared/protocol.c | 15 +++++- src/shared/protocol.h | 11 +++++ tests/runner.c | 2 + tests/test_protocol.c | 30 ++++++++++++ tests/test_receiver_timeout.c | 88 +++++++++++++++++++++++++++++++++++ tests/test_receiver_timeout.h | 6 +++ 13 files changed, 279 insertions(+), 5 deletions(-) create mode 100644 tests/test_receiver_timeout.c create mode 100644 tests/test_receiver_timeout.h diff --git a/README.md b/README.md index 2d59b66..ee7ac3a 100644 --- a/README.md +++ b/README.md @@ -132,7 +132,7 @@ partial, alternate, and planned behavior. | `--existing` | Skip files not already present at the destination; update existing files normally. | | `--bwlimit ` | Bandwidth limit in kilobytes per second | | `--chunk-size ` | Chunk size in bytes (default: 10485760) | -| `--timeout ` | I/O timeout in seconds (default: 30) | +| `--timeout ` | I/O timeout in seconds. Applied to both the socket (`SO_RCVTIMEO`/`SO_SNDTIMEO`, built-in default 30 s) and the per-message protocol poll deadline (built-in default 60 s). `0` (the default/unset sentinel) keeps both built-ins; a positive value overrides both. | | `--contimeout ` | Connection timeout in seconds (default: 10) | | `--backup` | Backup existing destination files before overwriting | | `--backup-dir ` | Target directory for backups (requires `--backup`) | @@ -151,6 +151,15 @@ partial, alternate, and planned behavior. | `--ca ` | TLS CA certificate file for verification (PEM) | | `--client-cn ` | TLS client certificate common name; mandatory with `--tls` (a TLS connection always verifies the client CN) | +**Per-message vs. connection timeouts.** `--timeout` bounds each individual protocol +send/receive (the `poll()` deadline), so a peer that stops mid-frame is dropped. It +does not, by itself, stop a peer that keeps sending well-formed frames forever. The +receiver therefore also enforces two wall-clock (`CLOCK_MONOTONIC`) bounds on a +connection: a **1 hour** idle limit (only `STATUS_KEEPALIVE`/`STATUS_ABORT` frames +seen for that long counts as no forward progress) and a **24 hour** overall session +cap. Both are deliberately generous so a legitimate long-running transfer is never +aborted; they exist to defeat keepalive slowloris squatting on a connection slot. + ### Server | Argument | Description | @@ -384,7 +393,7 @@ features without changing the meaning of ordinary compatibility options. | `--bwlimit ` | Apply token-bucket bandwidth limiting. | | `--progress` | Show transfer progress and throughput. | | `--stats` | Print transfer statistics. | -| `--timeout ` | Set I/O timeout. | +| `--timeout ` | Set the socket **and** per-message protocol I/O timeout. `0` keeps the built-in 30 s socket / 60 s protocol defaults; a positive value overrides both. | | `--contimeout ` | Set connection timeout. | Short-option conflicts with rsync have been resolved for the CLI namespace diff --git a/src/client/client_send.c b/src/client/client_send.c index 16ed906..a0bb9dc 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -1435,6 +1435,7 @@ static int send_chunks_multithreaded(void* pipeline_context) { } ProtocolSession session; protocol_session_init(&session, client->file_descriptor, client->file_descriptor); + protocol_session_set_io_timeout(&session, context->config->timeout); protocol_session_set_ssl(&session, (SSL*)client->ssl); protocol_session_bind(&session); if (!config_send(client->file_descriptor, context->config)) { @@ -1896,6 +1897,7 @@ int send_files(Config* config) { } ProtocolSession session; protocol_session_init(&session, client->file_descriptor, client->file_descriptor); + protocol_session_set_io_timeout(&session, config->timeout); protocol_session_set_ssl(&session, (SSL*)client->ssl); protocol_session_bind(&session); int ret = 1; diff --git a/src/server/receiver.c b/src/server/receiver.c index 4b84831..04e9f3d 100644 --- a/src/server/receiver.c +++ b/src/server/receiver.c @@ -12,6 +12,7 @@ #include "utils.h" #include #include +#include bool receiver_outcomes_append(ReceiverOutcomes* outcomes, unsigned char code) { if (!outcomes) @@ -153,6 +154,74 @@ static bool receiver_process_batch(Config* config, int file_descriptor) { return true; } +/* ---- Anti-slowloris connection bounds ---- + * A legitimate transfer either streams data frames continuously or, when it + * must pause, sends STATUS_KEEPALIVE so the peer sees the connection is alive. + * An attacker can therefore squat on a connection slot indefinitely by sending + * only keepalives under the per-message timeout. Two CLOCK_MONOTONIC bounds + * defeat that without ever punishing a real transfer: + * + * MAX_SESSION_IDLE_SEC (1 h): the longest a stream may make no forward + * progress. Data/status frames count as progress and refresh the timer; + * keepalives do not. One hour is far longer than any real pause between + * data frames, yet small enough to reap a slowloris well before the 24 h + * session cap. + * + * MAX_SESSION_WALL_SEC (24 h): an absolute ceiling on one connection's + * lifetime as defense-in-depth against a trickle of progress frames that + * resets the idle timer just below its limit. Larger than any plausible + * single transfer while still bounding resource occupancy. + * + * Both are wall-clock deltas, so the per-message poll timeout (60 s by default, + * or --timeout) can never fool them, and both the single-threaded and the -m + * receiver paths (receiver_process_pending) share the same logic. */ +#define MAX_SESSION_IDLE_SEC 3600u +#define MAX_SESSION_WALL_SEC 86400u + +static unsigned int g_max_session_idle_sec = MAX_SESSION_IDLE_SEC; +static unsigned int g_max_session_wall_sec = MAX_SESSION_WALL_SEC; + +void receiver_set_time_limits(unsigned int idle_sec, unsigned int wall_sec) { + g_max_session_idle_sec = idle_sec; + g_max_session_wall_sec = wall_sec; +} + +void receiver_reset_time_limits(void) { + g_max_session_idle_sec = MAX_SESSION_IDLE_SEC; + g_max_session_wall_sec = MAX_SESSION_WALL_SEC; +} + +bool receiver_time_limit_exceeded(const struct timespec* session_start, + const struct timespec* last_progress, + const struct timespec* now) { + if (!session_start || !last_progress || !now) + return false; + if (now->tv_sec - session_start->tv_sec >= (time_t)g_max_session_wall_sec) + return true; + if (now->tv_sec - last_progress->tv_sec >= (time_t)g_max_session_idle_sec) + return true; + return false; +} + +/* Refresh the progress timestamp for a forward-moving frame and enforce the + * bounds above. Returns false (after best-effort STATUS_ERROR) when the + * connection must be dropped. */ +static bool receiver_note_status(const struct timespec* session_start, + struct timespec* last_progress, Status status, + int file_descriptor) { + struct timespec now; + clock_gettime(CLOCK_MONOTONIC, &now); + if (status != STATUS_KEEPALIVE && status != STATUS_ABORT) + *last_progress = now; + if (!receiver_time_limit_exceeded(session_start, last_progress, &now)) + return true; + log_message(LOG_LEVEL_ERROR, + "Receive session exceeded its time bound (idle %us / total %us); aborting connection", + g_max_session_idle_sec, g_max_session_wall_sec); + send_status(file_descriptor, STATUS_ERROR); + return false; +} + int receiver_process(Config* config, int file_descriptor, const ReceiverSink* sink) { return receiver_process_pending(config, file_descriptor, sink, NULL); } @@ -172,6 +241,15 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver Status status; if (!receive_status(file_descriptor, &status)) return -1; + /* Wall-clock (=CLOCK_MONOTONIC) anti-slowloris bookkeeping. session_start is + * fixed for the whole connection; last_progress is refreshed by every frame + * that is not a keepalive/abort. */ + struct timespec session_start; + struct timespec last_progress; + clock_gettime(CLOCK_MONOTONIC, &session_start); + last_progress = session_start; + if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor)) + return -1; bool early_delete = config_delete_timing_early(config); /* Parked keep-set for the late/commit timing. Every exit path below frees it exactly once; the only exception is the successful FINISHED handoff, which @@ -271,6 +349,8 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver next_status: if (!receive_status(file_descriptor, &status)) goto receive_error; + if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor)) + goto fail; } if (status != STATUS_FINISHED) { log_message(LOG_LEVEL_ERROR, "Did not receive FINISHED Status"); diff --git a/src/server/receiver.h b/src/server/receiver.h index 1b07e13..9f3efa3 100644 --- a/src/server/receiver.h +++ b/src/server/receiver.h @@ -4,6 +4,8 @@ #include "config.h" #include "file.h" #include "file_receive.h" +#include +#include typedef bool (*ReceiverFileSink)(File* file, void* context); @@ -45,4 +47,22 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver DeleteManifest** pending_manifest); int receiver_receive_files(Config* config, int file_descriptor); +/* ---- Connection time bounds (anti-slowloris) ---- + * receiver_process_pending() aborts a connection that makes no forward progress + * (only STATUS_KEEPALIVE/STATUS_ABORT frames) beyond a wall-clock idle limit, + * and enforces a hard cap on the whole session. Both are CLOCK_MONOTONIC + * deltas, independent of the per-message poll deadline, so a 60 s (or + * --timeout) receive window can never reset them. Defaults are deliberately + * generous (see MAX_SESSION_IDLE_SEC / MAX_SESSION_WALL_SEC in receiver.c). */ + +/* Test seam: override the idle/session wall-clock limits (0 = abort on the + * next status). Always restore with receiver_reset_time_limits(). */ +void receiver_set_time_limits(unsigned int idle_sec, unsigned int wall_sec); +void receiver_reset_time_limits(void); +/* Pure predicate over explicit monotonic timestamps, exposed so the bound is + * unit-testable without sleeping. True when either the idle or the overall + * session limit has elapsed. */ +bool receiver_time_limit_exceeded(const struct timespec* session_start, + const struct timespec* last_progress, const struct timespec* now); + #endif diff --git a/src/server/server.c b/src/server/server.c index 92b65cf..13b854c 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -609,6 +609,11 @@ void handler(int file_descriptor) { if (gate_ctx.super_mode_override != -1) config->super_mode = (SuperMode)gate_ctx.super_mode_override; protocol_set_8_bit_output(config->eight_bit_output); + /* Honor the negotiated --timeout for every protocol frame from here on (the + * config handshake itself used the built-in 60 s window). A positive value + * also tightens the socket SO_RCVTIMEO/SO_SNDTIMEO already applied by the + * transport; 0 leaves both built-in defaults in place. */ + protocol_session_set_io_timeout(&session, config->timeout); if (!authorized_root) { log_message(LOG_LEVEL_ERROR, "No server-side destination root configured"); goto done; @@ -743,6 +748,7 @@ void handler(int file_descriptor) { goto done; } protocol_session_set_max_alloc(&context->session, config->max_alloc); + protocol_session_set_io_timeout(&context->session, config->timeout); atomic_store(&context->session.total_allocated_bytes, atomic_load(&session.total_allocated_bytes)); pipeline_context_receiver_set_queue_byte_limit(context, RECEIVER_QUEUE_MAX_BYTES); diff --git a/src/shared/config.c b/src/shared/config.c index 4816e06..6730639 100644 --- a/src/shared/config.c +++ b/src/shared/config.c @@ -67,7 +67,11 @@ static void config_set_defaults(Config* config) { config->tls_ca = NULL; config->server_host = str_dup("127.0.0.1"); config->server_port = 8080; - config->timeout = 30; + /* 0 means "--timeout not given": the transport keeps its own built-in 30 s + * socket timeout (tcp_set_timeouts ignores non-positive values) and the + * protocol layer keeps its built-in 60 s per-message deadline. A positive + * value overrides BOTH (see protocol_session_set_io_timeout). */ + config->timeout = 0; config->contimeout = 10; config->quiet = false; config->backup = false; diff --git a/src/shared/config.h b/src/shared/config.h index d582be2..77febe8 100644 --- a/src/shared/config.h +++ b/src/shared/config.h @@ -157,7 +157,12 @@ typedef struct Config { char* tls_cert; char* tls_key; char* tls_ca; + /* --timeout: per-message I/O deadline in seconds. 0 (the default/unset + * sentinel) leaves the transport's built-in 30 s socket timeout and the + * protocol's built-in 60 s per-message deadline in place; a positive value + * overrides both. See protocol_session_set_io_timeout. */ int timeout; + /* --contimeout: connect()/accept timeout, transport layer only. */ int contimeout; bool quiet; bool backup; diff --git a/src/shared/protocol.c b/src/shared/protocol.c index 3f13667..7604379 100644 --- a/src/shared/protocol.c +++ b/src/shared/protocol.c @@ -76,10 +76,17 @@ void protocol_session_init(ProtocolSession* session, int read_fd, int write_fd) session->read_fd = read_fd; session->write_fd = write_fd; session->max_alloc = DEFAULT_MAX_ALLOC; + session->io_timeout_sec = RECEIVE_TIMEOUT_SEC; atomic_init(&session->total_allocated_bytes, 0); protocol_session_set_bwlimit(session, global_bwlimit()); } +void protocol_session_set_io_timeout(ProtocolSession* session, int sec) { + if (!session) + return; + session->io_timeout_sec = sec; +} + void protocol_session_set_max_alloc(ProtocolSession* session, unsigned long long max_alloc) { if (!session) session = bound_session ? bound_session : &legacy_io_session; @@ -257,10 +264,11 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat log_debug_message(LOG_DEBUG_IO, " Sending n Data: %zu", data_size); if (!session) return false; + int timeout_sec = session->io_timeout_sec > 0 ? session->io_timeout_sec : SEND_TIMEOUT_SEC; int fd = session->write_fd; struct timespec deadline; clock_gettime(CLOCK_MONOTONIC, &deadline); - deadline.tv_sec += SEND_TIMEOUT_SEC; + deadline.tv_sec += timeout_sec; short wait_events = POLLOUT; ssize_t total_bytes_send = 0; while ((size_t)total_bytes_send < data_size) { @@ -306,7 +314,10 @@ bool protocol_receive_n_data_timed(ProtocolSession* session, void* data, size_t int timeout_sec); bool protocol_receive_n_data(ProtocolSession* session, void* data, size_t data_size) { - return protocol_receive_n_data_timed(session, data, data_size, RECEIVE_TIMEOUT_SEC); + /* Honor the session's configured deadline; protocol_receive_n_data_timed + * re-applies the built-in 60 s default when the value is <= 0. */ + int timeout_sec = session ? session->io_timeout_sec : 0; + return protocol_receive_n_data_timed(session, data, data_size, timeout_sec); } bool protocol_receive_n_data_timed(ProtocolSession* session, void* data, size_t data_size, diff --git a/src/shared/protocol.h b/src/shared/protocol.h index e23a68f..5015971 100644 --- a/src/shared/protocol.h +++ b/src/shared/protocol.h @@ -52,6 +52,12 @@ typedef struct ProtocolSession { atomic_ullong total_allocated_bytes; bool eight_bit_output; unsigned long long max_alloc; + /* Per-session deadline (seconds) applied to every protocol send/receive by + * protocol_send_n_data / protocol_receive_n_data. Defaults to the built-in + * 60 s window; a value <= 0 falls back to that default. Set from the + * negotiated Config->timeout so --timeout is honored by the poll()-driven + * protocol I/O, not just the socket SO_RCVTIMEO/SO_SNDTIMEO. */ + int io_timeout_sec; } ProtocolSession; typedef int Status; @@ -141,6 +147,11 @@ void protocol_session_unbind(void); void protocol_session_set_ssl(ProtocolSession* session, SSL* ssl); void protocol_session_set_bwlimit(ProtocolSession* session, unsigned long long bytes_per_sec); void protocol_session_set_max_alloc(ProtocolSession* session, unsigned long long max_alloc); +/* Override the per-message send/receive deadline for this session. + * `sec` <= 0 restores the built-in 60 s default (used for --timeout=0/unset). + * An explicit long deadline (e.g. the delete-ack wait) is applied per-call by + * protocol_receive_status_timed and is unaffected by this setter. */ +void protocol_session_set_io_timeout(ProtocolSession* session, int sec); void* protocol_alloc(size_t size); void* protocol_realloc(void* ptr, size_t size); void protocol_session_set_8_bit_output(ProtocolSession* session, bool enabled); diff --git a/tests/runner.c b/tests/runner.c index 3837a6f..ec58de9 100644 --- a/tests/runner.c +++ b/tests/runner.c @@ -23,6 +23,7 @@ #include "test_property.h" #include "test_protocol.h" #include "test_queue.h" +#include "test_receiver_timeout.h" #include "test_robustness.h" #include "test_scanner.h" #include "test_server.h" @@ -61,6 +62,7 @@ int main() { RUN_TEST(test_delta); RUN_TEST(test_data); RUN_TEST(test_protocol); + RUN_TEST(test_receiver_timeout); RUN_TEST(test_metadata); RUN_TEST(test_glob); RUN_TEST(test_iconv); diff --git a/tests/test_protocol.c b/tests/test_protocol.c index 79fb8a0..273861e 100644 --- a/tests/test_protocol.c +++ b/tests/test_protocol.c @@ -412,6 +412,35 @@ static void test_protocol_accounting_release_does_not_underflow() { protocol_session_unbind(); } +static void test_protocol_session_io_timeout() { + /* Default is the built-in 60 s window; the setter stores exactly what it is + * given (<= 0 means "fall back to the default") so callers can propagate + * --timeout without special-casing 0. */ + ProtocolSession session; + protocol_session_init(&session, -1, -1); + EXPECT_EQ_INT(session.io_timeout_sec, 60); + + protocol_session_set_io_timeout(&session, 120); + EXPECT_EQ_INT(session.io_timeout_sec, 120); + protocol_session_set_io_timeout(&session, 0); + EXPECT_EQ_INT(session.io_timeout_sec, 0); + /* A NULL session is a no-op, not a crash. */ + protocol_session_set_io_timeout(NULL, 5); + + /* A short per-session deadline must actually bound a non-responsive read: + * with no writer the poll waits for the configured 1 s and then fails, + * rather than the built-in 60 s. */ + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + ProtocolSession timed; + protocol_session_init(&timed, p[0], p[1]); + protocol_session_set_io_timeout(&timed, 1); + char buf[4]; + EXPECT_FALSE(protocol_receive_n_data(&timed, buf, sizeof(buf))); + close(p[0]); + close(p[1]); +} + static void test_send_receive_status_timed() { int p[2]; EXPECT_EQ_INT(pipe(p), 0); @@ -440,6 +469,7 @@ void test_protocol() { test_send_receive_data(); test_send_receive_int(); test_send_receive_status(); + test_protocol_session_io_timeout(); test_send_receive_status_timed(); test_receive_n_data_truncated(); test_receive_str_truncated(); diff --git a/tests/test_receiver_timeout.c b/tests/test_receiver_timeout.c new file mode 100644 index 0000000..1bb65c8 --- /dev/null +++ b/tests/test_receiver_timeout.c @@ -0,0 +1,88 @@ +#include "test_receiver_timeout.h" + +#include "protocol.h" +#include "receiver.h" +#include "test_utils.h" +#include +#include +#include + +static bool sink_discard(File* file, void* context) { + (void)context; + file_destroy(file); + return true; +} + +/* The idle/session bound is a pure function of three monotonic timestamps, so + * it can be exercised deterministically without sleeping an hour. A tiny + * overridden limit covers the same arithmetic the loop uses. */ +static void test_receiver_time_limit_predicate() { + receiver_set_time_limits(10, 100); + struct timespec start = {.tv_sec = 1000, .tv_nsec = 0}; + struct timespec fresh = {.tv_sec = 1000, .tv_nsec = 0}; + struct timespec just_idle = {.tv_sec = 1009, .tv_nsec = 0}; /* 9 s no progress */ + struct timespec idle = {.tv_sec = 1010, .tv_nsec = 0}; /* 10 s no progress */ + struct timespec just_wall = {.tv_sec = 1099, .tv_nsec = 0}; + struct timespec wall = {.tv_sec = 1100, .tv_nsec = 0}; /* 100 s session */ + struct timespec wp_just = {.tv_sec = 1098, .tv_nsec = 0}; /* idle 1 s */ + struct timespec wp_wall = {.tv_sec = 1099, .tv_nsec = 0}; /* idle 1 s */ + + EXPECT_FALSE(receiver_time_limit_exceeded(&start, &fresh, &fresh)); + EXPECT_FALSE(receiver_time_limit_exceeded(&start, &fresh, &just_idle)); + EXPECT_TRUE(receiver_time_limit_exceeded(&start, &fresh, &idle)); + EXPECT_FALSE(receiver_time_limit_exceeded(&start, &wp_just, &just_wall)); + EXPECT_TRUE(receiver_time_limit_exceeded(&start, &wp_wall, &wall)); + + /* Reset restores the generous production defaults (1 h idle / 24 h total). */ + receiver_reset_time_limits(); + struct timespec under_hour = {.tv_sec = 1000 + 3599, .tv_nsec = 0}; + EXPECT_FALSE(receiver_time_limit_exceeded(&start, &start, &under_hour)); + receiver_reset_time_limits(); +} + +/* Drive the actual receive loop with a test-only idle limit of 0 so the very + * first keepalive is rejected: this exercises the loop's abort path (log + + * STATUS_ERROR + return -1) with no timing dependence. */ +static void test_receiver_aborts_idle_keepalive() { + receiver_set_time_limits(0, 3600); + int sv[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, sv), 0); + Config* config = config_create(); + EXPECT_NOT_NULL(config); + ReceiverSink sink = {.store_file = sink_discard, + .context = NULL, + .send_error = true, + .send_success = false, + .send_success_frame = NULL}; + + /* Bind an explicit session so the fd-based receive helpers use the + * socketpair rather than any transport left over from an earlier test. */ + ProtocolSession session; + protocol_session_init(&session, sv[1], sv[1]); + protocol_session_bind(&session); + + Status keepalive = STATUS_KEEPALIVE; + EXPECT_EQ_INT((int)write(sv[0], &keepalive, sizeof(keepalive)), (int)sizeof(keepalive)); + int result = receiver_process_pending(config, sv[1], &sink, NULL); + EXPECT_EQ_INT(result, -1); + + Status reply = STATUS_OK; + EXPECT_EQ_INT((int)read(sv[0], &reply, sizeof(reply)), (int)sizeof(reply)); + EXPECT_EQ_INT((int)reply, (int)STATUS_ERROR); + + protocol_session_unbind(); + config_delete(config); + close(sv[0]); + close(sv[1]); + receiver_reset_time_limits(); +} + +void test_receiver_timeout(void) { + /* EXPECT_* returns from the current function on failure, so reset the + * process-global limits around the subtests (and again after) to guarantee a + * failed assertion cannot leave the receiver aborted for later tests. */ + receiver_reset_time_limits(); + test_receiver_time_limit_predicate(); + test_receiver_aborts_idle_keepalive(); + receiver_reset_time_limits(); +} diff --git a/tests/test_receiver_timeout.h b/tests/test_receiver_timeout.h new file mode 100644 index 0000000..c320aec --- /dev/null +++ b/tests/test_receiver_timeout.h @@ -0,0 +1,6 @@ +#ifndef TEST_RECEIVER_TIMEOUT_H +#define TEST_RECEIVER_TIMEOUT_H + +void test_receiver_timeout(void); + +#endif From ffa1d24625ea97b651d6ec6e8e36d372afdb4c8b Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 03:46:20 +0200 Subject: [PATCH 019/155] fix(receiver): harden idle-progress definition, single error frame, sendfile timeout --- README.md | 16 +++++++++------ src/server/receiver.c | 37 ++++++++++++++++++++++++++--------- src/server/server.c | 9 +++++---- src/shared/file_send.c | 2 +- src/shared/protocol.c | 6 ++++++ src/shared/protocol.h | 4 ++++ tests/test_receiver_timeout.c | 21 ++++++++++++++------ 7 files changed, 69 insertions(+), 26 deletions(-) diff --git a/README.md b/README.md index ee7ac3a..34059c2 100644 --- a/README.md +++ b/README.md @@ -132,7 +132,7 @@ partial, alternate, and planned behavior. | `--existing` | Skip files not already present at the destination; update existing files normally. | | `--bwlimit ` | Bandwidth limit in kilobytes per second | | `--chunk-size ` | Chunk size in bytes (default: 10485760) | -| `--timeout ` | I/O timeout in seconds. Applied to both the socket (`SO_RCVTIMEO`/`SO_SNDTIMEO`, built-in default 30 s) and the per-message protocol poll deadline (built-in default 60 s). `0` (the default/unset sentinel) keeps both built-ins; a positive value overrides both. | +| `--timeout ` | Positive I/O timeout in seconds, applied to both the socket (`SO_RCVTIMEO`/`SO_SNDTIMEO`, built-in default 30 s) and the per-message protocol poll deadline (built-in default 60 s). Omit the option to keep both built-ins; `0` is rejected. The server side keeps the built-in 60 s protocol window (the value is not sent on the wire). | | `--contimeout ` | Connection timeout in seconds (default: 10) | | `--backup` | Backup existing destination files before overwriting | | `--backup-dir ` | Target directory for backups (requires `--backup`) | @@ -155,10 +155,14 @@ partial, alternate, and planned behavior. send/receive (the `poll()` deadline), so a peer that stops mid-frame is dropped. It does not, by itself, stop a peer that keeps sending well-formed frames forever. The receiver therefore also enforces two wall-clock (`CLOCK_MONOTONIC`) bounds on a -connection: a **1 hour** idle limit (only `STATUS_KEEPALIVE`/`STATUS_ABORT` frames -seen for that long counts as no forward progress) and a **24 hour** overall session -cap. Both are deliberately generous so a legitimate long-running transfer is never -aborted; they exist to defeat keepalive slowloris squatting on a connection slot. +connection: a **1 hour** idle limit and a **24 hour** overall session cap. Only +frames that move real work (not `STATUS_KEEPALIVE`/`STATUS_ABORT` and not an +empty `STATUS_CHECK_BATCH`/`STATUS_DIR_TIMES`) refresh the idle timestamp, so a +peer cannot hold a connection slot by emitting cheap empty frames; a peer that +fabricates minimal non-empty frames can still occupy a slot until the 24 hour +cap, since no bound can require actual payload without risking a legitimate +long operation. Both are deliberately generous so a legitimate long-running +transfer is never aborted. ### Server @@ -393,7 +397,7 @@ features without changing the meaning of ordinary compatibility options. | `--bwlimit ` | Apply token-bucket bandwidth limiting. | | `--progress` | Show transfer progress and throughput. | | `--stats` | Print transfer statistics. | -| `--timeout ` | Set the socket **and** per-message protocol I/O timeout. `0` keeps the built-in 30 s socket / 60 s protocol defaults; a positive value overrides both. | +| `--timeout ` | Set the socket **and** per-message protocol I/O timeout (positive seconds). Omit to keep the built-in 30 s socket / 60 s protocol defaults. | | `--contimeout ` | Set connection timeout. | Short-option conflicts with rsync have been resolved for the CLI namespace diff --git a/src/server/receiver.c b/src/server/receiver.c index 04e9f3d..a0a3071 100644 --- a/src/server/receiver.c +++ b/src/server/receiver.c @@ -203,22 +203,41 @@ bool receiver_time_limit_exceeded(const struct timespec* session_start, return false; } +/* A frame proves forward progress only when it cannot be fabricated for free. + * KEEPALIVE/ABORT are pure liveness, and CHECK_BATCH/DIR_TIMES may carry zero + * entries, so a peer must not be able to hold a connection slot forever by + * merely emitting empty frames. */ +static bool status_counts_as_progress(Status status) { + switch (status) { + case STATUS_KEEPALIVE: + case STATUS_ABORT: + case STATUS_CHECK_BATCH: + case STATUS_DIR_TIMES: + return false; + default: + return true; + } +} + /* Refresh the progress timestamp for a forward-moving frame and enforce the - * bounds above. Returns false (after best-effort STATUS_ERROR) when the - * connection must be dropped. */ + * bounds above. Returns false when the connection must be dropped; the + * terminal STATUS_ERROR is sent only when the sink owns error reporting (the + * -m sink sets send_error=false so the main thread emits exactly one). */ static bool receiver_note_status(const struct timespec* session_start, - struct timespec* last_progress, Status status, - int file_descriptor) { + struct timespec* last_progress, Status status, int file_descriptor, + const ReceiverSink* sink) { struct timespec now; - clock_gettime(CLOCK_MONOTONIC, &now); - if (status != STATUS_KEEPALIVE && status != STATUS_ABORT) + if (clock_gettime(CLOCK_MONOTONIC, &now) != 0) + now = *last_progress; + if (status_counts_as_progress(status)) *last_progress = now; if (!receiver_time_limit_exceeded(session_start, last_progress, &now)) return true; log_message(LOG_LEVEL_ERROR, "Receive session exceeded its time bound (idle %us / total %us); aborting connection", g_max_session_idle_sec, g_max_session_wall_sec); - send_status(file_descriptor, STATUS_ERROR); + if (!sink || sink->send_error) + send_status(file_descriptor, STATUS_ERROR); return false; } @@ -248,7 +267,7 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver struct timespec last_progress; clock_gettime(CLOCK_MONOTONIC, &session_start); last_progress = session_start; - if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor)) + if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor, sink)) return -1; bool early_delete = config_delete_timing_early(config); /* Parked keep-set for the late/commit timing. Every exit path below frees it @@ -349,7 +368,7 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver next_status: if (!receive_status(file_descriptor, &status)) goto receive_error; - if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor)) + if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor, sink)) goto fail; } if (status != STATUS_FINISHED) { diff --git a/src/server/server.c b/src/server/server.c index 13b854c..da3e65d 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -609,10 +609,11 @@ void handler(int file_descriptor) { if (gate_ctx.super_mode_override != -1) config->super_mode = (SuperMode)gate_ctx.super_mode_override; protocol_set_8_bit_output(config->eight_bit_output); - /* Honor the negotiated --timeout for every protocol frame from here on (the - * config handshake itself used the built-in 60 s window). A positive value - * also tightens the socket SO_RCVTIMEO/SO_SNDTIMEO already applied by the - * transport; 0 leaves both built-in defaults in place. */ + /* Server-side per-message protocol deadline for every frame from here on. + * `timeout` is not serialized, so this is the server's own config (the server + * has no --timeout CLI and defaults it to 0): the built-in 60 s window stays + * in effect. A client's --timeout tightens only that client's own protocol + * I/O and the server's socket read/write timeout is the transport default. */ protocol_session_set_io_timeout(&session, config->timeout); if (!authorized_root) { log_message(LOG_LEVEL_ERROR, "No server-side destination root configured"); diff --git a/src/shared/file_send.c b/src/shared/file_send.c index e7bcffb..f15b77a 100644 --- a/src/shared/file_send.c +++ b/src/shared/file_send.c @@ -145,7 +145,7 @@ bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_meta off_t offset = 0; struct timespec deadline; clock_gettime(CLOCK_MONOTONIC, &deadline); - deadline.tv_sec += 60; + deadline.tv_sec += protocol_get_io_timeout_sec(); while ((unsigned long long)offset < file_size) { struct timespec now; clock_gettime(CLOCK_MONOTONIC, &now); diff --git a/src/shared/protocol.c b/src/shared/protocol.c index 7604379..3c3eee0 100644 --- a/src/shared/protocol.c +++ b/src/shared/protocol.c @@ -87,6 +87,12 @@ void protocol_session_set_io_timeout(ProtocolSession* session, int sec) { session->io_timeout_sec = sec; } +int protocol_get_io_timeout_sec(void) { + const ProtocolSession* session = bound_session ? bound_session : &legacy_io_session; + int sec = session->io_timeout_sec; + return sec > 0 ? sec : RECEIVE_TIMEOUT_SEC; +} + void protocol_session_set_max_alloc(ProtocolSession* session, unsigned long long max_alloc) { if (!session) session = bound_session ? bound_session : &legacy_io_session; diff --git a/src/shared/protocol.h b/src/shared/protocol.h index 5015971..60f6dec 100644 --- a/src/shared/protocol.h +++ b/src/shared/protocol.h @@ -152,6 +152,10 @@ void protocol_session_set_max_alloc(ProtocolSession* session, unsigned long long * An explicit long deadline (e.g. the delete-ack wait) is applied per-call by * protocol_receive_status_timed and is unaffected by this setter. */ void protocol_session_set_io_timeout(ProtocolSession* session, int sec); +/* Effective per-message I/O deadline (seconds) for the currently-bound session, + * falling back to the built-in default. Used by the plaintext sendfile path + * which bypasses the protocol send primitive. */ +int protocol_get_io_timeout_sec(void); void* protocol_alloc(size_t size); void* protocol_realloc(void* ptr, size_t size); void protocol_session_set_8_bit_output(ProtocolSession* session, bool enabled); diff --git a/tests/test_receiver_timeout.c b/tests/test_receiver_timeout.c index 1bb65c8..8066b5c 100644 --- a/tests/test_receiver_timeout.c +++ b/tests/test_receiver_timeout.c @@ -62,19 +62,28 @@ static void test_receiver_aborts_idle_keepalive() { protocol_session_bind(&session); Status keepalive = STATUS_KEEPALIVE; - EXPECT_EQ_INT((int)write(sv[0], &keepalive, sizeof(keepalive)), (int)sizeof(keepalive)); - int result = receiver_process_pending(config, sv[1], &sink, NULL); - EXPECT_EQ_INT(result, -1); - + ssize_t wrote = write(sv[0], &keepalive, sizeof(keepalive)); + int result = -2; + if (wrote == (ssize_t)sizeof(keepalive)) + result = receiver_process_pending(config, sv[1], &sink, NULL); Status reply = STATUS_OK; - EXPECT_EQ_INT((int)read(sv[0], &reply, sizeof(reply)), (int)sizeof(reply)); - EXPECT_EQ_INT((int)reply, (int)STATUS_ERROR); + ssize_t got = -1; + if (result == -1) + got = read(sv[0], &reply, sizeof(reply)); + /* Tear down the binding/descriptors BEFORE asserting: an EXPECT_* failure + * returns immediately, and a dangling bound_session would poison later + * fd-level protocol I/O tests. */ protocol_session_unbind(); config_delete(config); close(sv[0]); close(sv[1]); receiver_reset_time_limits(); + + EXPECT_EQ_INT((int)wrote, (int)sizeof(keepalive)); + EXPECT_EQ_INT(result, -1); + EXPECT_EQ_INT((int)got, (int)sizeof(reply)); + EXPECT_EQ_INT((int)reply, (int)STATUS_ERROR); } void test_receiver_timeout(void) { From 69fe7f3c9fde6f4d0118e860e5b61d88513c1068 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 04:06:30 +0200 Subject: [PATCH 020/155] perf(send,scanner): byte-bound sender queues; drop redundant stat --- src/client/client_send.c | 30 +++++--- src/client/scanner.c | 10 ++- src/shared/multiprocessing.c | 81 +++++++++++++++++++++ src/shared/multiprocessing.h | 26 +++++++ tests/test_multiprocessing.c | 119 +++++++++++++++++++++++++++++++ tests/test_scanner.c | 134 +++++++++++++++++++++++++++++++++++ 6 files changed, 389 insertions(+), 11 deletions(-) diff --git a/src/client/client_send.c b/src/client/client_send.c index a0bb9dc..bfec60a 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -37,6 +37,13 @@ #define STREAM_THRESHOLD (64ULL * 1024 * 1024) +/* Aggregate loaded payload bytes the sender may buffer across the loader queue + and the chunk in flight. Sending one chunk adds up to ~2 * MAX_CHUNK_SIZE of + transient serialize/compress buffers on top of the queued payloads, so this + ceiling keeps total pipeline memory within MAX_CONNECTION_MEMORY (mirrors the + receiver's RECEIVER_QUEUE_MAX_BYTES). */ +#define SENDER_QUEUE_MAX_BYTES (MAX_CONNECTION_MEMORY - 2 * MAX_CHUNK_SIZE) + /* Forward declaration for progress-reporting thread used in multithreaded send. */ static int progress_thread_fn(void* arg); @@ -1493,6 +1500,10 @@ static int send_chunks_multithreaded(void* pipeline_context) { protocol_session_unbind(); return thrd_error; } + /* Payload bytes this chunk was charged for on the loader's byte budget. + Computed before destruction and released after the memory is actually + freed, so a loader blocked on the budget wakes only once room exists. */ + size_t queued_payload = pipeline_context_sender_chunk_bytes(current_chunk); unsigned long long chunk_bytes = 0; int chunk_files = 0; for (int i = 0; i < current_chunk->element_count; i++) { @@ -1507,6 +1518,7 @@ static int send_chunks_multithreaded(void* pipeline_context) { context->progress_bytes = context->total_bytes; mtx_unlock(&context->mutex_progress); chunk_destroy(current_chunk); + pipeline_context_sender_note_bytes_released(context, queued_payload); } /* Completion tail: reached on natural exhaustion or an early stop deadline. @@ -1728,11 +1740,7 @@ static int load_files_multithreaded(void* pipeline_context) { } } } - if (!queue_enqueue_multithreaded_cancel(context->queue_loader, chunk, &context->mutex_loader, - &context->condition_not_empty_loader, - &context->condition_not_full_loader, - &context->cancelled)) { - chunk_destroy(chunk); + if (!pipeline_context_sender_enqueue_chunk(context, chunk)) { pipeline_cancel(context); protocol_session_unbind(); return thrd_error; @@ -2197,8 +2205,13 @@ int send_files_multithreaded(Config** config_ptr) { unsigned long long available_memory = pages > 0 && page_size > 0 ? (unsigned long long)pages * (unsigned long long)page_size : 512ULL * 1024 * 1024; - unsigned long long avg_file_size = 1024 * 1024; - int qsize = (int)(available_memory / avg_file_size); + /* Size the chunk queues from the actual chunk size rather than a fixed 1 MiB + average: a chunk holds roughly `chunk_size` bytes of file data, so counting + chunks at 1 MiB over-estimated the queue capacity by up to 10x. The byte + budget below is the authoritative bound; this count keeps the unloaded + chunks waiting in the scanner queue bounded too. */ + unsigned long long chunk_size = config->chunk_size > 0 ? config->chunk_size : DEFAULT_CHUNK_SIZE; + int qsize = (int)(available_memory / chunk_size); if (qsize < 10) qsize = 10; if (qsize > 1000) @@ -2223,7 +2236,8 @@ int send_files_multithreaded(Config** config_ptr) { } context->missing_args = missing_args; missing_args = NULL; /* owned by the context from here on */ - *config_ptr = NULL; /* context now owns config through all remaining paths */ + pipeline_context_sender_set_queue_byte_limit(context, SENDER_QUEUE_MAX_BYTES); + *config_ptr = NULL; /* context now owns config through all remaining paths */ struct timespec now_mono; if (clock_gettime(CLOCK_MONOTONIC, &now_mono) != 0) { now_mono.tv_sec = 0; diff --git a/src/client/scanner.c b/src/client/scanner.c index 6fd3407..7db3f58 100644 --- a/src/client/scanner.c +++ b/src/client/scanner.c @@ -388,9 +388,13 @@ static int scanner_inspect_entry(const ScannerOptions* options, const char* sour goto apply_filters; regular: - if (stat(entry->path, &entry->stats) != 0) - goto skip; - entry->is_directory = S_ISDIR(entry->stats.st_mode); + /* Not a symlink: the lstat() above already described this entry, and lstat + and stat are identical for every non-symlink, so reuse that result instead + of issuing a redundant stat() on the scanner hot path. stat() is still + used on the dereference paths above/below for actual symlinks (copy-links, + safe/copy-unsafe links, and -k symlinks-to-directories). */ + entry->stats = link_stats; + entry->is_directory = S_ISDIR(link_stats.st_mode); if (entry->is_directory) return 1; diff --git a/src/shared/multiprocessing.c b/src/shared/multiprocessing.c index b93c9f0..697e461 100644 --- a/src/shared/multiprocessing.c +++ b/src/shared/multiprocessing.c @@ -11,6 +11,7 @@ #include "protocol.h" #include "queue.h" #include "utils.h" +#include #include #include #include @@ -26,6 +27,8 @@ PipelineContextSender* pipeline_context_sender_create(Config* config, Queue* que context->queue_loader = queue_loader; context->scanner_done = false; context->loader_done = false; + context->queued_bytes = 0; + context->max_queue_bytes = 0; context->manifest = NULL; context->excluded_paths = NULL; context->missing_args = NULL; @@ -97,6 +100,84 @@ fail: return NULL; } +void pipeline_context_sender_set_queue_byte_limit(PipelineContextSender* context, + size_t max_bytes) { + if (context == NULL) + return; + mtx_lock(&context->mutex_loader); + context->max_queue_bytes = max_bytes; + context->queued_bytes = 0; + cnd_broadcast(&context->condition_not_full_loader); + mtx_unlock(&context->mutex_loader); +} + +size_t pipeline_context_sender_chunk_bytes(const Chunk* chunk) { + if (chunk == NULL || chunk->items == NULL) + return 0; + size_t total = 0; + for (int i = 0; i < chunk->element_count; i++) { + const File* file = chunk->items[i]; + if (file == NULL || file->data == NULL || file->data->data == NULL) + continue; + if (file->data->size > SIZE_MAX - total) + return SIZE_MAX; + total += file->data->size; + } + return total; +} + +void pipeline_context_sender_note_bytes_released(PipelineContextSender* context, + size_t released_bytes) { + if (context == NULL || context->max_queue_bytes == 0 || released_bytes == 0) + return; + mtx_lock(&context->mutex_loader); + if (released_bytes >= context->queued_bytes) + context->queued_bytes = 0; + else + context->queued_bytes -= released_bytes; + cnd_signal(&context->condition_not_full_loader); + mtx_unlock(&context->mutex_loader); +} + +bool pipeline_context_sender_enqueue_chunk(PipelineContextSender* context, Chunk* chunk) { + if (context == NULL || chunk == NULL) + return false; + size_t chunk_bytes = pipeline_context_sender_chunk_bytes(chunk); + mtx_lock(&context->mutex_loader); + while (!atomic_load(&context->cancelled)) { + bool blocked_by_count = queue_is_full(context->queue_loader); + bool blocked_by_budget = false; + if (context->max_queue_bytes > 0) { + size_t budget = context->max_queue_bytes; + size_t used = context->queued_bytes; + if (used >= budget) { + blocked_by_budget = true; + } else if (chunk_bytes > budget - used) { + /* A single payload larger than the whole budget is only admitted to an + empty pipeline so the wait can never deadlock. */ + blocked_by_budget = used != 0; + } + } + if (!blocked_by_count && !blocked_by_budget) + break; + cnd_wait(&context->condition_not_full_loader, &context->mutex_loader); + } + if (atomic_load(&context->cancelled)) { + mtx_unlock(&context->mutex_loader); + chunk_destroy(chunk); + return false; + } + if (!queue_enqueue(context->queue_loader, chunk)) { + mtx_unlock(&context->mutex_loader); + chunk_destroy(chunk); + return false; + } + context->queued_bytes += chunk_bytes; + cnd_signal(&context->condition_not_empty_loader); + mtx_unlock(&context->mutex_loader); + return true; +} + void pipeline_context_sender_destroy(PipelineContextSender* context) { if (context->manifest) { array_list_delete(context->manifest); diff --git a/src/shared/multiprocessing.h b/src/shared/multiprocessing.h index 5cbf236..73162a4 100644 --- a/src/shared/multiprocessing.h +++ b/src/shared/multiprocessing.h @@ -5,6 +5,7 @@ #include #include "array_list.h" +#include "chunk.h" #include "config.h" #include "file.h" #include "protocol.h" @@ -25,6 +26,15 @@ typedef struct { cnd_t condition_not_full_loader; cnd_t condition_not_empty_loader; bool loader_done; + /* Aggregate loaded payload bytes queued on queue_loader but not yet released + by the sender. Guarded by `mutex_loader`. When `max_queue_bytes` is + non-zero the loader blocks before enqueueing a chunk that would push this + total over it, so the sender buffers a bounded number of bytes rather than + an unbounded count of chunks that may each be up to chunk_size (or a single + file) in size. Files streamed straight from disk by sendfile hold no + payload, so only in-memory (`data->data`) payloads are counted. */ + size_t queued_bytes; + size_t max_queue_bytes; ArrayList* manifest; /* Protected prefixes (paths the source scan excluded by user rules) sent with the keep-set manifest so --delete leaves them alone unless @@ -113,6 +123,22 @@ typedef struct PipelineContextReceiver { PipelineContextSender* pipeline_context_sender_create(Config* config, Queue* queue_scanner, Queue* queue_loader); void pipeline_context_sender_destroy(PipelineContextSender* context); +/* Bound the loaded payload bytes the sender may buffer ahead of the network + writer (see max_queue_bytes). */ +void pipeline_context_sender_set_queue_byte_limit(PipelineContextSender* context, size_t max_bytes); +/* Total payload bytes a chunk currently holds in memory (loaded file data + only; zero for entries with no payload or data streamed from disk). */ +size_t pipeline_context_sender_chunk_bytes(const Chunk* chunk); +/* Blocking enqueue used by the sender's loader stage. Blocks while + queue_loader is full by element count or when adding `chunk` would push the + queued payload bytes over the configured byte limit; waits until the sender + releases bytes. Takes ownership of `chunk` on success and destroys it on + failure/cancel. */ +bool pipeline_context_sender_enqueue_chunk(PipelineContextSender* context, Chunk* chunk); +/* Account for `released_bytes` of payload memory that the sender freed after + destroying a chunk, unblocking a loader waiting on the byte limit. */ +void pipeline_context_sender_note_bytes_released(PipelineContextSender* context, + size_t released_bytes); PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue_receiver, int file_descriptor, SSL* ssl); void pipeline_context_receiver_destroy(PipelineContextReceiver* context); diff --git a/tests/test_multiprocessing.c b/tests/test_multiprocessing.c index fb0241d..427b5c6 100644 --- a/tests/test_multiprocessing.c +++ b/tests/test_multiprocessing.c @@ -332,6 +332,123 @@ static void test_receiver_enqueue_byte_budget() { config_delete(cfg); } +/* Build a one-file chunk that appears to hold `bytes` of loaded payload by + handing it a real buffer of that size (the byte accounting counts only + in-memory `data->data`, mirroring sendfile's streamed chunks). */ +static Chunk* make_loaded_chunk(const char* name, size_t bytes) { + File* file = file_create(name); + if (!file) + return NULL; + void* buffer = malloc(bytes > 0 ? bytes : 1); + if (!buffer) { + file_destroy(file); + return NULL; + } + file->data->data = buffer; + file->data->size = bytes; + File* items[1] = {file}; + Chunk* chunk = chunk_create(items, 1); + if (!chunk) + file_destroy(file); + return chunk; +} + +/* Only loaded (in-memory) payload is charged: a chunk whose files have no + buffer (e.g. sendfile streams the bytes from disk) accounts for zero. */ +static void test_sender_chunk_bytes_accounting() { + EXPECT_EQ_INT((int)pipeline_context_sender_chunk_bytes(NULL), 0); + + File* streamed = file_create("sender_account_streamed"); + EXPECT_NOT_NULL(streamed); + streamed->data->size = 4096; /* declared size, but no in-memory buffer */ + File* streamed_items[1] = {streamed}; + Chunk* streamed_chunk = chunk_create(streamed_items, 1); + EXPECT_NOT_NULL(streamed_chunk); + EXPECT_EQ_INT((int)pipeline_context_sender_chunk_bytes(streamed_chunk), 0); + chunk_destroy(streamed_chunk); + + Chunk* loaded = make_loaded_chunk("sender_account_loaded", 2000); + EXPECT_NOT_NULL(loaded); + EXPECT_EQ_INT((int)pipeline_context_sender_chunk_bytes(loaded), 2000); + chunk_destroy(loaded); +} + +typedef struct { + PipelineContextSender* context; + Chunk* chunk; + atomic_bool* done; + atomic_bool* result; +} SenderByteBudgetArg; + +static int sender_byte_budget_worker(void* arg) { + SenderByteBudgetArg* worker = arg; + bool ok = pipeline_context_sender_enqueue_chunk(worker->context, worker->chunk); + atomic_store(worker->result, ok); + atomic_store(worker->done, true); + return thrd_success; +} + +/* The sender's loader stage must not buffer more loaded payload bytes ahead of + the network writer than the configured byte budget: an enqueue that would + exceed the budget blocks until the sender releases bytes. */ +static void test_sender_enqueue_byte_budget() { + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + free(cfg->version); + cfg->version = str_dup(PROTOCOL_VERSION); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + + Queue* q_scanner = queue_create(16, chunk_destroy); + Queue* q_loader = queue_create(16, chunk_destroy); + EXPECT_NOT_NULL(q_scanner); + EXPECT_NOT_NULL(q_loader); + PipelineContextSender* ctx = pipeline_context_sender_create(cfg, q_scanner, q_loader); + EXPECT_NOT_NULL(ctx); + pipeline_context_sender_set_queue_byte_limit(ctx, 3000); + + Chunk* first = make_loaded_chunk("sender_budget_1", 2000); + EXPECT_NOT_NULL(first); + EXPECT_TRUE(pipeline_context_sender_enqueue_chunk(ctx, first)); + EXPECT_EQ_INT((int)ctx->queued_bytes, 2000); + + /* A second 2000-byte chunk would push the pipeline to 4000 > 3000 budget, so + its enqueue must block until the first chunk's bytes are released. */ + Chunk* second = make_loaded_chunk("sender_budget_2", 2000); + EXPECT_NOT_NULL(second); + atomic_bool done; + atomic_bool result; + atomic_init(&done, false); + atomic_init(&result, false); + SenderByteBudgetArg arg = {ctx, second, &done, &result}; + thrd_t enqueuer; + EXPECT_EQ_INT(thrd_create(&enqueuer, sender_byte_budget_worker, &arg), thrd_success); + + /* Give a broken (unbounded) implementation every chance to enqueue. */ + struct timespec wait = {0, 200 * 1000000L}; + thrd_sleep(&wait, NULL); + EXPECT_FALSE(atomic_load(&done)); + EXPECT_EQ_INT((int)ctx->queued_bytes, 2000); /* budget still honored */ + + /* Simulate the sender: dequeue + destroy the first chunk, then release its + bytes. Only the post-join state (below) is deterministic. */ + Chunk* drained = + queue_dequeue_multithreaded(q_loader, &ctx->mutex_loader, &ctx->condition_not_empty_loader, + &ctx->condition_not_full_loader, &ctx->loader_done); + EXPECT_NOT_NULL(drained); + chunk_destroy(drained); + pipeline_context_sender_note_bytes_released(ctx, 2000); + + EXPECT_EQ_INT(thrd_join(enqueuer, NULL), thrd_success); + EXPECT_TRUE(atomic_load(&done)); + EXPECT_TRUE(atomic_load(&result)); + EXPECT_EQ_INT((int)ctx->queued_bytes, 2000); /* second payload now in flight */ + + /* pipeline_context_sender_destroy frees the still-queued second chunk and + owns cfg/q_scanner/q_loader from here on. */ + pipeline_context_sender_destroy(ctx); +} + void test_multiprocessing() { test_sender_create_destroy(); test_receiver_create_destroy(); @@ -344,4 +461,6 @@ void test_multiprocessing() { } test_write_thread_done(); test_receiver_enqueue_byte_budget(); + test_sender_chunk_bytes_accounting(); + test_sender_enqueue_byte_budget(); } diff --git a/tests/test_scanner.c b/tests/test_scanner.c index 3ba35ed..9c3bcb6 100644 --- a/tests/test_scanner.c +++ b/tests/test_scanner.c @@ -1369,6 +1369,139 @@ static void test_scanner_chunk_ownership() { rmdir(dir); } +/* The scanner derives entry type from a single lstat() for non-symlinks + * (regular files and directories) and only calls stat() to dereference real + * symlinks. Guard the regular-file/directory/symlink distinction across the + * default (symlinks skipped), --copy-links (dereferenced) and -l (carried) + * modes so the lstat/stat reuse cannot misclassify entries. */ +static void test_scanner_entry_classification() { + const char* root = "test_scan_classify"; + const char* sub = "test_scan_classify/sub"; + const char* file = "test_scan_classify/file.txt"; + const char* nested = "test_scan_classify/sub/nested.txt"; + const char* link_file = "test_scan_classify/link_file"; + const char* link_dir = "test_scan_classify/link_dir"; + + EXPECT_EQ_INT(mkdir(root, 0755), 0); + EXPECT_EQ_INT(mkdir(sub, 0755), 0); + create_test_file(file, "hello"); /* 5 bytes */ + create_test_file(nested, "nested"); /* 6 bytes */ + EXPECT_EQ_INT(symlink("file.txt", link_file), 0); + EXPECT_EQ_INT(symlink("sub", link_dir), 0); + + /* Default: no link option -> symlinks are skipped entirely; regular files and + directories (descended, not emitted) are classified as before. */ + { + ScannerOptions options = {0}; + DirectoryScanner* scanner = directory_scanner_create_with_options(root, &options); + EXPECT_NOT_NULL(scanner); + size_t root_len = strlen(root); + bool file_ok = false, nested_ok = false, link_seen = false; + Chunk* chunk; + while ((chunk = directory_scanner_next(scanner)) != NULL) { + for (int i = 0; i < chunk->element_count; i++) { + const File* f = chunk->items[i]; + const char* rel = f->path + root_len; + if (*rel == '/') + rel++; + if (strcmp(rel, "file.txt") == 0) { + file_ok = !f->is_dir && !f->is_symlink && f->data->size == 5; + } else if (strcmp(rel, "sub/nested.txt") == 0) { + nested_ok = !f->is_dir && !f->is_symlink && f->data->size == 6; + } else { + link_seen = true; + } + } + chunk_destroy(chunk); + } + EXPECT_FALSE(directory_scanner_failed(scanner)); + EXPECT_TRUE(file_ok); + EXPECT_TRUE(nested_ok); + EXPECT_FALSE(link_seen); + directory_scanner_destroy(scanner); + } + + /* --copy-links: symlinks are dereferenced. A link to a file becomes a + regular file with the referent's size; a link to a directory is traversed. */ + { + ScannerOptions options = {0}; + options.copy_links = true; + DirectoryScanner* scanner = directory_scanner_create_with_options(root, &options); + EXPECT_NOT_NULL(scanner); + size_t root_len = strlen(root); + bool file_ok = false, nested_ok = false, link_file_ok = false; + bool link_dir_nested_ok = false, symlink_leaked = false; + Chunk* chunk; + while ((chunk = directory_scanner_next(scanner)) != NULL) { + for (int i = 0; i < chunk->element_count; i++) { + const File* f = chunk->items[i]; + const char* rel = f->path + root_len; + if (*rel == '/') + rel++; + if (f->is_symlink) + symlink_leaked = true; + if (strcmp(rel, "file.txt") == 0) + file_ok = !f->is_dir && f->data->size == 5; + else if (strcmp(rel, "sub/nested.txt") == 0) + nested_ok = !f->is_dir && f->data->size == 6; + else if (strcmp(rel, "link_file") == 0) + link_file_ok = !f->is_dir && f->data->size == 5; + else if (strcmp(rel, "link_dir/nested.txt") == 0) + link_dir_nested_ok = !f->is_dir && f->data->size == 6; + } + chunk_destroy(chunk); + } + EXPECT_FALSE(directory_scanner_failed(scanner)); + EXPECT_TRUE(file_ok); + EXPECT_TRUE(nested_ok); + EXPECT_TRUE(link_file_ok); + EXPECT_TRUE(link_dir_nested_ok); + EXPECT_FALSE(symlink_leaked); + directory_scanner_destroy(scanner); + } + + /* -l (--links): symlinks are carried through as symlinks, not dereferenced. */ + { + ScannerOptions options = {0}; + options.follow_symlinks = true; + DirectoryScanner* scanner = directory_scanner_create_with_options(root, &options); + EXPECT_NOT_NULL(scanner); + size_t root_len = strlen(root); + bool file_ok = false, link_file_ok = false, link_dir_ok = false, leaked_dir = false; + Chunk* chunk; + while ((chunk = directory_scanner_next(scanner)) != NULL) { + for (int i = 0; i < chunk->element_count; i++) { + const File* f = chunk->items[i]; + const char* rel = f->path + root_len; + if (*rel == '/') + rel++; + if (strcmp(rel, "file.txt") == 0) + file_ok = !f->is_dir && !f->is_symlink && f->data->size == 5; + else if (strcmp(rel, "link_file") == 0) + link_file_ok = f->is_symlink && !f->is_dir; + else if (strcmp(rel, "link_dir") == 0) + link_dir_ok = f->is_symlink && !f->is_dir; + else if (strcmp(rel, "link_dir/nested.txt") == 0) + leaked_dir = true; + } + chunk_destroy(chunk); + } + EXPECT_FALSE(directory_scanner_failed(scanner)); + EXPECT_TRUE(file_ok); + EXPECT_TRUE(link_file_ok); + EXPECT_TRUE(link_dir_ok); + EXPECT_FALSE(leaked_dir); + directory_scanner_destroy(scanner); + } + + unlink(link_file); + unlink(link_dir); + unlink(nested); + unlink(file); + rmdir(sub); + rmdir(root); +} + void test_scanner() { test_scanner_single_file(); test_scanner_multiple_files(); @@ -1406,4 +1539,5 @@ void test_scanner() { test_files_from_relative_send_path(); test_scanner_captures_directory_times(); test_scanner_chunk_ownership(); + test_scanner_entry_classification(); } From 317d5d081a5c1d706e678e2f1aba7341d797769e Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 04:08:36 +0200 Subject: [PATCH 021/155] perf(protocol): pack metadata into one frame (PROTOCOL 2.20.0) --- README.md | 2 +- RSYNC_COMPAT.md | 19 +++- src/shared/config.h | 20 +++- src/shared/metadata.c | 145 ++++-------------------- src/shared/metadata.h | 8 +- tests/integration/test_preflight.py | 6 +- tests/test_client_cli.c | 8 +- tests/test_metadata.c | 170 +++++++++++++++++++++++----- 8 files changed, 210 insertions(+), 168 deletions(-) diff --git a/README.md b/README.md index 34059c2..6cd07d0 100644 --- a/README.md +++ b/README.md @@ -551,7 +551,7 @@ before the module list, before authentication, and the connecting peer address ## Protocol and Security -FastSync protocol version `2.19.0` is shared by the client and server. The +FastSync protocol version `2.20.0` is shared by the client and server. The current protocol is sender-driven and includes configuration negotiation, including the maximum allocation limit, incremental checks, checksums, manifests, keep-alives, abort handling, per-file remove-source results, and diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 5692ad0..830605e 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -679,7 +679,7 @@ now transmits targets (the prior behavior was broken/partial); its status moved | `--stop-after=MINS` | Stop after N minutes | ✅ Implemented | Client-only sender stop deadline (Phase 6): computing `--stop-after=MINS` (a positive minute count; 0/negative/garbage rejected) and `--stop-at=TIME` (`HH:MM`, `HH:MM:SS`, or `now+N[smhd]`; a past time stops immediately). The transfer stops ELEGANTLY at the next chunk boundary: everything already fully sent is kept and applied, the run returns 0, and --delete (late/delete-after timing) does NOT wipe the destination — when the scan is cut short the partial keep-set manifest is suppressed with a warning (the delete walk is skipped rather than acting on an incomplete keep-set, so unscanned source mirrors survive). `--delete-before`/`--delete-during` still run their complete pre-scan (which ignores the deadline). Local client-only fields: never serialized into the wire config frame, so no PROTOCOL_VERSION bump. `--stop-after` uses CLOCK_MONOTONIC; `--stop-at` uses the wall clock. Works single-threaded and under `-j`/`--threads` (multithreaded). Divergence: rsync computes `--stop-after` from the run start; FastSync likewise. When both are given, the earlier of the two deadlines wins (checked per iteration). See the Phase-6 stop notes below | | `--stop-at=TIME` | Stop at specified time | ✅ Implemented | Same feature as `--stop-after` (deadline transfer stop), absolute wall-clock form (`HH:MM[:SS]` or `now+N[smhd]`). See the row above and the Phase-6 stop notes | | `--fsync` | Fsync every written file before publication | ✅ Implemented | | -| `--protocol=NUM` | Force older protocol version | ✅ Implemented | Forces the wire protocol version for this transfer. FastSync has exactly ONE wire format (`PROTOCOL_VERSION`, currently 2.19.0) with no downgrade/backward-compat code paths, so `--protocol=2.19.0` is accepted (it sets the version claim the client sends, which the server already requires to match exactly) and **every other value is rejected up front** with a clear error before any connection — it does not and cannot speak an older or virtual wire format. Divergence from rsync (which negotiates a range and downgrades to an integer 0..31): FastSync's honest contract is force-to-the-one-supported-value; a genuine downgrade would require a per-version compatibility layer that does not exist. Client-only; the server-side exact-match check is unchanged. `--protocol=2.18.0`/`2.18`/`2.17.0`/`2.16.0`/`2.15.0`/`216`/`31`/garbage are all rejected. See the Phase-6 protocol note below | +| `--protocol=NUM` | Force older protocol version | ✅ Implemented | Forces the wire protocol version for this transfer. FastSync has exactly ONE wire format (`PROTOCOL_VERSION`, currently 2.20.0) with no downgrade/backward-compat code paths, so `--protocol=2.20.0` is accepted (it sets the version claim the client sends, which the server already requires to match exactly) and **every other value is rejected up front** with a clear error before any connection — it does not and cannot speak an older or virtual wire format. Divergence from rsync (which negotiates a range and downgrades to an integer 0..31): FastSync's honest contract is force-to-the-one-supported-value; a genuine downgrade would require a per-version compatibility layer that does not exist. Client-only; the server-side exact-match check is unchanged. `--protocol=2.19.0`/`2.18.0`/`2.18`/`2.17.0`/`2.16.0`/`2.15.0`/`216`/`31`/garbage are all rejected. See the Phase-6 protocol note below | | `--iconv=CONVERT_SPEC` | Charset conversion | ✅ Implemented | Charset conversion of FILE NAMES (not content) at the protocol boundary via iconv(3): `--iconv=LOCAL[,REMOTE]` — the sender converts each local filename LOCAL→REMOTE before transmitting, and the receiver converts each wire filename REMOTE→LOCAL before creating/writing. The full CONVERT_SPEC is serialized into the config frame as a new trailing string field so the peer knows the wire charset; **PROTOCOL_VERSION bumped 2.15.0 → 2.16.0**. `LOCAL[,REMOTE]` parse: single charset ⇒ LOCAL==REMOTE (identity both ways); garbage rejected up front. Validation probes BOTH directions (a spec that only opens one way is refused, as is a NUL-emitting target charset like utf-16/utf-32/ucs-2, since filenames cannot contain NUL). An unrepresentable name (EILSEQ/EINVAL) fails that path cleanly with a logged `--iconv: cannot convert file name ...` and is never written mangled/truncated. Conversion is applied at EVERY wire-path site (regular/MKDIR/hardlink path+target/symlink path+target/SPECIAL, the delete manifest, the incremental-check path, and the `-s`/`chunk_serialize` embedded blob path), on both client and server (`--iconv` is also a server/daemon option). Zero overhead when unset. See the Phase-6 iconv notes below | | `--checksum-seed=NUM` | Set checksum seed | ✅ Implemented | Sets the seed for FastSync's whole-file xxHash64 digest (full 64-bit seed) and for the delta path's per-block xxHash32 strong checksum (low 32 bits of the seed). An explicit seed deterministically changes every computed digest on BOTH endpoints (sender and receiver share the seed via the config frame, protocol 2.10.0), so identical runs with the same seed skip the same files and a changed seed changes the digests — the explicit-seed path that makes xxHash comparisons deterministic. `--checksum-choice=md5` has no seed and ignores it (documented). The value is a strict decimal 0..2⁶⁴-1 (blank, signed, or non-numeric values are rejected). Like rsync, a seed only matters where a digest is actually computed (`--checksum` or a basis-dir run, or a delta transfer); it does not by itself enable `--checksum`/`--delta`. Divergence from rsync: the default is seed 0, and FastSync never randomizes the seed (rsync uses a random per-transfer seed when `--checksum-seed` is unset); FastSync's unset default therefore reproduces its historical byte-for-byte behavior | | `--secluded-args`, `-s` | Use protocol to send args | ⛔ Impossible/Divergence | Accepted for CLI compatibility (including the rsync short `-s`, Phase 7 Wave A) but a documented **no-op / divergence**. rsync's `-s` protects arguments from shell expansion by shipping them over the protocol; FastSync never passes remote arguments through a shell expansion boundary in the first place — its SSH transport builds the remote argv as **single-quote-escaped shell words** (`ssh_build_remote_command`), so the injection/leak that `-s` guards against does not exist and there is nothing to "seclude". Implementing a true arg-send protocol would mean replacing the argv-based SSH launch with an in-band argument channel, a large redesign of the transport that buys no security here. Chunk serialization remains the long-only `--chunk-serialization`. | @@ -795,7 +795,7 @@ These are the hardest compatibility items because they require durable formats o **Phase 6, Wave B (iconv) shipping note (PROTOCOL 2.15.0 → 2.16.0):** `--iconv=LOCAL[,REMOTE]` converts file NAMES at the wire boundary (never content). The full CONVERT_SPEC is serialized into the config frame as a new trailing string field (empty→NULL canonicalized), so both ends share the same wire charset interpretation; this required the PROTOCOL bump because the frame is a strict ordered sequence and a peer that does not parse the new trailing field would desynchronize. Each end derives LOCAL (its own charset) and REMOTE (the wire charset): the sender opens LOCAL→REMOTE and converts every transmitted filename; the receiver opens REMOTE→LOCAL and converts every received filename before creating/writing. Conversion is applied at every wire-path site (regular/MKDIR/hardlink path+target/symlink path+target/SPECIAL, the delete manifest keep/protected/missing entries, the incremental-check path, and the embedded `-s`/chunk-blob path). A name it cannot convert (EILSEQ/EINVAL) is failed cleanly with a logged `--iconv: cannot convert file name ...` and is never written truncated/mangled. Validation probes both directions up front (both the sender local→remote and the receiver remote→local, and, for a server/daemon with its own `--iconv`, the client-REMOTE→server-LOCAL pair) so an unusable spec is rejected before the connection rather than mid-transfer, and NUL-emitting target charsets (utf-16/utf-32/ucs-2) are refused because filenames cannot contain NUL. Divergence documented upstream: the receiver does NOT half-swap; the wire charset always comes from the sender's REMOTE half, so a server whose local charset differs from the client's LOCAL must declare it with its own `--iconv`. Conversion is process-global and runs on a single thread per process (sender thread / receiver-loop thread), initialized before worker threads start and freed after they join. -**Phase 6, Wave C (protocol-version) shipping note (no PROTOCOL_VERSION change):** `--protocol=NUM` lets the client force the wire protocol version for a transfer. FastSync's protocol is a single lockstep format: the config frame is a strict ordered sequence and the server requires the client's version string to equal `PROTOCOL_VERSION` exactly (`config_receive_with_validate`, src/shared/config.c) — there are no older-format code paths and no downgrade/negotiation machinery, so a lower/higher/virtual version can never be spoken. The honest contract is therefore: `--protocol=2.19.0` (the current `PROTOCOL_VERSION`, as of the A7 auth redesign) is accepted and stored into the client's `version` claim (which `config_send` already transmits), and every other value — `2.18.0`, `2.18`, `2.17.0`, `2.16.0`, `2.15.0`, `3.0.0`, rsync-integer spellings like `216`/`31`, garbage, empty — is rejected up front in `validate_config()` before any connection, with a clear error that FastSync supports only its current wire protocol and cannot speak an older or virtual one. Implementation is client-only: a server-side `--protocol` is intentionally not added because the server has no negotiation (it only enforces exact match), and it could only ever be the current version. This preserves (and slightly tightens) existing validation: the client now also refuses to launch with a version it cannot actually speak, rather than only the server rejecting it later. A genuine downgrade would require a per-version compatibility layer for every frame/feature added since (append 2.10, preallocate 2.11, hardlinks 2.12, devices/specials/symlink-trust/xattr 2.13, remote-option 2.14, daemon module/auth 2.15, iconv 2.16, dir/symlink times 2.17, privilege flags --super/--copy-as 2.18, SCRAM daemon auth 2.19) and is intentionally out of scope — documented divergences from rsync's integer-negotiated downgrade remain. +**Phase 6, Wave C (protocol-version) shipping note (no PROTOCOL_VERSION change):** `--protocol=NUM` lets the client force the wire protocol version for a transfer. FastSync's protocol is a single lockstep format: the config frame is a strict ordered sequence and the server requires the client's version string to equal `PROTOCOL_VERSION` exactly (`config_receive_with_validate`, src/shared/config.c) — there are no older-format code paths and no downgrade/negotiation machinery, so a lower/higher/virtual version can never be spoken. The honest contract is therefore: `--protocol=2.20.0` (the current `PROTOCOL_VERSION`, as of the packed-metadata wave) is accepted and stored into the client's `version` claim (which `config_send` already transmits), and every other value — `2.19.0`, `2.18.0`, `2.18`, `2.17.0`, `2.16.0`, `2.15.0`, `3.0.0`, rsync-integer spellings like `216`/`31`, garbage, empty — is rejected up front in `validate_config()` before any connection, with a clear error that FastSync supports only its current wire protocol and cannot speak an older or virtual one. Implementation is client-only: a server-side `--protocol` is intentionally not added because the server has no negotiation (it only enforces exact match), and it could only ever be the current version. This preserves (and slightly tightens) existing validation: the client now also refuses to launch with a version it cannot actually speak, rather than only the server rejecting it later. A genuine downgrade would require a per-version compatibility layer for every frame/feature added since (append 2.10, preallocate 2.11, hardlinks 2.12, devices/specials/symlink-trust/xattr 2.13, remote-option 2.14, daemon module/auth 2.15, iconv 2.16, dir/symlink times 2.17, privilege flags --super/--copy-as 2.18, SCRAM daemon auth 2.19, packed metadata 2.20) and is intentionally out of scope — documented divergences from rsync's integer-negotiated downgrade remain. **Phase-1/2 selection-and-update status correction (docs):** `-I/--ignore-times`, `--size-only`, `-@/--modify-window`, `--existing`, `--ignore-existing`, `-u/--update`, `-W/--whole-file`, and `--compress-threads` were previously listed as not-implemented in this document but are in fact fully implemented and tested on `dev`. This pass corrects the matrix to match the code. The realistic model of these is that FastSync is a *sender-driven* whole-tree copy, so the size+mtime quick-check and all three receiver-policy skips (`--existing`, `--ignore-existing`, `-u`) are evaluated against the **destination** on the receiver side, and their booleans cross the wire in the config frame. `-I`/`--size-only`/`--modify-window` modify the `--incremental` per-file `STATUS_CHECK` handshake's match predicate (`-I` disables the mtime leg and forces transfer; `--size-only` drops only the mtime leg; `--modify-window` adds tolerance to `metadata_mtime_matches`); they require `--incremental` (or a basis dir) to have a handshake to affect, mirroring how they only matter where a quick-check exists in rsync. `--existing`/`--ignore-existing`/`-u` are receiver write-time policies (skipping the write / newer-destination guard) applied across the regular-file, `--delay-updates`-staged, hardlink-sibling, and special/device paths; `-u` implies `-M` metadata and uses a second-then-nanosecond strict `>` newer check; both correctly influence `--remove-source-files` (a skipped source is not removed). `-W/--whole-file` disables block-level delta (opt-in via `--delta`), folded into the wire `use_delta` so no protocol bump was needed, and makes `--fuzzy` inert; `--append`/`--append-verify` are rejected with `-W`. `--compress-threads=NUM` (1..64, client-only, never crosses the wire) sizes the zstd compression worker pool. No code was changed by this correction; the implementation had landed in earlier merge waves (feat/ignore-times, feat/ignore-existing via the newer `file_to_disk_secure_no_replace`/`linkat EEXIST` path, feat/size-only, feat/modify-window, feat/whole-file, feat/update, compression-threads). @@ -840,6 +840,21 @@ These are the last compatibility items and the closing phase toward rsync flag p **Post-Phase-7 Summary (after Waves A–E).** ✅143 / 🔀0 / ⛔4 / ⚠️0 / 🔄0 / ❌0 = 147. The 3 `🔀 Alt Arg` rows (`-a`, `-p`, `-z`) are ✅ (Wave A). All 10 prior `⚠️ Partial` rows are resolved to ✅ (`-S`, `-P`, `--block-size`, `--fake-super`, `--devices`, `--copy-devices`, `--write-devices`) or ⛔ (`--stderr=client`, `-N/--crtimes`, `--specials` for the impossible socket case). The 3 `🔄 Compatibility No-op` rows are resolved: `-O`/`-J` are now real ✅ (Wave D), `--secluded-args` is ⛔. The **Impossible/Divergence** bucket holds the 4 physically-impossible/divergent flags: `--stderr=client`, `-N/--crtimes`, `--specials` (sockets), `--secluded-args`. The last two `❌ Not Implemented` rows — `--super` and `--copy-as=USER[:GROUP]` — are now ✅ (Wave E). **No `❌ Not Implemented` rows remain.** +## Packed Metadata Frame (protocol 2.20.0) + +A file's metadata used to cross the wire as up to 12 separate per-field framed +messages (a present flag followed by mode/uid/gid/mtime/atime/crtime writes), +which cost ~11 extra protocol frames per file on many-small-file trees. FastSync +now sends the metadata as ONE packed frame: a single `int32` present flag +(`0` = absent) followed, when present, by the fixed +`FILE_METADATA_WIRE_SIZE`-byte (68-byte) field record already emitted by the +shared `metadata_to_buf()`/`metadata_from_buf()` chunk codec. Absent metadata is +a lone `int32` zero. The encoded field layout is unchanged (only the framing +collapses), so chunk-serialized blobs remain byte-identical; `PROTOCOL_VERSION` +was bumped `2.19.0 → 2.20.0` because a 2.19 peer would desynchronize on the +removed frames. The strict same-version handshake rejects any mismatch before a +byte of the frame is parsed. + ### Recommended Delivery Order 1. Resolve short-option conflicts (`-m`, `-M`, `-T`, `-f`, `-s`) and define the compatibility contract. diff --git a/src/shared/config.h b/src/shared/config.h index 77febe8..23b811b 100644 --- a/src/shared/config.h +++ b/src/shared/config.h @@ -642,8 +642,24 @@ typedef struct Config { * anything else) is what keeps a 2.19 client and a 2.18 server from ever * reaching that state. SECURITY: a 2.19 store holds a salted PBKDF2 verifier * and cannot verify (and refuses to load) a legacy unsalted-SHA-256 store line, - * so an old bearer digest can never be replayed against a 2.19 daemon. */ -#define PROTOCOL_VERSION "2.19.0" + * so an old bearer digest can never be replayed against a 2.19 daemon. + * + * Packed Metadata Wave: 2.19.0 -> 2.20.0. + * + * WHY the bump, grounded in the wire: metadata_send()/metadata_receive() no + * longer emit/consume the metadata as up to 12 separate per-field framed + * writes. A file's metadata now crosses the wire as ONE packed frame: a + * single int32 present flag (0 = absent, 1 = present) followed, when present, + * by the fixed FILE_METADATA_WIRE_SIZE-byte (68-byte) field record produced by + * metadata_to_buf(). A 2.19 peer would desynchronize on the removed frames + * (it would read the packed record's bytes as a stream of separate field + * frames), so the strict same-version handshake (config_receive rejects a + * mismatched version before parsing anything else) is what keeps a 2.20 client + * and a 2.19 server from ever reaching that state. The encoded field layout + * itself is unchanged (only its framing collapses), so the chunk codec, which + * already used the packed metadata_to_buf()/metadata_from_buf() codec, is + * byte-identical to before. */ +#define PROTOCOL_VERSION "2.20.0" #define DEFAULT_CHUNK_SIZE (10 * 1024 * 1024) /* Upper bound on total basis-dir entries (rsync caps --link-dest at 20). */ #define MAX_BASIS_DIRS 64 diff --git a/src/shared/metadata.c b/src/shared/metadata.c index b1cff1e..d1f14e5 100644 --- a/src/shared/metadata.c +++ b/src/shared/metadata.c @@ -157,33 +157,17 @@ FileMetadata* metadata_from_buf(char** buf) { bool metadata_send(int file_descriptor, const FileMetadata* m) { if (m == NULL) { - int32_t zero = 0; - return send_n_data(file_descriptor, &zero, sizeof(zero)); + int32_t absent = 0; + return send_n_data(file_descriptor, &absent, sizeof(absent)); } - int32_t present = 1; - int32_t mode = (int32_t)m->mode; - int32_t uid = (int32_t)m->uid; - int32_t gid = (int32_t)m->gid; - int64_t mtime_sec = (int64_t)m->mtime_sec; - int64_t mtime_nsec = (int64_t)m->mtime_nsec; - int32_t atime_valid = m->atime_valid ? 1 : 0; - int64_t atime_sec = (int64_t)m->atime_sec; - int64_t atime_nsec = (int64_t)m->atime_nsec; - int32_t crtime_valid = m->crtime_valid ? 1 : 0; - int64_t crtime_sec = (int64_t)m->crtime_sec; - int64_t crtime_nsec = (int64_t)m->crtime_nsec; - return send_n_data(file_descriptor, &present, sizeof(present)) && - send_n_data(file_descriptor, &mode, sizeof(mode)) && - send_n_data(file_descriptor, &uid, sizeof(uid)) && - send_n_data(file_descriptor, &gid, sizeof(gid)) && - send_n_data(file_descriptor, &mtime_sec, sizeof(mtime_sec)) && - send_n_data(file_descriptor, &mtime_nsec, sizeof(mtime_nsec)) && - send_n_data(file_descriptor, &atime_valid, sizeof(atime_valid)) && - send_n_data(file_descriptor, &atime_sec, sizeof(atime_sec)) && - send_n_data(file_descriptor, &atime_nsec, sizeof(atime_nsec)) && - send_n_data(file_descriptor, &crtime_valid, sizeof(crtime_valid)) && - send_n_data(file_descriptor, &crtime_sec, sizeof(crtime_sec)) && - send_n_data(file_descriptor, &crtime_nsec, sizeof(crtime_nsec)); + /* One packed frame (protocol 2.20.0): the int32 present flag followed by the + fixed FILE_METADATA_WIRE_SIZE-byte field record. metadata_to_buf() emits + exactly that layout (present + fields), so build it once and write the + whole record in a single call instead of one frame per field. */ + char packed[sizeof(int32_t) + FILE_METADATA_WIRE_SIZE]; + char* cursor = packed; + metadata_to_buf(&cursor, m); + return send_n_data(file_descriptor, packed, sizeof(packed)); } FileMetadata* metadata_receive(int file_descriptor, int* ok) { @@ -203,109 +187,22 @@ FileMetadata* metadata_receive(int file_descriptor, int* ok) { *ok = 0; return NULL; } - FileMetadata* m = protocol_alloc(sizeof(FileMetadata)); + /* Rebuild the packed record metadata_from_buf() expects: the present flag we + just read, followed by exactly FILE_METADATA_WIRE_SIZE field bytes. */ + char packed[sizeof(int32_t) + FILE_METADATA_WIRE_SIZE]; + memcpy(packed, &present, sizeof(present)); + if (!receive_n_data(file_descriptor, packed + sizeof(present), FILE_METADATA_WIRE_SIZE)) { + if (ok) + *ok = 0; + return NULL; + } + char* cursor = packed; + FileMetadata* m = metadata_from_buf(&cursor); if (m == NULL) { if (ok) *ok = 0; return NULL; } - int32_t mode; - if (!receive_n_data(file_descriptor, &mode, sizeof(mode))) { - free(m); - if (ok) - *ok = 0; - return NULL; - } - m->mode = (mode_t)mode; - int32_t uid; - if (!receive_n_data(file_descriptor, &uid, sizeof(uid))) { - free(m); - if (ok) - *ok = 0; - return NULL; - } - m->uid = (uid_t)uid; - int32_t gid; - if (!receive_n_data(file_descriptor, &gid, sizeof(gid))) { - free(m); - if (ok) - *ok = 0; - return NULL; - } - m->gid = (gid_t)gid; - int64_t mtime_sec; - if (!receive_n_data(file_descriptor, &mtime_sec, sizeof(mtime_sec))) { - free(m); - if (ok) - *ok = 0; - return NULL; - } - m->mtime_sec = (time_t)mtime_sec; - int64_t mtime_nsec; - if (!receive_n_data(file_descriptor, &mtime_nsec, sizeof(mtime_nsec))) { - free(m); - if (ok) - *ok = 0; - return NULL; - } - m->mtime_nsec = (long)mtime_nsec; - int32_t atime_valid; - if (!receive_n_data(file_descriptor, &atime_valid, sizeof(atime_valid))) { - free(m); - if (ok) - *ok = 0; - return NULL; - } - int64_t atime_sec; - if (!receive_n_data(file_descriptor, &atime_sec, sizeof(atime_sec))) { - free(m); - if (ok) - *ok = 0; - return NULL; - } - int64_t atime_nsec; - if (!receive_n_data(file_descriptor, &atime_nsec, sizeof(atime_nsec))) { - free(m); - if (ok) - *ok = 0; - return NULL; - } - int32_t crtime_valid; - if (!receive_n_data(file_descriptor, &crtime_valid, sizeof(crtime_valid))) { - free(m); - if (ok) - *ok = 0; - return NULL; - } - int64_t crtime_sec; - if (!receive_n_data(file_descriptor, &crtime_sec, sizeof(crtime_sec))) { - free(m); - if (ok) - *ok = 0; - return NULL; - } - int64_t crtime_nsec; - if (!receive_n_data(file_descriptor, &crtime_nsec, sizeof(crtime_nsec))) { - free(m); - if (ok) - *ok = 0; - return NULL; - } - m->atime_valid = atime_valid != 0; - m->atime_sec = (time_t)atime_sec; - m->atime_nsec = (long)atime_nsec; - m->crtime_valid = crtime_valid != 0; - m->crtime_sec = (time_t)crtime_sec; - m->crtime_nsec = (long)crtime_nsec; - if (mtime_nsec < 0 || mtime_nsec >= 1000000000LL || mode < 0 || uid < 0 || gid < 0 || - atime_valid < 0 || atime_valid > 1 || crtime_valid < 0 || crtime_valid > 1 || - (atime_valid && (atime_nsec < 0 || atime_nsec >= 1000000000LL)) || - (crtime_valid && (crtime_nsec < 0 || crtime_nsec >= 1000000000LL))) { - free(m); - if (ok) - *ok = 0; - return NULL; - } if (ok) *ok = 1; return m; diff --git a/src/shared/metadata.h b/src/shared/metadata.h index faf5194..0540037 100644 --- a/src/shared/metadata.h +++ b/src/shared/metadata.h @@ -29,7 +29,13 @@ /* Size of metadata fields on wire, excluding the int32_t `present` field that * is always sent first. The total wire size for present metadata is - * sizeof(int32_t) + FILE_METADATA_WIRE_SIZE (68 bytes on most platforms). */ + * sizeof(int32_t) + FILE_METADATA_WIRE_SIZE (68 bytes on most platforms). + * + * metadata_send()/metadata_receive() (protocol 2.20.0) frame the metadata as a + * single packed record: one int32 present flag (0 = absent) followed, when + * present, by exactly FILE_METADATA_WIRE_SIZE bytes of field data. This is the + * same present+fields byte layout metadata_to_buf()/metadata_from_buf() use, so + * the wire metadata is now one frame instead of one frame per field. */ #define FILE_METADATA_WIRE_SIZE (sizeof(int32_t) * 5 + sizeof(int64_t) * 6) void metadata_to_buf(char** buf, const FileMetadata* m); diff --git a/tests/integration/test_preflight.py b/tests/integration/test_preflight.py index cbc1449..4cbfda3 100644 --- a/tests/integration/test_preflight.py +++ b/tests/integration/test_preflight.py @@ -94,14 +94,14 @@ def _seed_protocol_source(source): class TestProtocol: @pytest.mark.ci def test_protocol_current_version_accepted(self, shared_server): - """--protocol=2.19.0 (the current PROTOCOL_VERSION) is accepted and the + """--protocol=2.20.0 (the current PROTOCOL_VERSION) is accepted and the transfer completes normally.""" source = os.path.join(TEST_DATA_DIR, "proto_ok_src") dest = os.path.join(TEST_DATA_DIR, "proto_ok_dst") shutil.rmtree(dest, ignore_errors=True) os.makedirs(dest) _seed_protocol_source(source) - result, _ = run_client(source, dest, flags=["--protocol=2.19.0"], + result, _ = run_client(source, dest, flags=["--protocol=2.20.0"], port=shared_server.port) assert result.returncode == 0, \ f"--protocol current run failed: {(result.stderr or result.stdout)[:400]}" @@ -118,7 +118,7 @@ class TestProtocol: shutil.rmtree(dest, ignore_errors=True) os.makedirs(dest) _seed_protocol_source(source) - for bad in ("2.18.0", "2.17.0", "2.15.0", "2.16.0", "216", "31"): + for bad in ("2.19.0", "2.18.0", "2.17.0", "2.15.0", "2.16.0", "216", "31"): result, _ = run_client(source, dest, flags=[f"--protocol={bad}"], port=shared_server.port) assert result.returncode != 0, f"--protocol={bad} should be rejected" diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c index 44d6de7..786efe9 100644 --- a/tests/test_client_cli.c +++ b/tests/test_client_cli.c @@ -259,7 +259,7 @@ static void test_parse_args_protocol_accept_current() { Config* cfg = valid_client_config(); EXPECT_NOT_NULL(cfg); char* argv_equals[] = {"fastsync", "--source-dir", "/src", - "--dest-dir", "/dst", "--protocol=2.19.0"}; + "--dest-dir", "/dst", "--protocol=2.20.0"}; int positional_args[2]; int positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 6, argv_equals, positional_args, &positional_count), 0); @@ -269,7 +269,7 @@ static void test_parse_args_protocol_accept_current() { cfg = valid_client_config(); EXPECT_NOT_NULL(cfg); char* argv_space[] = {"fastsync", "--source-dir", "/src", "--dest-dir", - "/dst", "--protocol", "2.19.0"}; + "/dst", "--protocol", "2.20.0"}; positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 7, argv_space, positional_args, &positional_count), 0); EXPECT_EQ_STR(cfg->version, PROTOCOL_VERSION); @@ -279,8 +279,8 @@ static void test_parse_args_protocol_accept_current() { /* Any --protocol value other than the current PROTOCOL_VERSION must end in * failure (parse_args simply stores it; validate_config rejects it up front). */ static void test_parse_args_protocol_rejects_other_versions() { - static const char* const bad_versions[] = {"2.17", "2.16", "2.15.0", "2.16.0", "2.17.0", - "2.18.0", "216", "31", "abc", ""}; + static const char* const bad_versions[] = { + "2.17", "2.16", "2.15.0", "2.16.0", "2.17.0", "2.18.0", "2.19.0", "216", "31", "abc", ""}; for (size_t i = 0; i < sizeof(bad_versions) / sizeof(bad_versions[0]); i++) { Config* cfg = valid_client_config(); EXPECT_NOT_NULL(cfg); diff --git a/tests/test_metadata.c b/tests/test_metadata.c index 74a5f27..b6c7870 100644 --- a/tests/test_metadata.c +++ b/tests/test_metadata.c @@ -5,6 +5,8 @@ #include "test_utils.h" #include #include +#include +#include #include #include @@ -134,6 +136,115 @@ static void test_metadata_send_null() { close(p[1]); } +/* protocol 2.20.0: metadata is one packed frame. With metadata present the + * wire record is exactly sizeof(int32_t) + FILE_METADATA_WIRE_SIZE bytes (the + * present flag followed by the fixed field record); absent metadata is a lone + * int32 zero. */ +static void test_metadata_wire_is_one_packed_frame() { + io_set_bwlimit(0); + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + io_set_fds(p[0], p[1]); + + FileMetadata original = {.mode = 0640, + .uid = 42, + .gid = 43, + .mtime_sec = 111, + .mtime_nsec = 222, + .atime_valid = true, + .atime_sec = 333, + .atime_nsec = 444, + .crtime_valid = false, + .crtime_sec = 0, + .crtime_nsec = 0}; + EXPECT_TRUE(metadata_send(p[1], &original)); + + unsigned char wire[sizeof(int32_t) + FILE_METADATA_WIRE_SIZE]; + EXPECT_EQ_INT((int)read(p[0], wire, sizeof(wire)), (int)sizeof(wire)); + int32_t flag; + memcpy(&flag, wire, sizeof(flag)); + EXPECT_EQ_INT(flag, 1); + int avail = -1; + EXPECT_EQ_INT(ioctl(p[0], FIONREAD, &avail), 0); + EXPECT_EQ_INT(avail, 0); + + /* The present frame decodes in one shot with the shared codec. */ + char* cursor = (char*)wire; + FileMetadata* decoded = metadata_from_buf(&cursor); + EXPECT_NOT_NULL(decoded); + EXPECT_EQ_INT((int)(cursor - (char*)wire), (int)sizeof(wire)); + EXPECT_EQ_INT(decoded->mode, 0640); + EXPECT_EQ_INT(decoded->uid, 42); + EXPECT_EQ_INT(decoded->gid, 43); + EXPECT_EQ_INT(decoded->mtime_sec, 111); + EXPECT_EQ_INT(decoded->mtime_nsec, 222); + EXPECT_TRUE(decoded->atime_valid); + EXPECT_EQ_INT(decoded->atime_sec, 333); + EXPECT_EQ_INT(decoded->atime_nsec, 444); + EXPECT_FALSE(decoded->crtime_valid); + free(decoded); + + /* Absent metadata is a lone int32 zero (4 bytes). */ + EXPECT_TRUE(metadata_send(p[1], NULL)); + EXPECT_EQ_INT((int)read(p[0], wire, sizeof(int32_t)), (int)sizeof(int32_t)); + memcpy(&flag, wire, sizeof(flag)); + EXPECT_EQ_INT(flag, 0); + avail = -1; + EXPECT_EQ_INT(ioctl(p[0], FIONREAD, &avail), 0); + EXPECT_EQ_INT(avail, 0); + + close(p[0]); + close(p[1]); +} + +/* Round-trip over a socketpair (not just a pipe): present metadata compares + * equal field-by-field and absent metadata yields NULL with ok == 1. */ +static void test_metadata_send_receive_socketpair() { + io_set_bwlimit(0); + int sv[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, sv), 0); + io_set_fds(sv[0], sv[1]); + + FileMetadata original = {.mode = 0600, + .uid = 7, + .gid = 8, + .mtime_sec = 1000, + .mtime_nsec = 1, + .atime_valid = true, + .atime_sec = 2000, + .atime_nsec = 2, + .crtime_valid = true, + .crtime_sec = 3000, + .crtime_nsec = 3}; + EXPECT_TRUE(metadata_send(sv[1], &original)); + + int ok = 0; + FileMetadata* received = metadata_receive(sv[0], &ok); + EXPECT_NOT_NULL(received); + EXPECT_EQ_INT(ok, 1); + EXPECT_EQ_INT(received->mode, 0600); + EXPECT_EQ_INT(received->uid, 7); + EXPECT_EQ_INT(received->gid, 8); + EXPECT_EQ_INT(received->mtime_sec, 1000); + EXPECT_EQ_INT(received->mtime_nsec, 1); + EXPECT_TRUE(received->atime_valid); + EXPECT_EQ_INT(received->atime_sec, 2000); + EXPECT_EQ_INT(received->atime_nsec, 2); + EXPECT_TRUE(received->crtime_valid); + EXPECT_EQ_INT(received->crtime_sec, 3000); + EXPECT_EQ_INT(received->crtime_nsec, 3); + free(received); + + EXPECT_TRUE(metadata_send(sv[1], NULL)); + ok = 0; + received = metadata_receive(sv[0], &ok); + EXPECT_NULL(received); + EXPECT_EQ_INT(ok, 1); + + close(sv[0]); + close(sv[1]); +} + static void test_metadata_rejects_invalid_values() { int p[2]; EXPECT_EQ_INT(pipe(p), 0); @@ -148,36 +259,30 @@ static void test_metadata_rejects_invalid_values() { } /* metadata_receive must reject an out-of-range atime/crtime nsec even when the - * flag would otherwise be valid (defense-in-depth on the -U/-N wire fields). */ + * flag would otherwise be valid (defense-in-depth on the -U/-N wire fields). + * The packed record is built by the shared codec so the out-of-range value + * actually reaches the wire. */ static void test_metadata_receive_rejects_bad_optional_times() { int p[2]; EXPECT_EQ_INT(pipe(p), 0); io_set_fds(p[0], p[1]); - int32_t present = 1; - int32_t mode = 0644; - int32_t uid = 1000; - int32_t gid = 1000; - int64_t mtime_sec = 1; - int64_t mtime_nsec = 0; - int32_t atime_valid = 1; - int64_t atime_sec = 1; - int64_t atime_nsec = 2000000000; /* invalid: >= 1e9 */ - EXPECT_TRUE(send_n_data(p[1], &present, sizeof(present))); - EXPECT_TRUE(send_n_data(p[1], &mode, sizeof(mode))); - EXPECT_TRUE(send_n_data(p[1], &uid, sizeof(uid))); - EXPECT_TRUE(send_n_data(p[1], &gid, sizeof(gid))); - EXPECT_TRUE(send_n_data(p[1], &mtime_sec, sizeof(mtime_sec))); - EXPECT_TRUE(send_n_data(p[1], &mtime_nsec, sizeof(mtime_nsec))); - EXPECT_TRUE(send_n_data(p[1], &atime_valid, sizeof(atime_valid))); - EXPECT_TRUE(send_n_data(p[1], &atime_sec, sizeof(atime_sec))); - EXPECT_TRUE(send_n_data(p[1], &atime_nsec, sizeof(atime_nsec))); - int32_t crtime_valid = 0; - int64_t crtime_sec = 0; - int64_t crtime_nsec = 0; - EXPECT_TRUE(send_n_data(p[1], &crtime_valid, sizeof(crtime_valid))); - EXPECT_TRUE(send_n_data(p[1], &crtime_sec, sizeof(crtime_sec))); - EXPECT_TRUE(send_n_data(p[1], &crtime_nsec, sizeof(crtime_nsec))); + FileMetadata bad = {.mode = 0644, + .uid = 1000, + .gid = 1000, + .mtime_sec = 1, + .mtime_nsec = 0, + .atime_valid = true, + .atime_sec = 1, + .atime_nsec = 2000000000, /* invalid: >= 1e9 */ + .crtime_valid = false, + .crtime_sec = 0, + .crtime_nsec = 0}; + char packed[sizeof(int32_t) + FILE_METADATA_WIRE_SIZE]; + char* write_ptr = packed; + metadata_to_buf(&write_ptr, &bad); + EXPECT_TRUE(send_n_data(p[1], packed, sizeof(packed))); + int ok = 1; EXPECT_NULL(metadata_receive(p[0], &ok)); EXPECT_EQ_INT(ok, 0); @@ -235,12 +340,13 @@ static void test_file_restore_metadata() { const char* content = "test content"; EXPECT_TRUE(file_write_to_disk(path, content, strlen(content), false, false)); - FileMetadata m; - m.mode = 0644; - m.uid = getuid(); - m.gid = getgid(); - m.mtime_sec = 1234567890; - m.mtime_nsec = 0; + FileMetadata m = {.mode = 0644, + .uid = getuid(), + .gid = getgid(), + .mtime_sec = 1234567890, + .mtime_nsec = 0, + .atime_valid = false, + .crtime_valid = false}; file_restore_metadata(path, &m, false); @@ -347,6 +453,8 @@ void test_metadata() { test_metadata_from_buf_null(); test_metadata_send_receive_roundtrip(); test_metadata_send_null(); + test_metadata_wire_is_one_packed_frame(); + test_metadata_send_receive_socketpair(); test_metadata_rejects_invalid_values(); test_metadata_receive_rejects_bad_optional_times(); test_metadata_mtime_window(); From 1a83e284c448bcb95ad2c8a4b08f7dc1f4b10528 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 04:16:28 +0200 Subject: [PATCH 022/155] perf(utils,file-list): index delete keep-set and --files-from lookups --- src/shared/file_list.c | 77 ++++++++++--- src/shared/file_list.h | 10 +- src/shared/utils.c | 229 +++++++++++++++++++++++++++++++++----- src/shared/utils.h | 35 ++++++ tests/runner.c | 2 + tests/test_file_list.c | 123 ++++++++++++++++++++ tests/test_file_list.h | 6 + tests/test_shared_utils.c | 38 +++++++ 8 files changed, 470 insertions(+), 50 deletions(-) create mode 100644 tests/test_file_list.c create mode 100644 tests/test_file_list.h diff --git a/src/shared/file_list.c b/src/shared/file_list.c index 528c8d4..b01a114 100644 --- a/src/shared/file_list.c +++ b/src/shared/file_list.c @@ -104,8 +104,37 @@ static int normalize_entry(const char* raw, size_t len, bool strip_line_endings, return result; } +/* Build the membership index: each non-empty entry plus every ancestor + directory prefix of it. The entry flag lets file_list_affects tell an exact + listed path from an ancestor of a listed path. An empty entry (the source + root) short-circuits every query, so it is recorded as whole_tree. */ +static bool file_list_index_build(FileListSet* set, char* err, size_t err_size) { + if (!str_hash_set_init(&set->node_index, (size_t)set->count * 2 + 1)) { + snprintf(err, err_size, "memory allocation failed"); + return false; + } + for (int i = 0; i < set->count; i++) { + const char* entry = set->entries[i]; + if (entry[0] == '\0') { + set->whole_tree = true; + continue; + } + if (!str_hash_set_insert_ref(&set->node_index, entry, true)) { + snprintf(err, err_size, "memory allocation failed"); + return false; + } + for (const char* slash = entry; (slash = strchr(slash, '/')) != NULL; slash++) { + if (!str_hash_set_insert_copy_n(&set->node_index, entry, (size_t)(slash - entry), false)) { + snprintf(err, err_size, "memory allocation failed"); + return false; + } + } + } + return true; +} + static FileListSet* string_list_to_set(StringList* raw, char* err, size_t err_size) { - FileListSet* set = malloc(sizeof(FileListSet)); + FileListSet* set = calloc(1, sizeof(FileListSet)); if (!set) { snprintf(err, err_size, "memory allocation failed"); return NULL; @@ -114,6 +143,10 @@ static FileListSet* string_list_to_set(StringList* raw, char* err, size_t err_si set->entries = raw->items; raw->items = NULL; raw->count = 0; + if (!file_list_index_build(set, err, err_size)) { + file_list_destroy(set); + return NULL; + } return set; } @@ -163,31 +196,39 @@ void file_list_destroy(FileListSet* set) { for (int i = 0; i < set->count; i++) free(set->entries[i]); free(set->entries); + str_hash_set_free(&set->node_index); free(set); } -static bool path_has_prefix(const char* path, const char* prefix) { - size_t plen = strlen(prefix); - if (strncmp(path, prefix, plen) != 0) - return false; - return path[plen] == '/' || path[plen] == '\0'; -} - bool file_list_affects(const FileListSet* set, const char* rel) { if (!set) return true; if (!rel) return false; - for (int i = 0; i < set->count; i++) { - const char* entry = set->entries[i]; - if (entry[0] == '\0') - return true; /* whole tree listed */ - if (strcmp(rel, entry) == 0) - return true; /* the entry itself is listed */ - if (path_has_prefix(rel, entry)) - return true; /* rel lives under a listed directory */ - if (path_has_prefix(entry, rel)) - return true; /* rel is an ancestor directory of a listed entry */ + if (set->whole_tree) + return true; /* whole tree listed */ + /* A node hit means `rel` is a listed entry, or an ancestor directory of one + (rel lives on the path to some listed entry). */ + if (str_hash_set_lookup(&set->node_index, rel, NULL)) + return true; + /* Otherwise `rel` is affected only when a listed entry is an ancestor of it; + walk rel's directory prefixes (which preserve path-boundary semantics) and + test each for an exact entry. */ + size_t len = strlen(rel); + while (len > 0) { + const char* slash = NULL; + for (size_t i = len; i-- > 0;) { + if (rel[i] == '/') { + slash = rel + i; + break; + } + } + if (!slash) + break; + len = (size_t)(slash - rel); + bool is_entry = false; + if (str_hash_set_lookup_n(&set->node_index, rel, len, &is_entry) && is_entry) + return true; } return false; } diff --git a/src/shared/file_list.h b/src/shared/file_list.h index 0ced332..53c1a79 100644 --- a/src/shared/file_list.h +++ b/src/shared/file_list.h @@ -1,6 +1,7 @@ #ifndef FILE_LIST_H #define FILE_LIST_H +#include "utils.h" #include #include @@ -12,11 +13,16 @@ * of "." means the whole tree, absolute entries and ".." traversal are * rejected at parse time. The set is immutable and shared read-only across * scanner worker threads. - */ - + * + * Membership is answered from `node_index`, built once at load time: it holds + * every entry plus every ancestor directory prefix of an entry, with the entry + * flag distinguishing an exact listed path from a mere ancestor. A lookup is + * O(path length) instead of O(entry count). */ typedef struct { char** entries; /* normalized rel paths; "" means the whole tree */ int count; + StrHashSet node_index; + bool whole_tree; /* an entry of "" lists the source root */ } FileListSet; /* Load and validate a --files-from file. When `null_separated` (-0/--from0) diff --git a/src/shared/utils.c b/src/shared/utils.c index c34d4a6..e15d55b 100644 --- a/src/shared/utils.c +++ b/src/shared/utils.c @@ -13,6 +13,7 @@ #include #include #include +#include static int authorized_root_fd = -1; static char* authorized_root_path; @@ -96,6 +97,153 @@ char* str_dup(const char* string) { return new_string; } +#define STR_HASH_SET_MIN_CAPACITY 16 + +static size_t str_hash_set_hash(const char* key, size_t len) { + return (size_t)XXH64(key, len, 0); +} + +/* Store an already-allocated key. Returns 1 when a new slot was filled and 0 + * for a duplicate (the caller keeps ownership of `key` when owned is true). */ +static int str_hash_set_put(StrHashSet* set, const char* key, size_t len, bool owned, + bool is_entry) { + size_t mask = set->capacity - 1; + size_t index = str_hash_set_hash(key, len) & mask; + while (true) { + StrHashSetSlot* slot = &set->slots[index]; + if (!slot->key) { + slot->key = key; + slot->owned = owned; + slot->is_entry = is_entry; + set->size++; + return 1; + } + if (strlen(slot->key) == len && memcmp(slot->key, key, len) == 0) { + if (is_entry) + slot->is_entry = true; + return 0; + } + index = (index + 1) & mask; + } +} + +static bool str_hash_set_resize(StrHashSet* set, size_t new_capacity) { + StrHashSetSlot* old_slots = set->slots; + size_t old_capacity = set->capacity; + StrHashSetSlot* slots = calloc(new_capacity, sizeof(StrHashSetSlot)); + if (!slots) + return false; + set->slots = slots; + set->capacity = new_capacity; + set->size = 0; + for (size_t i = 0; i < old_capacity; i++) { + if (old_slots[i].key) + (void)str_hash_set_put(set, old_slots[i].key, strlen(old_slots[i].key), old_slots[i].owned, + old_slots[i].is_entry); + } + free(old_slots); + return true; +} + +static bool str_hash_set_grow(StrHashSet* set) { + if (set->capacity != 0 && (set->size + 1) * 4 <= set->capacity * 3) + return true; + size_t new_capacity = set->capacity ? set->capacity * 2 : STR_HASH_SET_MIN_CAPACITY; + return str_hash_set_resize(set, new_capacity); +} + +bool str_hash_set_init(StrHashSet* set, size_t hint) { + if (!set) + return false; + set->slots = NULL; + set->capacity = 0; + set->size = 0; + size_t capacity = STR_HASH_SET_MIN_CAPACITY; + while (capacity < (hint + 1) * 2) + capacity *= 2; + set->slots = calloc(capacity, sizeof(StrHashSetSlot)); + if (!set->slots) + return false; + set->capacity = capacity; + return true; +} + +void str_hash_set_free(StrHashSet* set) { + if (!set) + return; + for (size_t i = 0; i < set->capacity; i++) { + if (set->slots[i].key && set->slots[i].owned) + free((void*)set->slots[i].key); + } + free(set->slots); + set->slots = NULL; + set->capacity = 0; + set->size = 0; +} + +bool str_hash_set_insert_ref(StrHashSet* set, const char* key, bool is_entry) { + if (!set || !key) + return false; + if (!str_hash_set_grow(set)) + return false; + return str_hash_set_put(set, key, strlen(key), false, is_entry) >= 0; +} + +bool str_hash_set_insert_copy_n(StrHashSet* set, const char* key, size_t len, bool is_entry) { + if (!set || !key) + return false; + bool present = false; + if (str_hash_set_lookup_n(set, key, len, &present)) { + if (is_entry) + (void)str_hash_set_put(set, key, len, false, true); /* upgrade in place */ + return true; + } + if (!str_hash_set_grow(set)) + return false; + char* copy = malloc(len + 1); + if (!copy) + return false; + memcpy(copy, key, len); + copy[len] = '\0'; + int result = str_hash_set_put(set, copy, len, true, is_entry); + if (result <= 0) { + free(copy); + return result == 0; + } + return true; +} + +static const StrHashSetSlot* str_hash_set_find_n(const StrHashSet* set, const char* key, + size_t len) { + if (!set || set->capacity == 0 || !key) + return NULL; + size_t mask = set->capacity - 1; + size_t index = str_hash_set_hash(key, len) & mask; + while (true) { + const StrHashSetSlot* slot = &set->slots[index]; + if (!slot->key) + return NULL; + if (strlen(slot->key) == len && memcmp(slot->key, key, len) == 0) + return slot; + index = (index + 1) & mask; + } +} + +bool str_hash_set_lookup_n(const StrHashSet* set, const char* key, size_t len, bool* is_entry) { + const StrHashSetSlot* slot = str_hash_set_find_n(set, key, len); + if (!slot) + return false; + if (is_entry) + *is_entry = slot->is_entry; + return true; +} + +bool str_hash_set_lookup(const StrHashSet* set, const char* key, bool* is_entry) { + if (!key) + return false; + return str_hash_set_lookup_n(set, key, strlen(key), is_entry); +} + char* output_escape(const char* string, bool eight_bit_output) { if (!string) return NULL; @@ -196,17 +344,40 @@ bool format_human_bytes(unsigned long long bytes, char* buffer, size_t buffer_si return written >= 0 && (size_t)written < buffer_size; } -static bool is_dir_in_manifest(const char* rel_path, ArrayList* manifest) { - size_t len = strlen(rel_path); +/* Build the keep-set index: every manifest entry is inserted as an exact entry + and every ancestor directory prefix of it as a non-entry node. A lookup of + `rel` therefore succeeds iff `rel` is a kept file, a kept directory, or an + ancestor directory of kept content (the old is_dir_in_manifest predicate); + the entry flag distinguishes an exact kept file from a mere prefix. */ +static bool build_keep_index(ArrayList* manifest, StrHashSet* index) { + if (!str_hash_set_init(index, manifest && manifest->size > 0 ? (size_t)manifest->size : 1)) + return false; + if (!manifest) + return true; for (int i = 0; i < manifest->size; i++) { const char* entry = (const char*)manifest->items[i]; - // Check if entry starts with rel_path + '/' or matches exactly - if (strncmp(entry, rel_path, len) == 0 && (entry[len] == '/' || entry[len] == '\0')) - return true; + if (!str_hash_set_insert_ref(index, entry, true)) + goto fail; + for (const char* slash = entry; (slash = strchr(slash, '/')) != NULL; slash++) { + if (!str_hash_set_insert_copy_n(index, entry, (size_t)(slash - entry), false)) + goto fail; + } } + return true; +fail: + str_hash_set_free(index); return false; } +static bool keep_is_dir(const StrHashSet* index, const char* rel_path) { + return str_hash_set_lookup(index, rel_path, NULL); +} + +static bool keep_is_file(const StrHashSet* index, const char* rel_path) { + bool is_entry = false; + return str_hash_set_lookup(index, rel_path, &is_entry) && is_entry; +} + /* True when child_rel is, or lies below, a protected entry. A prefix "a" therefore protects "a" and "a/b/c" but not "ab". Entries with top_level_only set only protect DIRECT children of the receive root (at_root); nested @@ -234,7 +405,7 @@ bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSki prefixes) mark the enclosing directory as surviving, exactly as they would make a real rmdir fail with ENOTEMPTY. Stops early once *count reaches the cap (sets *exceeds). Returns false on a traversal error. */ -static bool count_extras_fd(int dirfd, const char* rel_path, ArrayList* manifest, size_t cap, +static bool count_extras_fd(int dirfd, const char* rel_path, const StrHashSet* keep, size_t cap, size_t* count, bool* exceeds, const DeleteSkipEntry* skips, int skip_count, bool* survives) { /* openat(dirfd, ".") opens an independent file description: a dup() would @@ -284,15 +455,15 @@ static bool count_extras_fd(int dirfd, const char* rel_path, ArrayList* manifest bool child_ok = true; bool child_survives = true; if (childfd >= 0) { - child_ok = count_extras_fd(childfd, child_rel, manifest, cap, count, exceeds, skips, - skip_count, &child_survives); + child_ok = count_extras_fd(childfd, child_rel, keep, cap, count, exceeds, skips, skip_count, + &child_survives); close(childfd); } else if (errno != ENOENT) { operation_ok = false; } if (!child_ok) operation_ok = false; - if (is_dir_in_manifest(child_rel, manifest)) { + if (keep_is_dir(keep, child_rel)) { /* A directory with kept content below it is never removed. */ local_survives = true; } else if (child_survives) { @@ -308,13 +479,7 @@ static bool count_extras_fd(int dirfd, const char* rel_path, ArrayList* manifest } } } else { - bool found = false; - for (int i = 0; i < manifest->size; i++) { - if (strcmp((char*)manifest->items[i], child_rel) == 0) { - found = true; - break; - } - } + bool found = keep_is_file(keep, child_rel); if (!found) { if (*count >= cap) { *exceeds = true; @@ -330,7 +495,7 @@ static bool count_extras_fd(int dirfd, const char* rel_path, ArrayList* manifest return operation_ok; } -static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifest, +static bool delete_extras_fd(int dirfd, const char* rel_path, const StrHashSet* keep, size_t max_delete, size_t* deleted_count, const DeleteSkipEntry* skips, int skip_count) { /* Independent file description (see count_extras_fd). */ @@ -379,15 +544,15 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifes int childfd = openat(dirfd, entry->d_name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); bool child_removed = false; if (childfd >= 0) { - child_removed = delete_extras_fd(childfd, child_rel, manifest, max_delete, deleted_count, - skips, skip_count); + child_removed = delete_extras_fd(childfd, child_rel, keep, max_delete, deleted_count, skips, + skip_count); if (!child_removed) operation_ok = false; close(childfd); } else if (errno != ENOENT) { operation_ok = false; } - if (child_removed && !is_dir_in_manifest(child_rel, manifest)) { + if (child_removed && !keep_is_dir(keep, child_rel)) { if (*deleted_count >= max_delete) { operation_ok = false; } else { @@ -406,13 +571,7 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifes } } else { // Check if relative path is in manifest - bool found = false; - for (int i = 0; i < manifest->size; i++) { - if (strcmp((char*)manifest->items[i], child_rel) == 0) { - found = true; - break; - } - } + bool found = keep_is_file(keep, child_rel); if (!found) { if (*deleted_count >= max_delete) { operation_ok = false; @@ -443,6 +602,11 @@ DeleteWalkResult delete_extras_limited(const char* dest_root, ArrayList* manifes *deleted_out = 0; if (!manifest) return DELETE_WALK_ERROR; + /* Index the keep-set once so both passes answer membership in O(path length) + instead of scanning every manifest entry for every destination entry. */ + StrHashSet keep; + if (!build_keep_index(manifest, &keep)) + return DELETE_WALK_ERROR; int rootfd; if (authorized_root_fd >= 0) { if (authorized_root_path) @@ -454,29 +618,34 @@ DeleteWalkResult delete_extras_limited(const char* dest_root, ArrayList* manifes } else { rootfd = open(dest_root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); } - if (rootfd < 0) + if (rootfd < 0) { + str_hash_set_free(&keep); return DELETE_WALK_ERROR; + } if (max_delete != SIZE_MAX) { /* Rehearse the deletion first so a run that would exceed the cap removes nothing (rsync's all-or-nothing --max-delete contract). */ size_t count = 0; bool exceeds = false; bool survives = false; - bool counted_ok = count_extras_fd(rootfd, "", manifest, max_delete, &count, &exceeds, skips, + bool counted_ok = count_extras_fd(rootfd, "", &keep, max_delete, &count, &exceeds, skips, skip_count, &survives); if (!counted_ok) { close(rootfd); + str_hash_set_free(&keep); return DELETE_WALK_ERROR; } if (exceeds) { close(rootfd); + str_hash_set_free(&keep); return DELETE_WALK_LIMIT_EXCEEDED; } } size_t deleted_count = 0; - bool ok = delete_extras_fd(rootfd, "", manifest, max_delete, &deleted_count, skips, skip_count); + bool ok = delete_extras_fd(rootfd, "", &keep, max_delete, &deleted_count, skips, skip_count); if (close(rootfd) != 0) ok = false; + str_hash_set_free(&keep); if (deleted_out) *deleted_out = deleted_count; return ok ? DELETE_WALK_OK : DELETE_WALK_ERROR; diff --git a/src/shared/utils.h b/src/shared/utils.h index 4c9144a..f27c0c0 100644 --- a/src/shared/utils.h +++ b/src/shared/utils.h @@ -6,6 +6,41 @@ #include #include +/* Small open-addressing string hash set used to turn quadratic membership + * scans into O(path length) lookups (the --delete keep-set and the + * --files-from allow-set). Keys are hashed with xxHash64 (seed 0); collisions + * are resolved by linear probing over a power-of-two table that grows at 75% + * load. Slots may either borrow a caller-owned key (insert_ref) or own an + * internal copy (insert_copy_n); owned copies are released by + * str_hash_set_free. The set is not thread-safe for mutation, but a fully + * built set supports concurrent read-only lookups. */ +typedef struct { + const char* key; /* NULL marks an empty slot */ + bool owned; /* key is an internal copy that free() must release */ + bool is_entry; /* key was inserted as an exact entry, not just a prefix */ +} StrHashSetSlot; + +typedef struct { + StrHashSetSlot* slots; + size_t capacity; /* power of two, zero before init */ + size_t size; +} StrHashSet; + +/* Initialize an empty set sized for roughly `hint` entries. Returns false on + * allocation failure. */ +bool str_hash_set_init(StrHashSet* set, size_t hint); +void str_hash_set_free(StrHashSet* set); +/* Insert a borrowed key (must outlive the set). A duplicate only upgrades + * is_entry. Returns false on allocation failure. */ +bool str_hash_set_insert_ref(StrHashSet* set, const char* key, bool is_entry); +/* Insert a copy of the first `len` bytes of `key` (which need not be + * NUL-terminated). Returns false on allocation failure. */ +bool str_hash_set_insert_copy_n(StrHashSet* set, const char* key, size_t len, bool is_entry); +/* Look up a NUL-terminated key / a key of `len` bytes. On a hit, optionally + * reports whether the stored key was inserted as an exact entry. */ +bool str_hash_set_lookup(const StrHashSet* set, const char* key, bool* is_entry); +bool str_hash_set_lookup_n(const StrHashSet* set, const char* key, size_t len, bool* is_entry); + char* str_dup(const char* string); char* output_escape(const char* string, bool eight_bit_output); char* path_cat(const char* path1, const char* path2); diff --git a/tests/runner.c b/tests/runner.c index ec58de9..51938e1 100644 --- a/tests/runner.c +++ b/tests/runner.c @@ -12,6 +12,7 @@ #include "test_delay_updates.h" #include "test_delta.h" #include "test_file.h" +#include "test_file_list.h" #include "test_file_sendfile.h" #include "test_fuzz_smoke.h" #include "test_glob.h" @@ -67,6 +68,7 @@ int main() { RUN_TEST(test_glob); RUN_TEST(test_iconv); RUN_TEST(test_file); + RUN_TEST(test_file_list); RUN_TEST(test_trust_sender); RUN_TEST(test_delay_updates); RUN_TEST(test_file_sendfile); diff --git a/tests/test_file_list.c b/tests/test_file_list.c new file mode 100644 index 0000000..bf31db5 --- /dev/null +++ b/tests/test_file_list.c @@ -0,0 +1,123 @@ +#include "test_file_list.h" +#include "file_list.h" +#include "test_utils.h" +#include +#include +#include + +/* Reference implementation of the ORIGINAL file_list_affects linear scan. The + indexed implementation must agree with it on every query; this pins the + subtle semantics: empty entry == whole tree, exact match, rel under a listed + directory, and rel an ancestor directory of a listed entry. */ +static bool reference_affects(const FileListSet* set, const char* rel) { + if (!set) + return true; + if (!rel) + return false; + for (int i = 0; i < set->count; i++) { + const char* entry = set->entries[i]; + if (entry[0] == '\0') + return true; + if (strcmp(rel, entry) == 0) + return true; + size_t entry_len = strlen(entry); + if (strncmp(rel, entry, entry_len) == 0 && (rel[entry_len] == '/' || rel[entry_len] == '\0')) + return true; + size_t rel_len = strlen(rel); + if (strncmp(entry, rel, rel_len) == 0 && (entry[rel_len] == '/' || entry[rel_len] == '\0')) + return true; + } + return false; +} + +static void write_list(const char* path, const char* bytes) { + FILE* fp = fopen(path, "wb"); + EXPECT_NOT_NULL(fp); + size_t len = strlen(bytes); + EXPECT_EQ_INT((int)fwrite(bytes, 1, len, fp), (int)len); + fclose(fp); +} + +static void check_queries(const FileListSet* set, const char* const* queries, int query_count) { + for (int i = 0; i < query_count; i++) { + bool expected = reference_affects(set, queries[i]); + bool actual = file_list_affects(set, queries[i]); + if (expected != actual) { + printf(" [FAIL] affects(\"%s\"): expected %d, got %d\n", queries[i], expected, actual); + current_test_failed = true; + return; + } + } +} + +static void test_membership_matches_reference() { + const char* path = "test_file_list_case.txt"; + char err[160]; + + /* Nested directories, an ancestor of a listed entry, an exact file, a + non-matching neighbor with the same prefix, and a literal '*'. */ + write_list(path, "a\na/b\na/b/c\nab\nc.txt\nsub/b.bin\n*\n"); + FileListSet* set = file_list_load(path, false, err, sizeof(err)); + EXPECT_NOT_NULL(set); + const char* queries[] = { + "a", "a/b", "a/b/c", "a/b/c/d", "a/bx", "a/x", "ab", + "abc", "c.txt", "c.txt/x", "c", "sub", "sub/b.bin", "sub/b.bin/z", + "sub2", "*", "x", "", "a/b/cd", "/", "a/b/", + }; + check_queries(set, queries, (int)(sizeof(queries) / sizeof(queries[0]))); + EXPECT_TRUE(file_list_affects(set, "a/b/c/d")); + EXPECT_TRUE(file_list_affects(set, "a/bx")); /* under listed directory "a" */ + EXPECT_TRUE(file_list_affects(set, "a/b/c")); + EXPECT_TRUE(file_list_affects(set, "a/x")); /* under listed directory "a" */ + EXPECT_FALSE(file_list_affects(set, "abc")); /* component boundary: not "a" */ + EXPECT_TRUE(file_list_affects(set, "sub")); + EXPECT_FALSE(file_list_affects(set, "sub2")); + file_list_destroy(set); + remove(path); + + /* A single "." entry means the whole tree: every non-NULL query is true. */ + write_list(path, ".\n"); + set = file_list_load(path, false, err, sizeof(err)); + EXPECT_NOT_NULL(set); + const char* root_queries[] = {"", "a", "a/b/c", "unrelated", "*", "/"}; + for (int i = 0; i < (int)(sizeof(root_queries) / sizeof(root_queries[0])); i++) + EXPECT_TRUE(file_list_affects(set, root_queries[i])); + EXPECT_FALSE(file_list_affects(set, NULL)); + check_queries(set, root_queries, (int)(sizeof(root_queries) / sizeof(root_queries[0]))); + file_list_destroy(set); + remove(path); + + /* Trailing slashes and "./" prefixes normalize away, so the query matches the + clean path (and not the raw spelling). */ + write_list(path, "./dir/\ndir2/./x\n"); + set = file_list_load(path, false, err, sizeof(err)); + EXPECT_NOT_NULL(set); + EXPECT_TRUE(file_list_affects(set, "dir")); + EXPECT_TRUE(file_list_affects(set, "dir/x")); + EXPECT_TRUE(file_list_affects(set, "dir2/x")); + EXPECT_TRUE(file_list_affects(set, "dir2")); + EXPECT_TRUE(file_list_affects(set, "dir/")); /* boundary prefix of listed "dir" */ + check_queries(set, (const char*[]){"dir", "dir/", "dir/x", "dir2", "dir2/x", "dir3"}, 6); + file_list_destroy(set); + remove(path); + + /* An empty file yields an empty set: nothing is affected, and NULL set still + means "everything". */ + write_list(path, ""); + set = file_list_load(path, false, err, sizeof(err)); + EXPECT_NOT_NULL(set); + EXPECT_EQ_INT(set->count, 0); + EXPECT_FALSE(file_list_affects(set, "a")); + EXPECT_FALSE(file_list_affects(set, "")); + check_queries(set, (const char*[]){"a", "a/b", ""}, 3); + file_list_destroy(set); + remove(path); + + /* NULL set is the unrestricted case. */ + EXPECT_TRUE(file_list_affects(NULL, "anything")); + EXPECT_TRUE(file_list_affects(NULL, NULL)); +} + +void test_file_list() { + test_membership_matches_reference(); +} diff --git a/tests/test_file_list.h b/tests/test_file_list.h new file mode 100644 index 0000000..4a8d6d8 --- /dev/null +++ b/tests/test_file_list.h @@ -0,0 +1,6 @@ +#ifndef TEST_FILE_LIST_H +#define TEST_FILE_LIST_H + +void test_file_list(); + +#endif diff --git a/tests/test_shared_utils.c b/tests/test_shared_utils.c index 86ce24b..6663695 100644 --- a/tests/test_shared_utils.c +++ b/tests/test_shared_utils.c @@ -150,6 +150,43 @@ static void test_walker_removes_extras_keeps_manifest_and_protected() { free(root); } +static void test_walker_keeps_nested_manifest_dirs() { + /* The keep-set index must preserve deep content: a directory is protected + when its own name is a keep entry OR when kept content lives below it, and + an exact kept file survives while its siblings are removed. */ + char* root = make_walk_root("nestedkeep"); + EXPECT_NOT_NULL(root); + EXPECT_TRUE(write_file_at(root, "extra.txt", "extra")); + EXPECT_EQ_INT(make_subdir(root, "keepdir"), 0); + EXPECT_EQ_INT(make_subdir(root, "keepdir/deep"), 0); + EXPECT_TRUE(write_file_at(root, "keepdir/deep/keep.txt", "kept")); + EXPECT_TRUE(write_file_at(root, "keepdir/extra2.txt", "extra")); + EXPECT_EQ_INT(make_subdir(root, "dropdir"), 0); + EXPECT_EQ_INT(make_subdir(root, "keep2"), 0); + EXPECT_TRUE(write_file_at(root, "keep2/inner.txt", "kept")); + EXPECT_EQ_INT(make_subdir(root, "keep3"), 0); + + const char* keeps[] = {"keepdir/deep/keep.txt", "keep2/inner.txt", "keep3"}; + ArrayList* manifest = make_manifest_strings(keeps, 3); + EXPECT_NOT_NULL(manifest); + size_t deleted = 0; + DeleteWalkResult result = delete_extras_limited(root, manifest, 100000, NULL, 0, &deleted); + EXPECT_EQ_INT((int)result, (int)DELETE_WALK_OK); + EXPECT_FALSE(file_exists(root, "extra.txt")); + EXPECT_TRUE(file_exists(root, "keepdir/deep/keep.txt")); + EXPECT_FALSE(file_exists(root, "keepdir/extra2.txt")); + EXPECT_TRUE(dir_exists(root, "keepdir")); + EXPECT_TRUE(dir_exists(root, "keepdir/deep")); + EXPECT_FALSE(dir_exists(root, "dropdir")); + EXPECT_TRUE(dir_exists(root, "keep2")); + EXPECT_TRUE(file_exists(root, "keep2/inner.txt")); + EXPECT_TRUE(dir_exists(root, "keep3")); /* an exact directory keep entry survives */ + EXPECT_EQ_INT((int)deleted, 3); + array_list_delete(manifest); + remove_walk_tree(root); + free(root); +} + static void test_walker_max_delete_exceeded_deletes_nothing() { char* root = make_walk_root("maxdel"); EXPECT_NOT_NULL(root); @@ -421,6 +458,7 @@ static void test_fd_peer_ip() { void test_shared_utils() { test_walker_removes_extras_keeps_manifest_and_protected(); + test_walker_keeps_nested_manifest_dirs(); test_walker_max_delete_exceeded_deletes_nothing(); test_walker_max_delete_exact_bound_deletes(); test_walker_unlimited_deletes_all(); From 6c636a19e64f2ec8032dabb4be910b09ba5ba3b4 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 04:18:55 +0200 Subject: [PATCH 023/155] perf(compression,tcp): reuse zstd contexts; enable TCP_NODELAY --- src/shared/compression.c | 227 +++++++++++++++++++++++++++++-------- src/shared/compression.h | 8 ++ src/shared/transport_tcp.c | 18 +++ tests/test_compression.c | 75 ++++++++++++ tests/test_transport_tcp.c | 45 ++++++++ 5 files changed, 328 insertions(+), 45 deletions(-) diff --git a/src/shared/compression.c b/src/shared/compression.c index e4d5b95..d9aa037 100644 --- a/src/shared/compression.c +++ b/src/shared/compression.c @@ -2,11 +2,12 @@ #include "data.h" #include "log.h" #include "protocol.h" -#include #include #include +#include #include #include +#include #include #include @@ -39,6 +40,96 @@ bool compression_should_skip_with_suffixes(const char* path, char* const* suffix return false; } +/* Per-thread cache of zstd contexts plus the grow-only compression scratch + * buffer. zstd contexts are stateful and not safe to share between threads, + * so each thread keeps its own (see compression_get_thread_ctx). The cache is + * stored in a C11 thread-specific storage slot whose destructor releases the + * contexts when the thread exits; this keeps LeakSanitizer clean for the + * short-lived sender/receiver/scanner worker threads without every worker + * entry point having to remember to call compression_free_thread_contexts(). + * The main thread's slot is not torn down by tss at process exit, so an atexit + * hook releases it (and compression_free_thread_contexts allows eager + * release). */ +typedef struct { + ZSTD_CCtx* cctx; + ZSTD_DCtx* dctx; + void* out_buf; /* reusable ZSTD_compressBound-sized output scratch */ + size_t out_cap; /* bytes currently allocated for out_buf */ + int level; /* compression level currently applied to cctx */ + int workers; /* nbWorkers currently applied to cctx */ + bool params_set; + bool cached; /* false when the TSS slot could not be used: caller owns */ +} CompressionThreadCtx; + +static once_flag compression_tls_once = ONCE_FLAG_INIT; +static tss_t compression_tls_key; +static bool compression_tls_ready; + +static void compression_tls_make_key(void); + +static void compression_ctx_free(CompressionThreadCtx* ctx) { + if (!ctx) + return; + if (ctx->cctx) + ZSTD_freeCCtx(ctx->cctx); + if (ctx->dctx) + ZSTD_freeDCtx(ctx->dctx); + free(ctx->out_buf); + free(ctx); +} + +static void compression_tls_destructor(void* value) { + compression_ctx_free((CompressionThreadCtx*)value); +} + +void compression_free_thread_contexts(void) { + call_once(&compression_tls_once, compression_tls_make_key); + if (!compression_tls_ready) + return; + CompressionThreadCtx* ctx = (CompressionThreadCtx*)tss_get(compression_tls_key); + if (!ctx) + return; + /* Clear the slot first so the thread-exit destructor cannot free it twice. */ + tss_set(compression_tls_key, NULL); + compression_ctx_free(ctx); +} + +static void compression_atexit_cleanup(void) { + compression_free_thread_contexts(); +} + +static void compression_tls_make_key(void) { + if (tss_create(&compression_tls_key, compression_tls_destructor) == thrd_success) { + compression_tls_ready = true; + atexit(compression_atexit_cleanup); + } +} + +static CompressionThreadCtx* compression_get_thread_ctx(void) { + call_once(&compression_tls_once, compression_tls_make_key); + if (!compression_tls_ready) { + /* Extremely unlikely: fall back to an uncached context the caller frees. */ + return (CompressionThreadCtx*)calloc(1, sizeof(CompressionThreadCtx)); + } + CompressionThreadCtx* ctx = (CompressionThreadCtx*)tss_get(compression_tls_key); + if (ctx) + return ctx; + ctx = (CompressionThreadCtx*)calloc(1, sizeof(CompressionThreadCtx)); + if (!ctx) + return NULL; + ctx->cached = true; + if (tss_set(compression_tls_key, ctx) != thrd_success) + ctx->cached = false; + return ctx; +} + +/* Release an uncached context immediately; cached contexts are owned by the + * thread's TSS slot and freed on thread exit / compression_free_thread_contexts. */ +static void compression_ctx_put(CompressionThreadCtx* ctx) { + if (ctx && !ctx->cached) + compression_ctx_free(ctx); +} + Data* data_compress(Data* data_to_compress, int compression_level) { return data_compress_with_threads(data_to_compress, compression_level, 0); } @@ -50,68 +141,102 @@ Data* data_compress_with_threads(Data* data_to_compress, int compression_level, return NULL; log_message(LOG_LEVEL_DEBUG, "Starting to compress data"); size_t dst_size = ZSTD_compressBound(data_to_compress->size); - Data* compressed_data = data_create_empty(dst_size); - if (compressed_data == NULL) - return NULL; - ZSTD_CCtx* cctx = ZSTD_createCCtx(); - if (!cctx) { - log_message(LOG_LEVEL_ERROR, "Failed to create ZSTD compression context"); - data_destroy(compressed_data); + CompressionThreadCtx* ctx = compression_get_thread_ctx(); + if (ctx == NULL) { + log_message(LOG_LEVEL_ERROR, "Failed to allocate ZSTD compression context"); return NULL; } + Data* compressed_data = NULL; - size_t zret = ZSTD_CCtx_setParameter(cctx, ZSTD_c_compressionLevel, compression_level); - if (ZSTD_isError(zret)) { - log_message(LOG_LEVEL_ERROR, "Failed to set compression level: %s", ZSTD_getErrorName(zret)); - ZSTD_freeCCtx(cctx); - data_destroy(compressed_data); - return NULL; + if (!ctx->cctx) { + ctx->cctx = ZSTD_createCCtx(); + if (!ctx->cctx) { + log_message(LOG_LEVEL_ERROR, "Failed to create ZSTD compression context"); + goto cleanup; + } + ctx->params_set = false; } + /* Reset only the session: parameters (and any already-allocated zstd worker + * pool) stay attached to the context, so compressing the next file does not + * rebuild the pool. */ + ZSTD_CCtx_reset(ctx->cctx, ZSTD_reset_session_only); + + if (!ctx->params_set || ctx->level != compression_level) { + size_t zret = ZSTD_CCtx_setParameter(ctx->cctx, ZSTD_c_compressionLevel, compression_level); + if (ZSTD_isError(zret)) { + log_message(LOG_LEVEL_ERROR, "Failed to set compression level: %s", ZSTD_getErrorName(zret)); + goto cleanup; + } + ctx->level = compression_level; + } + + int available_threads = 0; if (compression_threads > 0) { long online_cpus = sysconf(_SC_NPROCESSORS_ONLN); - int available_threads = online_cpus > 0 && online_cpus < compression_threads - ? (int)online_cpus - : compression_threads; - zret = ZSTD_CCtx_setParameter(cctx, ZSTD_c_nbWorkers, available_threads); + available_threads = online_cpus > 0 && online_cpus < compression_threads ? (int)online_cpus + : compression_threads; + } + if (!ctx->params_set || ctx->workers != available_threads) { + size_t zret = ZSTD_CCtx_setParameter(ctx->cctx, ZSTD_c_nbWorkers, available_threads); if (ZSTD_isError(zret)) { log_message(LOG_LEVEL_ERROR, "Failed to set compression threads: %s", ZSTD_getErrorName(zret)); - ZSTD_freeCCtx(cctx); - data_destroy(compressed_data); - return NULL; + goto cleanup; } + ctx->workers = available_threads; + } + ctx->params_set = true; + + if (available_threads > 0) { /* Streaming compression needs the source size before threaded mode can end a frame. */ - zret = ZSTD_CCtx_setPledgedSrcSize(cctx, data_to_compress->size); + size_t zret = ZSTD_CCtx_setPledgedSrcSize(ctx->cctx, data_to_compress->size); if (ZSTD_isError(zret)) { log_message(LOG_LEVEL_ERROR, "Failed to set compression source size: %s", ZSTD_getErrorName(zret)); - ZSTD_freeCCtx(cctx); - data_destroy(compressed_data); - return NULL; + goto cleanup; } } + if (ctx->out_cap < dst_size) { + void* grown = protocol_realloc(ctx->out_buf, dst_size); + if (grown == NULL) { + log_message(LOG_LEVEL_ERROR, "Failed to allocate compression buffer"); + goto cleanup; + } + ctx->out_buf = grown; + ctx->out_cap = dst_size; + } + ZSTD_inBuffer input = {data_to_compress->data, data_to_compress->size, 0}; - ZSTD_outBuffer output = {compressed_data->data, dst_size, 0}; + ZSTD_outBuffer output = {ctx->out_buf, dst_size, 0}; size_t ret; do { - ret = ZSTD_compressStream2(cctx, &output, &input, ZSTD_e_end); + ret = ZSTD_compressStream2(ctx->cctx, &output, &input, ZSTD_e_end); if (ZSTD_isError(ret)) { log_message(LOG_LEVEL_ERROR, "Compression failed: %s", ZSTD_getErrorName(ret)); - ZSTD_freeCCtx(cctx); - data_destroy(compressed_data); - return NULL; + goto cleanup; } } while (ret > 0); + /* Hand off an exactly-sized copy; the scratch buffer stays cached so the next + * call does not reallocate a ZSTD_compressBound-sized block. */ + compressed_data = data_create_empty(output.pos); + if (compressed_data == NULL) { + log_message(LOG_LEVEL_ERROR, "Failed to allocate compressed data"); + goto cleanup; + } + if (output.pos > 0) + memcpy(compressed_data->data, ctx->out_buf, output.pos); compressed_data->size = output.pos; - ZSTD_freeCCtx(cctx); log_debug_message(LOG_DEBUG_UTIL, "Data succesfully compressed from %zu to %zu", data_to_compress->size, compressed_data->size); + +cleanup: + compression_ctx_put(ctx); return compressed_data; } @@ -144,20 +269,30 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) { return NULL; } - ZSTD_DCtx* dctx = ZSTD_createDCtx(); - if (!dctx) { - log_message(LOG_LEVEL_ERROR, "Failed to create ZSTD decompression context"); + CompressionThreadCtx* ctx = compression_get_thread_ctx(); + if (ctx == NULL) { + log_message(LOG_LEVEL_ERROR, "Failed to allocate ZSTD decompression context"); return NULL; } + Data* uncompressed_data = NULL; + + if (!ctx->dctx) { + ctx->dctx = ZSTD_createDCtx(); + if (!ctx->dctx) { + log_message(LOG_LEVEL_ERROR, "Failed to create ZSTD decompression context"); + goto cleanup; + } + } + /* Reset only the session; decompression parameters are sticky. */ + ZSTD_DCtx_reset(ctx->dctx, ZSTD_reset_session_only); size_t buf_size = (dst_size > 0) ? (size_t)dst_size : INITIAL_DECOMPRESS_BUF_SIZE; if (buf_size > maximum_size) buf_size = maximum_size; - Data* uncompressed_data = data_create_empty(buf_size); + uncompressed_data = data_create_empty(buf_size); if (!uncompressed_data) { log_message(LOG_LEVEL_ERROR, "Failed to allocate decompression buffer"); - ZSTD_freeDCtx(dctx); - return NULL; + goto cleanup; } ZSTD_inBuffer input = {compressed_data->data, compressed_data->size, 0}; @@ -165,20 +300,20 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) { size_t ret; do { - ret = ZSTD_decompressStream(dctx, &output, &input); + ret = ZSTD_decompressStream(ctx->dctx, &output, &input); if (ZSTD_isError(ret)) { log_message(LOG_LEVEL_ERROR, "Decompression failed: %s", ZSTD_getErrorName(ret)); - ZSTD_freeDCtx(dctx); data_destroy(uncompressed_data); - return NULL; + uncompressed_data = NULL; + goto cleanup; } if (ret > 0 && output.pos == output.size) { if (buf_size >= hard_limit || buf_size > SIZE_MAX / 2) { log_message(LOG_LEVEL_ERROR, "Decompressed data exceeds %llu bytes", (unsigned long long)MAX_DECOMPRESSED_SIZE); - ZSTD_freeDCtx(dctx); data_destroy(uncompressed_data); - return NULL; + uncompressed_data = NULL; + goto cleanup; } buf_size *= 2; if (buf_size > hard_limit) @@ -186,9 +321,9 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) { void* new_data = protocol_realloc(uncompressed_data->data, buf_size); if (!new_data) { log_message(LOG_LEVEL_ERROR, "Failed to grow decompression buffer"); - ZSTD_freeDCtx(dctx); data_destroy(uncompressed_data); - return NULL; + uncompressed_data = NULL; + goto cleanup; } uncompressed_data->data = new_data; output.dst = new_data; @@ -197,9 +332,11 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) { } while (ret > 0); uncompressed_data->size = output.pos; - ZSTD_freeDCtx(dctx); log_debug_message(LOG_DEBUG_UTIL, "Decompressed data successfully"); + +cleanup: + compression_ctx_put(ctx); return uncompressed_data; } diff --git a/src/shared/compression.h b/src/shared/compression.h index b30622d..179c2b7 100644 --- a/src/shared/compression.h +++ b/src/shared/compression.h @@ -14,4 +14,12 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size); bool compression_should_skip(const char* path); bool compression_should_skip_with_suffixes(const char* path, char* const* suffixes, int count); +/* Release the calling thread's cached zstd contexts (compressor, decompressor + * and scratch buffer). The cache is thread-local and is also released + * automatically when a worker thread exits (via a C11 tss destructor) and for + * the main thread at process exit; this explicit entry point exists so tests + * and long-lived callers can drop the cache deterministically. Safe to call + * when no context has been created, and idempotent. */ +void compression_free_thread_contexts(void); + #endif diff --git a/src/shared/transport_tcp.c b/src/shared/transport_tcp.c index a3f9a07..42dad4c 100644 --- a/src/shared/transport_tcp.c +++ b/src/shared/transport_tcp.c @@ -19,6 +19,7 @@ static volatile sig_atomic_t g_active_connections = 0; static void tcp_apply_socket_timeout(int fd); +static void tcp_enable_nodelay_default(int fd, int family); static void sigchld_handler(int sig) { (void)sig; @@ -148,6 +149,7 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil continue; } tcp_apply_socket_timeout(fd); + tcp_enable_nodelay_default(fd, client_addr.ss_family); char peer[128]; if (!utils_sockaddr_to_string((const struct sockaddr*)&client_addr, peer, sizeof(peer))) snprintf(peer, sizeof(peer), "unknown"); @@ -231,6 +233,19 @@ static void tcp_apply_socket_timeout(int fd) { setsockopt(fd, SOL_SOCKET, SO_SNDTIMEO, &tv, sizeof(tv)); } +/* Enable TCP_NODELAY by default on a transfer socket: the protocol emits many + * small messages and Nagle's algorithm would otherwise coalesce/delay them. + * Best-effort only: the family guard keeps this to IP/TCP sockets, and a + * setsockopt failure is ignored. A caller-provided --sockopts TCP_NODELAY=0 + * is applied afterwards on the connect path, so an explicit user choice still + * wins. */ +static void tcp_enable_nodelay_default(int fd, int family) { + if (family != AF_INET && family != AF_INET6) + return; + int value = 1; + setsockopt(fd, IPPROTO_TCP, TCP_NODELAY, &value, sizeof(value)); +} + Client* client_create() { Client* client = (Client*)malloc(sizeof(Client)); if (client == NULL) { @@ -376,6 +391,9 @@ bool tcp_connect_socket_ex(Client* client, const char* host, int port, if (client->file_descriptor < 0) continue; + /* Default first; a user --sockopts TCP_NODELAY=0 applied below overrides. */ + tcp_enable_nodelay_default(client->file_descriptor, rp->ai_family); + if (opts && opts->sockopt_count > 0 && !tcp_apply_sockopts(client->file_descriptor, opts->sockopts, opts->sockopt_count)) { close(client->file_descriptor); diff --git a/tests/test_compression.c b/tests/test_compression.c index 8784046..da51531 100644 --- a/tests/test_compression.c +++ b/tests/test_compression.c @@ -6,6 +6,7 @@ #include "utils.h" #include #include +#include #include static void test_data_compress_decompress_roundtrip() { @@ -137,10 +138,84 @@ static void test_chunk_compress_decompress_roundtrip() { unlink(path2); } +typedef struct { + int id; + int iterations; + bool ok; +} CompressionThreadArg; + +/* Each worker exercises the per-thread cached zstd contexts: several + * compress/decompress round-trips with varying payload sizes, levels and + * worker counts so the context is reused (and its parameters re-applied) + * across calls, concurrently with other workers. */ +static int compression_reuse_worker(void* arg) { + CompressionThreadArg* a = (CompressionThreadArg*)arg; + a->ok = true; + for (int it = 0; it < a->iterations; it++) { + size_t size = 512 + (size_t)((a->id * 7919 + it * 104729) % (48 * 1024)); + char* original = malloc(size); + if (!original) { + a->ok = false; + break; + } + for (size_t i = 0; i < size; i++) + original[i] = (char)((i * 31 + (size_t)a->id + (size_t)it * 7) % 251); + Data* input = data_create(original, size); + if (!input) { /* data_create takes ownership of original, even on failure */ + a->ok = false; + break; + } + int level = 1 + ((it / 2) % 5); + int threads = ((it / 2) % 2 == 0) ? 2 : 0; + Data* compressed = data_compress_with_threads(input, level, threads); + if (!compressed) { + data_destroy(input); + a->ok = false; + break; + } + Data* decompressed = data_decompress(compressed); + bool roundtrip_ok = decompressed != NULL && decompressed->size == size && + memcmp(decompressed->data, original, size) == 0; + data_destroy(decompressed); + data_destroy(compressed); + data_destroy(input); + if (!roundtrip_ok) { + a->ok = false; + break; + } + } + /* Deliberately do NOT free the thread context here: the C11 tss destructor + * must release it when this thread exits (validated by LeakSanitizer). */ + return thrd_success; +} + +static void test_data_compress_reused_contexts_multithreaded() { + enum { NTHREADS = 8, ITERATIONS = 6 }; + thrd_t threads[NTHREADS]; + CompressionThreadArg args[NTHREADS]; + bool all_created = true; + for (int i = 0; i < NTHREADS; i++) { + args[i].id = i; + args[i].iterations = ITERATIONS; + args[i].ok = false; + if (thrd_create(&threads[i], compression_reuse_worker, &args[i]) != thrd_success) { + all_created = false; + break; + } + } + EXPECT_TRUE(all_created); + for (int i = 0; i < NTHREADS; i++) + EXPECT_EQ_INT(thrd_join(threads[i], NULL), thrd_success); + for (int i = 0; i < NTHREADS; i++) + EXPECT_TRUE(args[i].ok); + compression_free_thread_contexts(); +} + void test_compression() { test_data_compress_decompress_roundtrip(); test_data_compress_decompress_large(); test_skip_compress_suffix_matching(); test_data_compress_with_threads_roundtrip(); + test_data_compress_reused_contexts_multithreaded(); test_chunk_compress_decompress_roundtrip(); } diff --git a/tests/test_transport_tcp.c b/tests/test_transport_tcp.c index b4b7c89..a3043cf 100644 --- a/tests/test_transport_tcp.c +++ b/tests/test_transport_tcp.c @@ -2,6 +2,7 @@ #include "protocol.h" #include "test_utils.h" #include "transport_tcp.h" +#include #include #include #include @@ -195,6 +196,49 @@ static void test_client_disconnect_delete() { client_delete(c); } +/* TCP_NODELAY is enabled by default on a connected transfer socket, and an + * explicit --sockopts TCP_NODELAY=0 still overrides it. */ +static void test_tcp_nodelay_default_and_override() { + Server* s = server_create(0); + EXPECT_NOT_NULL(s); + EXPECT_EQ_INT(listen(s->file_descriptor, 1), 0); + struct sockaddr_in bound; + socklen_t bound_len = sizeof(bound); + EXPECT_EQ_INT(getsockname(s->file_descriptor, (struct sockaddr*)&bound, &bound_len), 0); + int port = (int)ntohs(bound.sin_port); + EXPECT_TRUE(port > 0); + + Client* c = client_create(); + EXPECT_NOT_NULL(c); + EXPECT_TRUE(client_connect(c, "127.0.0.1", port)); + int got = 0; + socklen_t len = sizeof(got); + EXPECT_EQ_INT(getsockopt(c->file_descriptor, IPPROTO_TCP, TCP_NODELAY, &got, &len), 0); + EXPECT_EQ_INT(got, 1); + client_disconnect(c); + client_delete(c); + + SockOptEntry* entries = NULL; + int count = 0; + EXPECT_EQ_INT(config_sockopts_parse("TCP_NODELAY=0", &entries, &count), 0); + TcpConnectOptions opts; + memset(&opts, 0, sizeof(opts)); + opts.sockopts = entries; + opts.sockopt_count = count; + + Client* c2 = client_create(); + EXPECT_NOT_NULL(c2); + EXPECT_TRUE(client_connect_ex(c2, "127.0.0.1", port, &opts)); + got = 0; + len = sizeof(got); + EXPECT_EQ_INT(getsockopt(c2->file_descriptor, IPPROTO_TCP, TCP_NODELAY, &got, &len), 0); + EXPECT_EQ_INT(got, 0); + client_disconnect(c2); + client_delete(c2); + free(entries); + server_delete(&s); +} + void test_transport_tcp() { test_server_create_ephemeral(); test_server_delete_null(); @@ -211,4 +255,5 @@ void test_transport_tcp() { test_sockopts_apply_sets_option(); test_server_create_bind_address(); test_server_create_bind_ipv6(); + test_tcp_nodelay_default_and_override(); } From 301cb0dbaf148129d265b2c41a601a61b513888a Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 04:52:10 +0200 Subject: [PATCH 024/155] fix(utils,file-list): bound keep/files-from indexes to O(M) memory --- src/shared/file_list.c | 46 +++---- src/shared/file_list.h | 12 +- src/shared/utils.c | 247 ++++++++++++++++++++++++-------------- src/shared/utils.h | 74 +++++++++--- tests/test_file_list.c | 73 +++++++++++ tests/test_shared_utils.c | 65 ++++++++++ 6 files changed, 377 insertions(+), 140 deletions(-) diff --git a/src/shared/file_list.c b/src/shared/file_list.c index b01a114..3297a87 100644 --- a/src/shared/file_list.c +++ b/src/shared/file_list.c @@ -104,30 +104,20 @@ static int normalize_entry(const char* raw, size_t len, bool strip_line_endings, return result; } -/* Build the membership index: each non-empty entry plus every ancestor - directory prefix of it. The entry flag lets file_list_affects tell an exact - listed path from an ancestor of a listed path. An empty entry (the source - root) short-circuits every query, so it is recorded as whole_tree. */ +/* Build the membership index over the exact entries only. `file_list_affects` + combines the exact/descendant lookups with a walk of the query's own ancestor + prefixes, so no ancestor prefix is ever materialized as a copy and the index + stays O(entry count) memory regardless of path depth. An empty entry (the + source root) sets whole_tree and short-circuits every query. */ static bool file_list_index_build(FileListSet* set, char* err, size_t err_size) { - if (!str_hash_set_init(&set->node_index, (size_t)set->count * 2 + 1)) { + if (!path_index_build(&set->index, (const char* const*)set->entries, (size_t)set->count)) { snprintf(err, err_size, "memory allocation failed"); return false; } for (int i = 0; i < set->count; i++) { - const char* entry = set->entries[i]; - if (entry[0] == '\0') { + if (set->entries[i][0] == '\0') { set->whole_tree = true; - continue; - } - if (!str_hash_set_insert_ref(&set->node_index, entry, true)) { - snprintf(err, err_size, "memory allocation failed"); - return false; - } - for (const char* slash = entry; (slash = strchr(slash, '/')) != NULL; slash++) { - if (!str_hash_set_insert_copy_n(&set->node_index, entry, (size_t)(slash - entry), false)) { - snprintf(err, err_size, "memory allocation failed"); - return false; - } + break; } } return true; @@ -193,10 +183,10 @@ FileListSet* file_list_load(const char* path, bool null_separated, char* err, si void file_list_destroy(FileListSet* set) { if (!set) return; + path_index_free(&set->index); for (int i = 0; i < set->count; i++) free(set->entries[i]); free(set->entries); - str_hash_set_free(&set->node_index); free(set); } @@ -207,13 +197,12 @@ bool file_list_affects(const FileListSet* set, const char* rel) { return false; if (set->whole_tree) return true; /* whole tree listed */ - /* A node hit means `rel` is a listed entry, or an ancestor directory of one - (rel lives on the path to some listed entry). */ - if (str_hash_set_lookup(&set->node_index, rel, NULL)) + /* An exact entry match means `rel` itself is listed. */ + if (path_index_contains(&set->index, rel)) return true; - /* Otherwise `rel` is affected only when a listed entry is an ancestor of it; - walk rel's directory prefixes (which preserve path-boundary semantics) and - test each for an exact entry. */ + /* Otherwise `rel` is affected when a listed entry is an ancestor directory of + it; walk rel's own directory prefixes (which preserve path-boundary + semantics) and test each for an exact entry. No prefixes are stored. */ size_t len = strlen(rel); while (len > 0) { const char* slash = NULL; @@ -226,9 +215,10 @@ bool file_list_affects(const FileListSet* set, const char* rel) { if (!slash) break; len = (size_t)(slash - rel); - bool is_entry = false; - if (str_hash_set_lookup_n(&set->node_index, rel, len, &is_entry) && is_entry) + if (path_index_contains_n(&set->index, rel, len)) return true; } - return false; + /* Finally `rel` is affected when it is an ancestor directory of a listed + entry (binary search for the first entry at or after `rel` + '/'). */ + return path_index_has_descendant(&set->index, rel); } diff --git a/src/shared/file_list.h b/src/shared/file_list.h index 53c1a79..b18a8ee 100644 --- a/src/shared/file_list.h +++ b/src/shared/file_list.h @@ -14,14 +14,16 @@ * rejected at parse time. The set is immutable and shared read-only across * scanner worker threads. * - * Membership is answered from `node_index`, built once at load time: it holds - * every entry plus every ancestor directory prefix of an entry, with the entry - * flag distinguishing an exact listed path from a mere ancestor. A lookup is - * O(path length) instead of O(entry count). */ + * Membership is answered from `index`, built once at load time over the exact + * entries only: `index.exact` matches a listed path, the sorted view detects an + * ancestor directory of a listed entry, and `rel`'s own directory prefixes are + * matched against the exact set while descending. No ancestor prefix is stored + * as a separate string, so the index is O(entry count) memory however deep the + * paths are, and each query is O(path length) comparisons. */ typedef struct { char** entries; /* normalized rel paths; "" means the whole tree */ int count; - StrHashSet node_index; + PathIndex index; bool whole_tree; /* an entry of "" lists the source root */ } FileListSet; diff --git a/src/shared/utils.c b/src/shared/utils.c index e15d55b..17b1d50 100644 --- a/src/shared/utils.c +++ b/src/shared/utils.c @@ -103,26 +103,20 @@ static size_t str_hash_set_hash(const char* key, size_t len) { return (size_t)XXH64(key, len, 0); } -/* Store an already-allocated key. Returns 1 when a new slot was filled and 0 - * for a duplicate (the caller keeps ownership of `key` when owned is true). */ -static int str_hash_set_put(StrHashSet* set, const char* key, size_t len, bool owned, - bool is_entry) { +/* Store a borrowed key. Returns 1 when a new slot was filled and 0 for a + * duplicate. */ +static int str_hash_set_put(StrHashSet* set, const char* key, size_t len) { size_t mask = set->capacity - 1; size_t index = str_hash_set_hash(key, len) & mask; while (true) { StrHashSetSlot* slot = &set->slots[index]; if (!slot->key) { slot->key = key; - slot->owned = owned; - slot->is_entry = is_entry; set->size++; return 1; } - if (strlen(slot->key) == len && memcmp(slot->key, key, len) == 0) { - if (is_entry) - slot->is_entry = true; + if (strlen(slot->key) == len && memcmp(slot->key, key, len) == 0) return 0; - } index = (index + 1) & mask; } } @@ -138,8 +132,7 @@ static bool str_hash_set_resize(StrHashSet* set, size_t new_capacity) { set->size = 0; for (size_t i = 0; i < old_capacity; i++) { if (old_slots[i].key) - (void)str_hash_set_put(set, old_slots[i].key, strlen(old_slots[i].key), old_slots[i].owned, - old_slots[i].is_entry); + (void)str_hash_set_put(set, old_slots[i].key, strlen(old_slots[i].key)); } free(old_slots); return true; @@ -159,7 +152,7 @@ bool str_hash_set_init(StrHashSet* set, size_t hint) { set->capacity = 0; set->size = 0; size_t capacity = STR_HASH_SET_MIN_CAPACITY; - while (capacity < (hint + 1) * 2) + while (capacity < (hint + 1) * 2 && capacity <= SIZE_MAX / 2) capacity *= 2; set->slots = calloc(capacity, sizeof(StrHashSetSlot)); if (!set->slots) @@ -171,46 +164,18 @@ bool str_hash_set_init(StrHashSet* set, size_t hint) { void str_hash_set_free(StrHashSet* set) { if (!set) return; - for (size_t i = 0; i < set->capacity; i++) { - if (set->slots[i].key && set->slots[i].owned) - free((void*)set->slots[i].key); - } free(set->slots); set->slots = NULL; set->capacity = 0; set->size = 0; } -bool str_hash_set_insert_ref(StrHashSet* set, const char* key, bool is_entry) { +bool str_hash_set_insert_ref(StrHashSet* set, const char* key) { if (!set || !key) return false; if (!str_hash_set_grow(set)) return false; - return str_hash_set_put(set, key, strlen(key), false, is_entry) >= 0; -} - -bool str_hash_set_insert_copy_n(StrHashSet* set, const char* key, size_t len, bool is_entry) { - if (!set || !key) - return false; - bool present = false; - if (str_hash_set_lookup_n(set, key, len, &present)) { - if (is_entry) - (void)str_hash_set_put(set, key, len, false, true); /* upgrade in place */ - return true; - } - if (!str_hash_set_grow(set)) - return false; - char* copy = malloc(len + 1); - if (!copy) - return false; - memcpy(copy, key, len); - copy[len] = '\0'; - int result = str_hash_set_put(set, copy, len, true, is_entry); - if (result <= 0) { - free(copy); - return result == 0; - } - return true; + return str_hash_set_put(set, key, strlen(key)) >= 0; } static const StrHashSetSlot* str_hash_set_find_n(const StrHashSet* set, const char* key, @@ -229,19 +194,140 @@ static const StrHashSetSlot* str_hash_set_find_n(const StrHashSet* set, const ch } } -bool str_hash_set_lookup_n(const StrHashSet* set, const char* key, size_t len, bool* is_entry) { - const StrHashSetSlot* slot = str_hash_set_find_n(set, key, len); - if (!slot) +bool str_hash_set_lookup_n(const StrHashSet* set, const char* key, size_t len) { + return str_hash_set_find_n(set, key, len) != NULL; +} + +bool str_hash_set_lookup(const StrHashSet* set, const char* key) { + if (!key) return false; - if (is_entry) - *is_entry = slot->is_entry; + return str_hash_set_lookup_n(set, key, strlen(key)); +} + +static int str_sorted_array_compare(const void* left, const void* right) { + const char* const* left_key = left; + const char* const* right_key = right; + return strcmp(*left_key, *right_key); +} + +bool str_sorted_array_build(StrSortedArray* array, const char* const* items, size_t count) { + if (!array) + return false; + array->items = NULL; + array->count = 0; + if (count == 0) + return true; + if (!items || count > SIZE_MAX / sizeof(const char*)) + return false; + const char** sorted = malloc(count * sizeof(*sorted)); + if (!sorted) + return false; + for (size_t i = 0; i < count; i++) + sorted[i] = items[i]; + qsort(sorted, count, sizeof(*sorted), str_sorted_array_compare); + array->items = sorted; + array->count = count; return true; } -bool str_hash_set_lookup(const StrHashSet* set, const char* key, bool* is_entry) { - if (!key) +void str_sorted_array_free(StrSortedArray* array) { + if (!array) + return; + free(array->items); + array->items = NULL; + array->count = 0; +} + +bool str_sorted_array_contains(const StrSortedArray* array, const char* key) { + if (!array || !key || array->count == 0) return false; - return str_hash_set_lookup_n(set, key, strlen(key), is_entry); + size_t lo = 0; + size_t hi = array->count; + while (lo < hi) { + size_t mid = lo + (hi - lo) / 2; + int cmp = strcmp(array->items[mid], key); + if (cmp < 0) + lo = mid + 1; + else if (cmp > 0) + hi = mid; + else + return true; + } + return false; +} + +/* Compare `entry` against the virtual key `key` + '/' without allocating the + * concatenation. Returns <0, 0 or >0 as `entry` sorts before, equal to, or + * after that virtual key. */ +static int str_sorted_array_compare_prefix(const char* entry, const char* key, size_t key_len) { + int cmp = strncmp(entry, key, key_len); + if (cmp != 0) + return cmp; + unsigned char next = (unsigned char)entry[key_len]; + if (next == '\0') + return -1; /* entry == key sorts before key + '/' */ + return (int)next - (int)'/'; +} + +bool str_sorted_array_has_child_prefix(const StrSortedArray* array, const char* key) { + if (!array || !key || array->count == 0 || key[0] == '\0') + return false; + size_t key_len = strlen(key); + size_t lo = 0; + size_t hi = array->count; + while (lo < hi) { + size_t mid = lo + (hi - lo) / 2; + if (str_sorted_array_compare_prefix(array->items[mid], key, key_len) < 0) + lo = mid + 1; + else + hi = mid; + } + if (lo >= array->count) + return false; + const char* entry = array->items[lo]; + return strncmp(entry, key, key_len) == 0 && entry[key_len] == '/'; +} + +bool path_index_build(PathIndex* index, const char* const* entries, size_t count) { + if (!index) + return false; + index->exact.slots = NULL; + index->exact.capacity = 0; + index->exact.size = 0; + index->sorted.items = NULL; + index->sorted.count = 0; + if (!str_hash_set_init(&index->exact, count)) + return false; + if (!str_sorted_array_build(&index->sorted, entries, count)) { + str_hash_set_free(&index->exact); + return false; + } + for (size_t i = 0; i < count; i++) { + if (!str_hash_set_insert_ref(&index->exact, entries[i])) { + path_index_free(index); + return false; + } + } + return true; +} + +void path_index_free(PathIndex* index) { + if (!index) + return; + str_hash_set_free(&index->exact); + str_sorted_array_free(&index->sorted); +} + +bool path_index_contains(const PathIndex* index, const char* path) { + return index && str_hash_set_lookup(&index->exact, path); +} + +bool path_index_contains_n(const PathIndex* index, const char* path, size_t len) { + return index && str_hash_set_lookup_n(&index->exact, path, len); +} + +bool path_index_has_descendant(const PathIndex* index, const char* path) { + return index && str_sorted_array_has_child_prefix(&index->sorted, path); } char* output_escape(const char* string, bool eight_bit_output) { @@ -344,38 +430,23 @@ bool format_human_bytes(unsigned long long bytes, char* buffer, size_t buffer_si return written >= 0 && (size_t)written < buffer_size; } -/* Build the keep-set index: every manifest entry is inserted as an exact entry - and every ancestor directory prefix of it as a non-entry node. A lookup of - `rel` therefore succeeds iff `rel` is a kept file, a kept directory, or an - ancestor directory of kept content (the old is_dir_in_manifest predicate); - the entry flag distinguishes an exact kept file from a mere prefix. */ -static bool build_keep_index(ArrayList* manifest, StrHashSet* index) { - if (!str_hash_set_init(index, manifest && manifest->size > 0 ? (size_t)manifest->size : 1)) - return false; - if (!manifest) - return true; - for (int i = 0; i < manifest->size; i++) { - const char* entry = (const char*)manifest->items[i]; - if (!str_hash_set_insert_ref(index, entry, true)) - goto fail; - for (const char* slash = entry; (slash = strchr(slash, '/')) != NULL; slash++) { - if (!str_hash_set_insert_copy_n(index, entry, (size_t)(slash - entry), false)) - goto fail; - } - } - return true; -fail: - str_hash_set_free(index); - return false; +/* Build the keep-set index from the exact manifest entries only. A lookup of + `rel` succeeds iff `rel` is a kept entry, a kept directory, or an ancestor + directory of kept content (the old is_dir_in_manifest predicate); the sorted + view answers "is an ancestor of kept content" without materializing any + per-component prefix copy, so the index is O(manifest size) memory. */ +static bool build_keep_index(const ArrayList* manifest, PathIndex* index) { + if (!manifest || manifest->size <= 0) + return path_index_build(index, NULL, 0); + return path_index_build(index, (const char* const*)manifest->items, (size_t)manifest->size); } -static bool keep_is_dir(const StrHashSet* index, const char* rel_path) { - return str_hash_set_lookup(index, rel_path, NULL); +static bool keep_is_dir(const PathIndex* index, const char* rel_path) { + return path_index_contains(index, rel_path) || path_index_has_descendant(index, rel_path); } -static bool keep_is_file(const StrHashSet* index, const char* rel_path) { - bool is_entry = false; - return str_hash_set_lookup(index, rel_path, &is_entry) && is_entry; +static bool keep_is_file(const PathIndex* index, const char* rel_path) { + return path_index_contains(index, rel_path); } /* True when child_rel is, or lies below, a protected entry. A prefix "a" @@ -405,7 +476,7 @@ bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSki prefixes) mark the enclosing directory as surviving, exactly as they would make a real rmdir fail with ENOTEMPTY. Stops early once *count reaches the cap (sets *exceeds). Returns false on a traversal error. */ -static bool count_extras_fd(int dirfd, const char* rel_path, const StrHashSet* keep, size_t cap, +static bool count_extras_fd(int dirfd, const char* rel_path, const PathIndex* keep, size_t cap, size_t* count, bool* exceeds, const DeleteSkipEntry* skips, int skip_count, bool* survives) { /* openat(dirfd, ".") opens an independent file description: a dup() would @@ -495,7 +566,7 @@ static bool count_extras_fd(int dirfd, const char* rel_path, const StrHashSet* k return operation_ok; } -static bool delete_extras_fd(int dirfd, const char* rel_path, const StrHashSet* keep, +static bool delete_extras_fd(int dirfd, const char* rel_path, const PathIndex* keep, size_t max_delete, size_t* deleted_count, const DeleteSkipEntry* skips, int skip_count) { /* Independent file description (see count_extras_fd). */ @@ -595,7 +666,7 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, const StrHashSet* return operation_ok; } -DeleteWalkResult delete_extras_limited(const char* dest_root, ArrayList* manifest, +DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest, size_t max_delete, const DeleteSkipEntry* skips, int skip_count, size_t* deleted_out) { if (deleted_out) @@ -604,7 +675,7 @@ DeleteWalkResult delete_extras_limited(const char* dest_root, ArrayList* manifes return DELETE_WALK_ERROR; /* Index the keep-set once so both passes answer membership in O(path length) instead of scanning every manifest entry for every destination entry. */ - StrHashSet keep; + PathIndex keep; if (!build_keep_index(manifest, &keep)) return DELETE_WALK_ERROR; int rootfd; @@ -619,7 +690,7 @@ DeleteWalkResult delete_extras_limited(const char* dest_root, ArrayList* manifes rootfd = open(dest_root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); } if (rootfd < 0) { - str_hash_set_free(&keep); + path_index_free(&keep); return DELETE_WALK_ERROR; } if (max_delete != SIZE_MAX) { @@ -632,12 +703,12 @@ DeleteWalkResult delete_extras_limited(const char* dest_root, ArrayList* manifes skip_count, &survives); if (!counted_ok) { close(rootfd); - str_hash_set_free(&keep); + path_index_free(&keep); return DELETE_WALK_ERROR; } if (exceeds) { close(rootfd); - str_hash_set_free(&keep); + path_index_free(&keep); return DELETE_WALK_LIMIT_EXCEEDED; } } @@ -645,13 +716,13 @@ DeleteWalkResult delete_extras_limited(const char* dest_root, ArrayList* manifes bool ok = delete_extras_fd(rootfd, "", &keep, max_delete, &deleted_count, skips, skip_count); if (close(rootfd) != 0) ok = false; - str_hash_set_free(&keep); + path_index_free(&keep); if (deleted_out) *deleted_out = deleted_count; return ok ? DELETE_WALK_OK : DELETE_WALK_ERROR; } -bool delete_extras(const char* dest_root, ArrayList* manifest) { +bool delete_extras(const char* dest_root, const ArrayList* manifest) { return delete_extras_limited(dest_root, manifest, SIZE_MAX, NULL, 0, NULL) == DELETE_WALK_OK; } diff --git a/src/shared/utils.h b/src/shared/utils.h index f27c0c0..e97c635 100644 --- a/src/shared/utils.h +++ b/src/shared/utils.h @@ -7,17 +7,15 @@ #include /* Small open-addressing string hash set used to turn quadratic membership - * scans into O(path length) lookups (the --delete keep-set and the + * scans into O(path length) exact-match lookups (the --delete keep-set and the * --files-from allow-set). Keys are hashed with xxHash64 (seed 0); collisions * are resolved by linear probing over a power-of-two table that grows at 75% - * load. Slots may either borrow a caller-owned key (insert_ref) or own an - * internal copy (insert_copy_n); owned copies are released by - * str_hash_set_free. The set is not thread-safe for mutation, but a fully - * built set supports concurrent read-only lookups. */ + * load. Keys are always borrowed from the caller and must outlive the set; the + * set never copies or owns keys, so indexing M entries costs O(M) memory. The + * set is not thread-safe for mutation, but a fully built set supports + * concurrent read-only lookups. */ typedef struct { const char* key; /* NULL marks an empty slot */ - bool owned; /* key is an internal copy that free() must release */ - bool is_entry; /* key was inserted as an exact entry, not just a prefix */ } StrHashSetSlot; typedef struct { @@ -30,16 +28,54 @@ typedef struct { * allocation failure. */ bool str_hash_set_init(StrHashSet* set, size_t hint); void str_hash_set_free(StrHashSet* set); -/* Insert a borrowed key (must outlive the set). A duplicate only upgrades - * is_entry. Returns false on allocation failure. */ -bool str_hash_set_insert_ref(StrHashSet* set, const char* key, bool is_entry); -/* Insert a copy of the first `len` bytes of `key` (which need not be - * NUL-terminated). Returns false on allocation failure. */ -bool str_hash_set_insert_copy_n(StrHashSet* set, const char* key, size_t len, bool is_entry); -/* Look up a NUL-terminated key / a key of `len` bytes. On a hit, optionally - * reports whether the stored key was inserted as an exact entry. */ -bool str_hash_set_lookup(const StrHashSet* set, const char* key, bool* is_entry); -bool str_hash_set_lookup_n(const StrHashSet* set, const char* key, size_t len, bool* is_entry); +/* Insert a borrowed key (must outlive the set). A duplicate is ignored. + * Returns false on allocation failure. */ +bool str_hash_set_insert_ref(StrHashSet* set, const char* key); +/* Look up a NUL-terminated key / a key of `len` bytes. */ +bool str_hash_set_lookup(const StrHashSet* set, const char* key); +bool str_hash_set_lookup_n(const StrHashSet* set, const char* key, size_t len); + +/* Sorted, non-owning view of NUL-terminated strings. Built from borrowed + * pointers (qsort), so indexing M entries costs O(M) memory and O(M log M) + * time; exact membership and ancestor-prefix existence are binary searches + * that never materialize a prefix copy. */ +typedef struct { + const char** items; /* sorted with strcmp; borrowed, never freed */ + size_t count; +} StrSortedArray; + +/* Build `array` over the borrowed `items`. Only the pointer array is copied, + * never the strings. Returns false on allocation failure. */ +bool str_sorted_array_build(StrSortedArray* array, const char* const* items, size_t count); +void str_sorted_array_free(StrSortedArray* array); +/* True when some item equals `key`. */ +bool str_sorted_array_contains(const StrSortedArray* array, const char* key); +/* True when some item starts with `key` followed by '/' (i.e. `key` is a proper + * ancestor directory of an item). Allocates nothing. */ +bool str_sorted_array_has_child_prefix(const StrSortedArray* array, const char* key); + +/* Read-only membership index over exact relative paths. `exact` answers + * O(path length) equality; `sorted` answers whether any indexed path lies + * strictly below a query directory. Both borrow their keys from the caller and + * no ancestor prefix is stored as a separate string, so an index over M entries + * is O(M) memory regardless of path depth. Not thread-safe to build, but safe + * for concurrent read-only queries once built. */ +typedef struct { + StrHashSet exact; + StrSortedArray sorted; +} PathIndex; + +/* Build an index borrowing `entries` (which must outlive the index). Returns + * false on allocation failure, freeing any partial state. */ +bool path_index_build(PathIndex* index, const char* const* entries, size_t count); +void path_index_free(PathIndex* index); +/* True when `path` is an indexed entry. */ +bool path_index_contains(const PathIndex* index, const char* path); +/* Length-bounded form of path_index_contains (`path` need not be terminated). */ +bool path_index_contains_n(const PathIndex* index, const char* path, size_t len); +/* True when some indexed entry lies strictly below `path` (starts with + * `path` + '/'). */ +bool path_index_has_descendant(const PathIndex* index, const char* path); char* str_dup(const char* string); char* output_escape(const char* string, bool eight_bit_output); @@ -83,10 +119,10 @@ bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSki and the delete pass are two separate walks, so a concurrent change between them (another process adding/removing entries) can make the second pass delete a different set than the first one counted. */ -DeleteWalkResult delete_extras_limited(const char* dest_root, ArrayList* manifest, +DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest, size_t max_delete, const DeleteSkipEntry* skips, int skip_count, size_t* deleted_out); -bool delete_extras(const char* dest_root, ArrayList* manifest); +bool delete_extras(const char* dest_root, const ArrayList* manifest); bool utils_set_authorized_root(int fd, const char* canonical_path); /* The fd-only compatibility form is fail-closed for path-based operations; * callers should use utils_set_authorized_root with the canonical identity. */ diff --git a/tests/test_file_list.c b/tests/test_file_list.c index bf31db5..80a0ca8 100644 --- a/tests/test_file_list.c +++ b/tests/test_file_list.c @@ -118,6 +118,79 @@ static void test_membership_matches_reference() { EXPECT_TRUE(file_list_affects(NULL, NULL)); } +/* Explicit ancestor/descendant coverage: a query that is a proper ancestor of + a listed entry is affected, and a query below a listed entry is affected, + while a component-boundary neighbor is not. */ +static void test_ancestor_and_descendant_queries() { + const char* path = "test_file_list_ancestor.txt"; + char err[160]; + write_list(path, "top/mid/leaf.txt\nsingle.txt\n"); + FileListSet* set = file_list_load(path, false, err, sizeof(err)); + EXPECT_NOT_NULL(set); + + /* q is an ancestor of a listed entry. */ + EXPECT_TRUE(file_list_affects(set, "top")); + EXPECT_TRUE(file_list_affects(set, "top/mid")); + EXPECT_FALSE(file_list_affects(set, "top/other")); /* neither direction */ + EXPECT_FALSE(file_list_affects(set, "to")); /* component boundary */ + + /* A listed entry is an ancestor of q. */ + EXPECT_TRUE(file_list_affects(set, "single.txt")); + EXPECT_TRUE(file_list_affects(set, "single.txt/deeper")); + EXPECT_FALSE(file_list_affects(set, "single.txtx")); /* boundary */ + + check_queries(set, + (const char*[]){"top", "top/mid", "top/mid/leaf.txt", "top/other", "single.txt", + "single.txt/deeper", "single.txtx", "to"}, + 8); + file_list_destroy(set); + remove(path); +} + +/* Regression for the remote OOM: an adversarial --files-from entry made of a + very deep chain of repeated components must be indexed with memory + proportional to the entry count. The old implementation stored one copied + ancestor prefix per component (O(L^2) bytes for a single entry); the sorted + index stores the exact entries only. */ +static void test_deep_paths_are_bounded() { + const char* path = "test_file_list_deep.txt"; + enum { COMPONENTS = 20000 }; + size_t entry_len = (size_t)COMPONENTS * 2; /* "a/" per component */ + char* entry = malloc(entry_len + 1); + EXPECT_NOT_NULL(entry); + for (size_t i = 0; i < entry_len; i += 2) { + entry[i] = 'a'; + entry[i + 1] = '/'; + } + entry[entry_len - 1] = 'z'; /* .../a/z: a deep leaf name */ + entry[entry_len] = '\0'; + + FILE* fp = fopen(path, "wb"); + EXPECT_NOT_NULL(fp); + EXPECT_EQ_INT((int)fwrite(entry, 1, entry_len, fp), (int)entry_len); + EXPECT_EQ_INT(fputc('\n', fp), '\n'); + fclose(fp); + + char err[160]; + FileListSet* set = file_list_load(path, false, err, sizeof(err)); + EXPECT_NOT_NULL(set); + EXPECT_EQ_INT(set->count, 1); + /* One exact entry stored, not one node per path component. */ + EXPECT_EQ_INT((int)set->index.sorted.count, 1); + EXPECT_EQ_INT((int)set->index.exact.size, 1); + EXPECT_TRUE(file_list_affects(set, entry)); /* exact */ + EXPECT_TRUE(file_list_affects(set, "a")); /* ancestor of the entry */ + EXPECT_TRUE(file_list_affects(set, "a/a")); /* deeper ancestor */ + EXPECT_FALSE(file_list_affects(set, "b")); /* unrelated */ + EXPECT_FALSE(file_list_affects(set, "aa")); /* component boundary */ + + file_list_destroy(set); + remove(path); + free(entry); +} + void test_file_list() { test_membership_matches_reference(); + test_ancestor_and_descendant_queries(); + test_deep_paths_are_bounded(); } diff --git a/tests/test_shared_utils.c b/tests/test_shared_utils.c index 6663695..c89c006 100644 --- a/tests/test_shared_utils.c +++ b/tests/test_shared_utils.c @@ -456,7 +456,72 @@ static void test_fd_peer_ip() { EXPECT_EQ_STR(peer_string, ""); } +/* The keep/files-from indexes must store exactly the input entries (one node + each), never a copied ancestor prefix per component. This builds a PathIndex + over paths thousands of components deep and checks the structural bound plus + the exact / descendant query semantics. */ +static void test_path_index_bounded() { + enum { COUNT = 8, COMPONENTS = 5000 }; + size_t entry_len = (size_t)COMPONENTS * 2 + 2; /* trailing "xN" */ + char* storage = malloc((size_t)COUNT * (entry_len + 1)); + EXPECT_NOT_NULL(storage); + const char** entries = calloc(COUNT, sizeof(char*)); + EXPECT_NOT_NULL(entries); + for (int i = 0; i < COUNT; i++) { + char* entry = storage + (size_t)i * (entry_len + 1); + size_t pos = 0; + for (int c = 0; c < COMPONENTS; c++) { + entry[pos++] = 'a'; + entry[pos++] = '/'; + } + entry[pos++] = 'x'; + entry[pos++] = (char)('0' + i); + entry[pos] = '\0'; + entries[i] = entry; + } + + PathIndex index; + EXPECT_TRUE(path_index_build(&index, entries, COUNT)); + EXPECT_EQ_INT((int)index.sorted.count, COUNT); + EXPECT_EQ_INT((int)index.exact.size, COUNT); + EXPECT_TRUE(path_index_contains(&index, entries[0])); + EXPECT_FALSE(path_index_contains(&index, "a")); + EXPECT_TRUE(path_index_has_descendant(&index, "a")); + EXPECT_TRUE(path_index_has_descendant(&index, "a/a")); + EXPECT_FALSE(path_index_has_descendant(&index, "aa")); + path_index_free(&index); + + free((void*)entries); + free(storage); +} + +static void test_path_index_semantics() { + const char* entries[] = {"a/b/c.txt", "a/b/d.txt", "x.txt", "deep/deeper/deepest"}; + PathIndex index; + EXPECT_TRUE(path_index_build(&index, entries, 4)); + EXPECT_TRUE(path_index_contains(&index, "a/b/c.txt")); + EXPECT_FALSE(path_index_contains(&index, "a/b")); + EXPECT_TRUE(path_index_contains_n(&index, "a/b/c.txt/ignored", 9)); + EXPECT_FALSE(path_index_contains_n(&index, "a/b/c.txt/ignored", 10)); + EXPECT_TRUE(path_index_has_descendant(&index, "a")); + EXPECT_TRUE(path_index_has_descendant(&index, "a/b")); + EXPECT_FALSE(path_index_has_descendant(&index, "a/b/c.txt")); + EXPECT_FALSE(path_index_has_descendant(&index, "ab")); + EXPECT_FALSE(path_index_has_descendant(&index, "")); + path_index_free(&index); + + /* A zero-entry index answers no queries. */ + PathIndex empty; + EXPECT_TRUE(path_index_build(&empty, NULL, 0)); + EXPECT_EQ_INT((int)empty.sorted.count, 0); + EXPECT_FALSE(path_index_contains(&empty, "a")); + EXPECT_FALSE(path_index_has_descendant(&empty, "a")); + path_index_free(&empty); +} + void test_shared_utils() { + test_path_index_bounded(); + test_path_index_semantics(); test_walker_removes_extras_keeps_manifest_and_protected(); test_walker_keeps_nested_manifest_dirs(); test_walker_max_delete_exceeded_deletes_nothing(); From 3cf2e2c91f4bbec3831e2c2a26e4bbc6c079a5f0 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 05:01:15 +0200 Subject: [PATCH 025/155] docs(version): align 2.20.0 artifacts; fix protocol-bump rationale --- CHANGELOG.md | 29 +++++++++++++++++++++++++++++ CMakeLists.txt | 2 +- RSYNC_COMPAT.md | 10 ++++++---- src/shared/config.h | 25 ++++++++++++------------- src/shared/utils.c | 4 +++- 5 files changed, 51 insertions(+), 19 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index bcdf3a4..66132a2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,35 @@ All notable changes to FastSync are documented here. Versions match `PROTOCOL_VERSION` (printed by `fastsync --version`); the client and server must run the same version because the handshake is strict. +## [2.20.0] - 2026-09-13 + +### Security + +- Cap cumulative `DirTimeList` growth and bound pre-auth config-string memory + (remote memory-exhaustion DoS). +- Daemon host access control (`hosts allow`/`hosts deny`, IPv4/IPv6/CIDR), + configurable global `max connections`, connection audit logging, and a + bounded `auth failure delay` throttle. IPv4-mapped peers are normalized and + invalid patterns are rejected at parse time (no silent fail-open). +- Honor `--timeout` for protocol I/O and bound idle/session time to defeat + keepalive slowloris; child-safe signal handling in the forked daemon. +- Compiler/linker hardening (`_FORTIFY_SOURCE`, stack protector, PIE, RELRO) + and pinned build dependencies. + +### Fixed + +- Use-after-free in the basis-dir oversize preflight. +- Placeholder `Data` leaks, `missing_args` leak, scanner chunk leak. +- Thread-safe logging; single fd owner and cleanup epilogue in the server + handler. + +### Performance + +- Metadata now crosses the wire as one packed frame (protocol 2.20.0). +- Delete keep-set and `--files-from` lookups indexed (O(n*m) → O(n)). +- Reused per-thread zstd contexts; `TCP_NODELAY` by default. +- Byte-bounded sender queues; removed a redundant scanner `stat()`. + ## [2.19.0] - 2026-09-12 ### Security diff --git a/CMakeLists.txt b/CMakeLists.txt index ff92672..f384f62 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1,6 +1,6 @@ cmake_minimum_required(VERSION 3.22) -project(FastFileTransfer VERSION 2.19.0) +project(FastFileTransfer VERSION 2.20.0) set(CMAKE_EXPORT_COMPILE_COMMANDS ON) set(CMAKE_C_STANDARD 11) diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 830605e..6fb91c8 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -850,10 +850,12 @@ now sends the metadata as ONE packed frame: a single `int32` present flag `FILE_METADATA_WIRE_SIZE`-byte (68-byte) field record already emitted by the shared `metadata_to_buf()`/`metadata_from_buf()` chunk codec. Absent metadata is a lone `int32` zero. The encoded field layout is unchanged (only the framing -collapses), so chunk-serialized blobs remain byte-identical; `PROTOCOL_VERSION` -was bumped `2.19.0 → 2.20.0` because a 2.19 peer would desynchronize on the -removed frames. The strict same-version handshake rejects any mismatch before a -byte of the frame is parsed. +collapses), so chunk-serialized blobs remain byte-identical. Protocol data is an +unframed byte stream, so the packed encoding is byte-for-byte identical to the +old field-by-field writes; `PROTOCOL_VERSION` was bumped `2.19.0 → 2.20.0` as a +deliberate lockstep-release marker rather than because of a +desynchronization. The strict same-version handshake rejects any mismatch before +a byte of the frame is parsed. ### Recommended Delivery Order diff --git a/src/shared/config.h b/src/shared/config.h index 23b811b..29b7baf 100644 --- a/src/shared/config.h +++ b/src/shared/config.h @@ -646,19 +646,18 @@ typedef struct Config { * * Packed Metadata Wave: 2.19.0 -> 2.20.0. * - * WHY the bump, grounded in the wire: metadata_send()/metadata_receive() no - * longer emit/consume the metadata as up to 12 separate per-field framed - * writes. A file's metadata now crosses the wire as ONE packed frame: a - * single int32 present flag (0 = absent, 1 = present) followed, when present, - * by the fixed FILE_METADATA_WIRE_SIZE-byte (68-byte) field record produced by - * metadata_to_buf(). A 2.19 peer would desynchronize on the removed frames - * (it would read the packed record's bytes as a stream of separate field - * frames), so the strict same-version handshake (config_receive rejects a - * mismatched version before parsing anything else) is what keeps a 2.20 client - * and a 2.19 server from ever reaching that state. The encoded field layout - * itself is unchanged (only its framing collapses), so the chunk codec, which - * already used the packed metadata_to_buf()/metadata_from_buf() codec, is - * byte-identical to before. */ + * WHY the bump: metadata_send()/metadata_receive() no longer emit/consume the + * metadata as up to 12 separate per-field writes. A file's metadata now + * crosses the wire as ONE packed frame: a single int32 present flag (0 = + * absent, 1 = present) followed, when present, by the fixed + * FILE_METADATA_WIRE_SIZE-byte (68-byte) field record produced by + * metadata_to_buf(). Protocol data is an unframed byte stream, so the packed + * encoding is byte-for-byte identical to the old field-by-field writes (same + * fields, same order, same widths); the change only removes per-field syscalls. + * The bump is therefore a deliberate lockstep-release marker, not a + * desynchronization fix — the strict same-version handshake still rejects a + * mixed 2.19/2.20 deployment. The chunk codec, which already used the packed + * metadata_to_buf()/metadata_from_buf() form, is unchanged. */ #define PROTOCOL_VERSION "2.20.0" #define DEFAULT_CHUNK_SIZE (10 * 1024 * 1024) /* Upper bound on total basis-dir entries (rsync caps --link-dest at 20). */ diff --git a/src/shared/utils.c b/src/shared/utils.c index 17b1d50..080785d 100644 --- a/src/shared/utils.c +++ b/src/shared/utils.c @@ -175,7 +175,9 @@ bool str_hash_set_insert_ref(StrHashSet* set, const char* key) { return false; if (!str_hash_set_grow(set)) return false; - return str_hash_set_put(set, key, strlen(key)) >= 0; + /* put() returns 1 for a new slot and 0 for a duplicate; both are success. */ + (void)str_hash_set_put(set, key, strlen(key)); + return true; } static const StrHashSetSlot* str_hash_set_find_n(const StrHashSet* set, const char* key, From c944787e032d0eb71eff285e84d6bfb19000e909 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 05:19:18 +0200 Subject: [PATCH 026/155] refactor: remove dead file_store subsystem and unused wrappers --- src/shared/compression.c | 4 - src/shared/compression.h | 1 - src/shared/file.c | 9 -- src/shared/file.h | 1 - src/shared/file_store.c | 186 ------------------------------------- src/shared/file_store.h | 7 -- src/shared/protocol.c | 4 - src/shared/protocol.h | 1 - src/shared/transport_tcp.c | 4 - src/shared/transport_tcp.h | 1 - src/shared/utils.c | 10 +- src/shared/utils.h | 6 ++ tests/test_file.c | 5 +- 13 files changed, 17 insertions(+), 222 deletions(-) diff --git a/src/shared/compression.c b/src/shared/compression.c index d9aa037..7609921 100644 --- a/src/shared/compression.c +++ b/src/shared/compression.c @@ -17,10 +17,6 @@ static char* SKIP_COMPRESSION_EXTENSIONS[] = {".jpg", ".jpeg", ".png", ".gif", ".mp4", ".mkv", ".zip", ".gz", ".xz", ".zst", NULL}; -bool compression_should_skip(const char* path) { - return compression_should_skip_with_suffixes(path, NULL, -1); -} - bool compression_should_skip_with_suffixes(const char* path, char* const* suffixes, int count) { if (!path) return false; diff --git a/src/shared/compression.h b/src/shared/compression.h index 179c2b7..2c3753c 100644 --- a/src/shared/compression.h +++ b/src/shared/compression.h @@ -11,7 +11,6 @@ Data* data_compress_with_threads(Data* data_to_compress, int compression_level, int compression_threads); Data* data_decompress(Data* compressed_data); Data* data_decompress_limited(Data* compressed_data, size_t maximum_size); -bool compression_should_skip(const char* path); bool compression_should_skip_with_suffixes(const char* path, char* const* suffixes, int count); /* Release the calling thread's cached zstd contexts (compressor, decompressor diff --git a/src/shared/file.c b/src/shared/file.c index 6220967..b85e988 100644 --- a/src/shared/file.c +++ b/src/shared/file.c @@ -287,11 +287,6 @@ size_t file_content_to_buffer(File* file) { static int authorized_root_fd = -1; static char* authorized_root_path; -static bool path_is_within_root(const char* root, const char* path) { - size_t root_len = strlen(root); - return strncmp(root, path, root_len) == 0 && (path[root_len] == '\0' || path[root_len] == '/'); -} - bool file_set_authorized_root(int fd, const char* canonical_path) { char* path_copy = canonical_path ? str_dup(canonical_path) : NULL; if (canonical_path && !path_copy) { @@ -362,10 +357,6 @@ void file_set_keep_dirlinks(bool enable) { file_keep_dirlinks = enable; } -bool file_get_keep_dirlinks(void) { - return file_keep_dirlinks; -} - /* --trust-sender (Phase 5) receiver process-wide policy: when set, the receiver * trusts the sender's file list and skips its own redundant up-front re- * validation (empty/".." path rejection, escaping-symlink-target containment). diff --git a/src/shared/file.h b/src/shared/file.h index 554170b..570d703 100644 --- a/src/shared/file.h +++ b/src/shared/file.h @@ -52,7 +52,6 @@ bool file_symlink_at_secure(const char* path, const char* target); /* --keep-dirlinks (-K) receiver process-wide policy: allow an in-root existing * symlink-to-directory to be followed as a directory. */ void file_set_keep_dirlinks(bool enable); -bool file_get_keep_dirlinks(void); /* --trust-sender receiver process-wide policy (Phase 5). When set, the * receiver trusts that the sender already produced a clean file list and skips diff --git a/src/shared/file_store.c b/src/shared/file_store.c index b14da25..5365598 100644 --- a/src/shared/file_store.c +++ b/src/shared/file_store.c @@ -1,129 +1,7 @@ #include -#include -#include -#include -#include -#include -#include #include #include "file_store.h" -#include "metadata.h" -#include "utils.h" - -static int authorized_root_fd = -1; -static char* authorized_root_path; - -static bool path_is_within_root(const char* root, const char* path) { - size_t root_length = strlen(root); - return strncmp(root, path, root_length) == 0 && - (path[root_length] == '\0' || path[root_length] == '/'); -} - -bool file_store_set_authorized_root(int fd, const char* canonical_path) { - char* new_path = canonical_path ? str_dup(canonical_path) : NULL; - if (canonical_path && !new_path) { - authorized_root_fd = -1; - free(authorized_root_path); - authorized_root_path = NULL; - return false; - } - free(authorized_root_path); - authorized_root_path = new_path; - authorized_root_fd = fd; - return true; -} - -int file_store_open_secure_parent(const char* path, char** leaf_out) { - char* copy = str_dup(path); - if (!copy) - return -1; - char* parent = dirname(copy); - const char* slash = strrchr(path, '/'); - char* leaf = str_dup(slash ? slash + 1 : path); - if (!leaf) { - free(copy); - return -1; - } - int fd; - if (authorized_root_fd >= 0) { - if (!authorized_root_path || path[0] != '/' || - !path_is_within_root(authorized_root_path, path)) { - free(copy); - free(leaf); - return -1; - } - fd = dup(authorized_root_fd); - if (fd < 0) { - free(copy); - free(leaf); - return -1; - } - size_t root_length = strlen(authorized_root_path); - char* relative = str_dup(path + root_length); - if (!relative) { - free(copy); - free(leaf); - close(fd); - return -1; - } - free(copy); - copy = relative; - parent = dirname(copy); - } else { - fd = (parent[0] == '/') ? open("/", O_RDONLY | O_DIRECTORY | O_CLOEXEC) - : open(".", O_RDONLY | O_DIRECTORY | O_CLOEXEC); - } - if (fd < 0) { - free(copy); - free(leaf); - return -1; - } - char* save = NULL; - char* component = strtok_r(parent, "/", &save); - while (component) { - if (strcmp(component, "..") == 0) { - close(fd); - free(copy); - free(leaf); - return -1; - } - if (strcmp(component, ".") != 0) { - int next = openat(fd, component, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - if (next < 0 && errno == ENOENT) { - if (mkdirat(fd, component, 0755) == 0 || errno == EEXIST) - next = openat(fd, component, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - } - if (next < 0) { - close(fd); - free(copy); - free(leaf); - return -1; - } - close(fd); - fd = next; - } - component = strtok_r(NULL, "/", &save); - } - free(copy); - *leaf_out = leaf; - return fd; -} - -bool file_store_rename_secure(const char* old_path, const char* new_path) { - char *old_leaf = NULL, *new_leaf = NULL; - int old_parent = file_store_open_secure_parent(old_path, &old_leaf); - int new_parent = file_store_open_secure_parent(new_path, &new_leaf); - bool ok = old_parent >= 0 && new_parent >= 0 && - renameat(old_parent, old_leaf, new_parent, new_leaf) == 0; - if (old_parent >= 0) - close(old_parent); - if (new_parent >= 0) - close(new_parent); - free(old_leaf); - free(new_leaf); - return ok; -} static bool write_all(int fd, const void* data, unsigned long long size) { const unsigned char* p = data; @@ -176,67 +54,3 @@ bool file_store_write_sparse(int fd, const unsigned char* data, unsigned long lo } return ftruncate(fd, (off_t)size) == 0; } - -bool file_store_write_secure(const char* path, const void* data, unsigned long long data_size, - bool inplace, bool sparse, const FileMetadata* metadata, - bool preserve_executability) { - char* leaf = NULL; - int dirfd = file_store_open_secure_parent(path, &leaf); - if (dirfd < 0) - return false; - int fd = -1; - bool ok = false; - if (inplace) { - fd = openat(dirfd, leaf, O_WRONLY | O_CREAT | O_TRUNC | O_CLOEXEC | O_NOFOLLOW, 0644); - if (fd >= 0) { - if (sparse && data_size > 0) { - if (ftruncate(fd, (off_t)data_size) == 0) - ok = file_store_write_sparse(fd, data, data_size); - } else { - ok = write_all(fd, data, data_size); - } - if (ok && metadata) - ok = file_restore_metadata_fd(fd, metadata, preserve_executability); - } - } else { - int tmp_size = snprintf(NULL, 0, ".%s.tmp.%ld.%u", leaf, (long)getpid(), 99U); - if (tmp_size < 0) { - close(dirfd); - free(leaf); - return false; - } - char* tmp = malloc((size_t)tmp_size + 1); - if (!tmp) { - close(dirfd); - free(leaf); - return false; - } - for (unsigned int i = 0; i < 100 && !ok; ++i) { - snprintf(tmp, (size_t)tmp_size + 1, ".%s.tmp.%ld.%u", leaf, (long)getpid(), i); - fd = openat(dirfd, tmp, O_WRONLY | O_CREAT | O_EXCL | O_CLOEXEC | O_NOFOLLOW, 0600); - if (fd < 0) - continue; - if (sparse && data_size > 0) - ok = ftruncate(fd, (off_t)data_size) == 0; - if (ok || (!sparse || data_size == 0)) - ok = (sparse && data_size > 0) - ? file_store_write_sparse(fd, (const unsigned char*)data, data_size) - : write_all(fd, data, data_size); - if (ok && metadata) - ok = file_restore_metadata_fd(fd, metadata, preserve_executability); - if (close(fd) != 0) - ok = false; - fd = -1; - if (ok && renameat(dirfd, tmp, dirfd, leaf) != 0) - ok = false; - if (!ok) - unlinkat(dirfd, tmp, 0); - } - free(tmp); - } - if (fd >= 0) - close(fd); - close(dirfd); - free(leaf); - return ok; -} diff --git a/src/shared/file_store.h b/src/shared/file_store.h index 4bfb39a..9514648 100644 --- a/src/shared/file_store.h +++ b/src/shared/file_store.h @@ -1,15 +1,8 @@ #ifndef FILE_STORE_H #define FILE_STORE_H -#include "file.h" #include -bool file_store_set_authorized_root(int fd, const char* canonical_path); -int file_store_open_secure_parent(const char* path, char** leaf_out); -bool file_store_rename_secure(const char* old_path, const char* new_path); -bool file_store_write_secure(const char* path, const void* data, unsigned long long data_size, - bool inplace, bool sparse, const FileMetadata* metadata, - bool preserve_executability); /* Sparse-aware write (--sparse/-S): every all-zero run of at least * SPARSE_HOLE_MIN bytes is skipped with lseek(SEEK_CUR) so it becomes a real * hole; every other byte is written. The caller pre-sizes the file with diff --git a/src/shared/protocol.c b/src/shared/protocol.c index 3c3eee0..f3dd81b 100644 --- a/src/shared/protocol.c +++ b/src/shared/protocol.c @@ -570,10 +570,6 @@ Data* protocol_receive_data_limited(ProtocolSession* session, unsigned long long return result; } -Data* protocol_receive_data(ProtocolSession* session) { - return protocol_receive_data_limited(session, MAX_DATA_PAYLOAD_SIZE); -} - bool protocol_send_int(ProtocolSession* session, int data) { if (!protocol_send_n_data(session, &data, sizeof(int))) return false; diff --git a/src/shared/protocol.h b/src/shared/protocol.h index 60f6dec..b27c2d1 100644 --- a/src/shared/protocol.h +++ b/src/shared/protocol.h @@ -172,7 +172,6 @@ char* protocol_receive_str(ProtocolSession* session); bool protocol_send_str_redacted(ProtocolSession* session, const char* data); char* protocol_receive_str_redacted(ProtocolSession* session); bool protocol_send_data(ProtocolSession* session, const Data* data); -Data* protocol_receive_data(ProtocolSession* session); Data* protocol_receive_data_limited(ProtocolSession* session, unsigned long long maximum_size); bool protocol_send_int(ProtocolSession* session, int data); bool protocol_receive_int(ProtocolSession* session, int* data); diff --git a/src/shared/transport_tcp.c b/src/shared/transport_tcp.c index 42dad4c..2758cbe 100644 --- a/src/shared/transport_tcp.c +++ b/src/shared/transport_tcp.c @@ -440,10 +440,6 @@ bool tcp_connect_socket_ex(Client* client, const char* host, int port, return true; } -bool tcp_connect_socket(Client* client, const char* host, int port) { - return tcp_connect_socket_ex(client, host, port, NULL); -} - bool client_connect_ex(Client* client, const char* host, int port, const TcpConnectOptions* opts) { if (!tcp_connect_socket_ex(client, host, port, opts)) return false; diff --git a/src/shared/transport_tcp.h b/src/shared/transport_tcp.h index 4439003..e37b879 100644 --- a/src/shared/transport_tcp.h +++ b/src/shared/transport_tcp.h @@ -57,7 +57,6 @@ bool client_connect_ex(Client* client, const char* host, int port, const TcpConn bool client_connect(Client* client, const char* host, int port); bool tcp_connect_socket_ex(Client* client, const char* host, int port, const TcpConnectOptions* opts); -bool tcp_connect_socket(Client* client, const char* host, int port); void client_disconnect(Client* client); void client_delete(Client* client); void tcp_set_timeouts(int timeout_sec, int contimeout_sec); diff --git a/src/shared/utils.c b/src/shared/utils.c index 080785d..d790172 100644 --- a/src/shared/utils.c +++ b/src/shared/utils.c @@ -36,11 +36,19 @@ void utils_set_authorized_root_fd(int fd) { (void)utils_set_authorized_root(fd, NULL); } -static bool path_is_within_root(const char* root, const char* path) { +bool path_is_within_root(const char* root, const char* path) { size_t root_len = strlen(root); return strncmp(root, path, root_len) == 0 && (path[root_len] == '\0' || path[root_len] == '/'); } +/* Open the destination root directory itself, confined to the authorized root. + * NOTE (do not merge with file_open_secure_parent): this walk opens dest_root + * (a directory that must already exist) and returns its fd, whereas + * file_open_secure_parent resolves the PARENT of a file path, optionally + * creating missing components and honouring --keep-dirlinks / --copy-as. The + * two differ in create-vs-no-create, in what path component they stop at, and + * in the extra receiver policies they apply, so they are intentionally kept + * separate. Both rely on the shared lexical path_is_within_root check. */ static int open_authorized_destination(const char* dest_root) { if (authorized_root_fd < 0 || !authorized_root_path || !dest_root || !path_is_within_root(authorized_root_path, dest_root)) diff --git a/src/shared/utils.h b/src/shared/utils.h index e97c635..f4a5bd6 100644 --- a/src/shared/utils.h +++ b/src/shared/utils.h @@ -127,6 +127,12 @@ bool utils_set_authorized_root(int fd, const char* canonical_path); /* The fd-only compatibility form is fail-closed for path-based operations; * callers should use utils_set_authorized_root with the canonical identity. */ void utils_set_authorized_root_fd(int fd); +/* True when `path` is `root` itself or lies directly beneath it: a lexical + * prefix test requiring the byte after `root` to be '\0' or '/'. Both `root` + * and `path` must be absolute canonical paths free of "."/".." components (the + * callers guarantee this); this is containment by string, not by resolved + * symlinks. Shared by the utils and file secure-walk root confinement. */ +bool path_is_within_root(const char* root, const char* path); bool has_path_traversal(const char* path); bool utils_valid_batch_path(const char* path); bool format_human_bytes(unsigned long long bytes, char* buffer, size_t buffer_size); diff --git a/tests/test_file.c b/tests/test_file.c index 4d53b70..4561f1a 100644 --- a/tests/test_file.c +++ b/tests/test_file.c @@ -3,7 +3,6 @@ #endif #include "test_file.h" #include "file.h" -#include "file_store.h" #include "file_receive.h" #include "data.h" #include "config.h" @@ -1225,7 +1224,7 @@ void test_trust_sender() { } /* --sparse/-S hole preservation: a buffer with a long zero run written via - * file_store_write_secure(sparse=true) must round-trip its content exactly and + * file_to_disk_secure(sparse=true) must round-trip its content exactly and * have the right logical size, and should additionally be genuinely sparse on * filesystems that support holes. The sparseness assertion is tolerant: if the * filesystem reports no holes (SEEK_HOLE/SEEK_DATA -> ENXIO) we skip the strict @@ -1247,7 +1246,7 @@ static void test_file_write_to_disk_sparse_preserves_holes() { buf[size - 1 - i] = (unsigned char)((i * 7) % 253); } - EXPECT_TRUE(file_store_write_secure(path, buf, size, false, true, NULL, false)); + EXPECT_TRUE(file_to_disk_secure(path, buf, size, false, true, false, NULL, false, NULL)); /* Logical size must equal data_size exactly. */ struct stat st; From eea66a7848164df13f7303c7a4d0df88ee27590b Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 05:24:23 +0200 Subject: [PATCH 027/155] refactor(config,metadata): shared invariants; length-bounded metadata parser --- src/client/client_validation.c | 113 ++------------------- src/shared/chunk.c | 152 ++++++++++------------------ src/shared/config.c | 90 +++++++++++----- src/shared/config.h | 10 ++ src/shared/metadata.c | 63 ++++++------ src/shared/metadata.h | 9 +- tests/fuzz/fuzz_metadata_from_buf.c | 16 +-- tests/test_client_cli.c | 47 +++++++++ tests/test_config.c | 152 ++++++++++++++++++++++++++++ tests/test_fuzz_smoke.c | 3 +- tests/test_metadata.c | 47 +++++++-- 11 files changed, 422 insertions(+), 280 deletions(-) diff --git a/src/client/client_validation.c b/src/client/client_validation.c index f6a8ad0..f19886b 100644 --- a/src/client/client_validation.c +++ b/src/client/client_validation.c @@ -1,6 +1,4 @@ #include "client_validation.h" -#include "charset.h" -#include "delay_updates.h" #include "log.h" #include "usage.h" #include "utils.h" @@ -41,17 +39,6 @@ bool validate_config(const Config* config) { print_usage(); return false; } - if (config_has_basis(config) && config->use_chunk_serialization) { - log_message(LOG_LEVEL_ERROR, - "--compare-dest/--copy-dest/--link-dest require per-file incremental checks and " - "cannot be combined with -s (chunk serialization)"); - return false; - } - if (config->use_sendfile && (config->use_chunk_serialization || config->use_compression)) { - log_message(LOG_LEVEL_ERROR, "-f/--sendfile cannot be combined with -c (compression) or -s " - "(chunk serialization)"); - return false; - } if (config->compression_threads > 0 && !config->use_compression) { log_message(LOG_LEVEL_ERROR, "--compress-threads requires compression (-c or -z)"); return false; @@ -60,73 +47,11 @@ bool validate_config(const Config* config) { log_message(LOG_LEVEL_ERROR, "-f/--sendfile is not supported with SSH transport"); return false; } - if (config->use_incremental && config->use_chunk_serialization) { - log_message(LOG_LEVEL_ERROR, "--incremental is not supported with -s (chunk serialization)"); - return false; - } /* -4 and -6 are mutually exclusive: a socket address family cannot be both. */ if (config->ipv4 && config->ipv6) { log_message(LOG_LEVEL_ERROR, "-4/--ipv4 and -6/--ipv6 are mutually exclusive"); return false; } - if (config->skip_compress_set && config->use_chunk_serialization) { - log_message(LOG_LEVEL_ERROR, - "--skip-compress cannot be combined with -s (chunk serialization)"); - return false; - } - if (config->use_delta && !config->whole_file && !config->use_incremental) { - log_message(LOG_LEVEL_ERROR, "--delta requires --incremental"); - return false; - } - if (config->use_delta && !config->whole_file && config->use_chunk_serialization) { - log_message(LOG_LEVEL_ERROR, "--delta cannot be combined with -s (chunk serialization)"); - return false; - } - if (config->use_delta && !config->whole_file && config->use_sendfile) { - log_message(LOG_LEVEL_ERROR, "--delta cannot be combined with -f (sendfile)"); - return false; - } - /* --append / --append-verify resume a shorter existing destination by - transmitting only the tail. The resume needs the per-file STATUS_CHECK - handshake (so the dest length is learned), which chunk serialization -s - disables; and whole-file is the opposite intent (send everything), so the - two would silently make the resume pointless. Both are rejected up front - rather than silently degrading to a full transfer. */ - if ((config->append || config->append_verify) && config->use_chunk_serialization) { - log_message(LOG_LEVEL_ERROR, - "--append/--append-verify require the per-file incremental check and cannot be " - "combined with -s (chunk serialization)"); - return false; - } - if ((config->append || config->append_verify) && config->whole_file) { - log_message(LOG_LEVEL_ERROR, - "--append/--append-verify are incompatible with --whole-file (which forces a " - "full transfer)"); - return false; - } - /* --hard-links/-H transmits each later group member as a dedicated per-file - STATUS_HARDLINK frame, which chunk serialization -s does not support; and a - hard-links sibling carries no payload, so the tail-resume of --append is - meaningless for it. Both combinations are rejected up front rather than - silently degrading. */ - if (config->preserve_hard_links && config->use_chunk_serialization) { - log_message(LOG_LEVEL_ERROR, - "--hard-links/-H cannot be combined with -s (chunk serialization)"); - return false; - } - /* -X/-A ride the per-file metadata frame; the buffer-based chunk-serialization - wire format does not carry the xattr block, so the pair is rejected up front - (mirroring -H + -s) rather than silently dropping attributes. */ - if ((config->preserve_xattrs || config->preserve_acls) && config->use_chunk_serialization) { - log_message(LOG_LEVEL_ERROR, - "--xattrs/-X and --acls/-A cannot be combined with -s (chunk serialization)"); - return false; - } - if (config->preserve_hard_links && (config->append || config->append_verify)) { - log_message(LOG_LEVEL_ERROR, - "--hard-links/-H cannot be combined with --append/--append-verify"); - return false; - } if (config->log_file_format && !config->log_file) { log_message(LOG_LEVEL_ERROR, "--log-file-format requires --log-file"); return false; @@ -146,28 +71,12 @@ bool validate_config(const Config* config) { log_message(LOG_LEVEL_ERROR, "sending daemon credentials to a non-local server requires --tls"); return false; } - if (config->delay_updates && config->inplace) { - log_message(LOG_LEVEL_ERROR, "--delay-updates does not work with --inplace"); - return false; - } - if (config->delay_updates && delay_updates_staging_name_conflict(config->backup_dir)) { - log_message(LOG_LEVEL_ERROR, - "--backup-dir is reserved when --delay-updates is active (used for the internal " - "staging directory)"); - return false; - } - if (!config_has_valid_delete_timing(config)) { - log_message(LOG_LEVEL_ERROR, - "--delete-before/--delete-during/--delete-delay/--delete-after select the delete " - "timing; at most one may be given and each implies --delete"); - return false; - } - /* --iconv: reject a malformed CONVERT_SPEC or an unsupported charset name at - startup (a probe iconv_open is attempted), so a typo'd charset never fails - the run mid-transfer with per-file errors. */ - if (!charset_spec_valid(config->iconv_spec)) { - log_message(LOG_LEVEL_ERROR, - "--iconv requires LOCAL[,REMOTE] charset names supported by iconv"); + /* Every cross-field invariant the receiver enforces lives in one shared + predicate so the client and the server can never disagree. The client + reports the specific reason here, before any network I/O. */ + const char* invariants_error = config_invariants_error(config); + if (invariants_error) { + log_message(LOG_LEVEL_ERROR, "%s", invariants_error); return false; } /* --protocol: FastSync has exactly one wire format, so the forced version @@ -180,15 +89,5 @@ bool validate_config(const Config* config) { PROTOCOL_VERSION); return false; } - /* --copy-as pushes the source ids through the metadata path (it implies - --preserve). A later --no-preserve would clear use_metadata, leaving the - transfer with nothing to chown while the receiver gate would still pass. - Refuse the combination up front rather than silently chowning nothing. */ - if (config->copy_as_set && !config->use_metadata) { - log_message(LOG_LEVEL_ERROR, - "--copy-as requires metadata preservation and cannot be combined with " - "--no-preserve"); - return false; - } return true; } diff --git a/src/shared/chunk.c b/src/shared/chunk.c index c5baeb4..1f3be3a 100644 --- a/src/shared/chunk.c +++ b/src/shared/chunk.c @@ -207,17 +207,20 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) { return NULL; char* data_pointer = data->data; size_t remaining_size = data->size; + /* The element currently being parsed is owned by `files` only after the + * array_list_add() at the end of the iteration; until then the error + * epilogue destroys it directly. Keeping this one pointer nulled after the + * hand-off makes the single cleanup path correct for every failure. */ + File* file = NULL; while (remaining_size > 0) { if ((unsigned int)files->size >= MAX_FILES_PER_CHUNK) { log_message(LOG_LEVEL_ERROR, "Chunk contains too many files"); - array_list_delete(files); - return NULL; + goto error; } if (remaining_size < sizeof(size_t)) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for path length"); - array_list_delete(files); - return NULL; + goto error; } size_t path_len; @@ -227,26 +230,19 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) { if (path_len > SIZE_MAX - 1 || remaining_size < path_len) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for path"); - array_list_delete(files); - return NULL; + goto error; } - if (path_len == SIZE_MAX) { - array_list_delete(files); - return NULL; - } char* path = protocol_alloc(path_len + 1); if (path == NULL) { log_perror("Could not allocate memory for file path"); - array_list_delete(files); - return NULL; + goto error; } memcpy(path, data_pointer, path_len); path[path_len] = '\0'; if (memchr(path, '\0', path_len) != NULL) { free(path); - array_list_delete(files); - return NULL; + goto error; } data_pointer += path_len; remaining_size -= path_len; @@ -260,8 +256,7 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) { if (local_path == NULL) { log_message(LOG_LEVEL_ERROR, "--iconv: received chunk file name cannot be converted to the local charset"); - array_list_delete(files); - return NULL; + goto error; } path = local_path; path_len = strlen(path); @@ -269,30 +264,23 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) { if (path_len == 0 || has_path_traversal(path)) { free(path); - array_list_delete(files); - return NULL; + goto error; } - File* file = file_create(path); + file = file_create(path); free(path); - if (file == NULL) { - array_list_delete(files); - return NULL; - } + if (file == NULL) + goto error; if (remaining_size < sizeof(int)) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for entry type"); - file_destroy(file); - array_list_delete(files); - return NULL; + goto error; } int entry_type; memcpy(&entry_type, data_pointer, sizeof(int)); if (entry_type != 0 && entry_type != 1 && entry_type != 2 && entry_type != 3) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: bad entry type"); - file_destroy(file); - array_list_delete(files); - return NULL; + goto error; } file->is_dir = entry_type == 1; file->is_symlink = entry_type == 2; @@ -303,9 +291,7 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) { if (file->is_special) { if (remaining_size < 2 * (int32_t)sizeof(int32_t)) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for special rdev"); - file_destroy(file); - array_list_delete(files); - return NULL; + goto error; } int32_t special_major, special_minor; memcpy(&special_major, data_pointer, sizeof(special_major)); @@ -320,9 +306,7 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) { if (special_major < 0 || special_minor < 0 || special_major > 0xffff || special_minor > 0x00ffffff) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: out-of-range special rdev"); - file_destroy(file); - array_list_delete(files); - return NULL; + goto error; } file->rdev_major = special_major; file->rdev_minor = special_minor; @@ -331,37 +315,32 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) { if (use_metadata) { if (remaining_size < sizeof(int)) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for metadata"); - file_destroy(file); - array_list_delete(files); - return NULL; + goto error; } - // Peek at present flag to determine total size needed before reading + /* Peek at the present flag to determine the total record size before + decoding. metadata_from_buf() independently bounds-checks every read + against remaining_size, so a short body can never over-read. */ int present_flag; memcpy(&present_flag, data_pointer, sizeof(int)); if ((present_flag != 0 && present_flag != 1) || (present_flag == 1 && remaining_size < sizeof(int) + FILE_METADATA_WIRE_SIZE)) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for metadata body"); - file_destroy(file); - array_list_delete(files); - return NULL; + goto error; } - file->metadata = metadata_from_buf(&data_pointer); - remaining_size -= sizeof(int); + file->metadata = metadata_from_buf((const uint8_t*)data_pointer, remaining_size); + size_t metadata_consumed = sizeof(int); if (present_flag == 1) { - if (file->metadata == NULL) { - file_destroy(file); - array_list_delete(files); - return NULL; - } - remaining_size -= FILE_METADATA_WIRE_SIZE; + if (file->metadata == NULL) + goto error; + metadata_consumed += FILE_METADATA_WIRE_SIZE; } + data_pointer += metadata_consumed; + remaining_size -= metadata_consumed; } if (remaining_size < sizeof(size_t)) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for data size"); - file_destroy(file); - array_list_delete(files); - return NULL; + goto error; } size_t file_data_size; @@ -371,35 +350,26 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) { if (remaining_size < file_data_size) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for file content"); - file_destroy(file); - array_list_delete(files); - return NULL; + goto error; } // Reject individual file data larger than the maximum allowed size. if (file_data_size > MAX_FILE_DATA_SIZE) { log_message(LOG_LEVEL_ERROR, "File data size %zu exceeds maximum %llu", file_data_size, (unsigned long long)MAX_FILE_DATA_SIZE); - file_destroy(file); - array_list_delete(files); - return NULL; + goto error; } size_t allocation_size = file_data_size > 0 ? file_data_size : 1; void* file_data = protocol_alloc(allocation_size); if (file_data == NULL) { log_perror("Could not allocate memory for file data"); - file_destroy(file); - array_list_delete(files); - return NULL; + goto error; } memcpy(file_data, data_pointer, file_data_size); Data* replacement = data_create(file_data, file_data_size); - if (replacement == NULL) { - file_destroy(file); - array_list_delete(files); - return NULL; - } + if (replacement == NULL) + goto error; data_destroy(file->data); file->data = replacement; data_pointer += file_data_size; @@ -408,9 +378,7 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) { if (file->is_symlink) { if (remaining_size < sizeof(size_t)) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for symlink target"); - file_destroy(file); - array_list_delete(files); - return NULL; + goto error; } size_t target_len; memcpy(&target_len, data_pointer, sizeof(size_t)); @@ -418,24 +386,18 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) { remaining_size -= sizeof(size_t); if (target_len == 0 || remaining_size < target_len) { log_message(LOG_LEVEL_ERROR, "Invalid chunk format: bad symlink target"); - file_destroy(file); - array_list_delete(files); - return NULL; + goto error; } char* target = protocol_alloc(target_len + 1); if (!target) { log_perror("Could not allocate memory for symlink target"); - file_destroy(file); - array_list_delete(files); - return NULL; + goto error; } memcpy(target, data_pointer, target_len); target[target_len] = '\0'; if (memchr(target, '\0', target_len) != NULL) { free(target); - file_destroy(file); - array_list_delete(files); - return NULL; + goto error; } /* The symlink target also rides the wire charset; decode it to the local charset like the path (a target is a path). */ @@ -446,9 +408,7 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) { log_message(LOG_LEVEL_ERROR, "--iconv: received chunk symlink target cannot be converted to the local " "charset"); - file_destroy(file); - array_list_delete(files); - return NULL; + goto error; } target = local_target; } @@ -457,29 +417,27 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) { remaining_size -= target_len; } - if (!array_list_add(files, file)) { - file_destroy(file); - array_list_delete(files); - return NULL; - } + if (!array_list_add(files, file)) + goto error; + file = NULL; } File** file_array = (File**)array_list_to_array(files); - if (files->size > 0 && file_array == NULL) { - array_list_delete(files); - return NULL; - } + if (files->size > 0 && file_array == NULL) + goto error; Chunk* chunk = chunk_create(file_array, files->size); - free(file_array); - if (chunk == NULL) { - array_list_delete(files); - return NULL; - } + if (chunk == NULL) + goto error; files->item_destroyer = NULL; array_list_delete(files); - return chunk; + +error: + if (file) + file_destroy(file); + array_list_delete(files); + return NULL; } Data* chunk_compress(Chunk* chunk, int compression_level, bool use_metadata) { diff --git a/src/shared/config.c b/src/shared/config.c index 6730639..c0f1412 100644 --- a/src/shared/config.c +++ b/src/shared/config.c @@ -252,6 +252,11 @@ static char* config_receive_str_redacted(int fd, ConfigStringBudget* budget) { } static bool validate_received_config(const Config* config) { + /* Cross-field invariants live in one place (config_invariants_error) so the + receiver enforces every combination the client relies on; a hostile peer + can forge a frame that violates any clause of the shared predicate. */ + if (config_invariants_error(config) != NULL) + return false; return valid_wire_bool(config->save_to_disk) && valid_wire_bool(config->use_multithreading) && valid_wire_bool(config->use_chunk_serialization) && valid_wire_bool(config->use_compression) && valid_wire_bool(config->use_metadata) && @@ -275,30 +280,14 @@ static bool validate_received_config(const Config* config) { valid_wire_bool(config->delete_delay) && valid_wire_bool(config->delete_during) && valid_wire_bool(config->relative) && valid_wire_bool(config->prune_empty_dirs) && valid_wire_bool(config->delay_updates) && valid_wire_bool(config->mkpath) && - !(config->delay_updates && config->inplace) && - !(config->delay_updates && delay_updates_staging_name_conflict(config->backup_dir)) && valid_wire_bool(config->partial) && valid_wire_bool(config->delete_before) && valid_wire_bool(config->checksum) && valid_wire_bool(config->eight_bit_output) && - checksum_algo_valid(config->checksum_algo) && config_has_valid_delete_timing(config) && - identity_wire_valid(config) && - !(config->skip_compress_set && config->use_chunk_serialization) && - /* --append / --append-verify tail resume needs the per-file check, - which chunk serialization -s disables: reject on the receiver too - so a -s sender cannot negotiate an inert append mode. */ - !((config->append || config->append_verify) && config->use_chunk_serialization) && - !(config->preserve_hard_links && config->use_chunk_serialization) && - !(config->preserve_hard_links && (config->append || config->append_verify)) && - /* The xattr block rides the per-file streaming frame, which -s drops. */ - !((config->preserve_xattrs || config->preserve_acls) && config->use_chunk_serialization) && + checksum_algo_valid(config->checksum_algo) && identity_wire_valid(config) && valid_wire_bool(config->preserve_atimes) && valid_wire_bool(config->preserve_crtimes) && valid_wire_bool(config->omit_dir_times) && valid_wire_bool(config->omit_link_times) && valid_wire_bool(config->munge_links) && valid_wire_bool(config->keep_dirlinks) && valid_wire_bool(config->fake_super) && (!config->copy_as_set || (config->copy_as_uid >= 0 && config->copy_as_gid >= 0)) && - /* --copy-as forces ownership through the metadata path; without - metadata it would pass the privilege gate but silently chown - nothing. Refuse the frame instead. */ - (!config->copy_as_set || config->use_metadata) && (!config->use_compression || (config->compression_level >= 1 && config->compression_level <= 22)) && config->chunk_size > 0 && config->chunk_size <= MAX_CHUNK_SIZE && @@ -309,14 +298,6 @@ static bool validate_received_config(const Config* config) { config->skip_compress_count <= MAX_SKIP_COMPRESS_SUFFIXES && config->max_alloc > 0 && (!config->chmod_spec || !*config->chmod_spec || chmod_apply(0, config->chmod_spec, &(mode_t){0})) && - /* The received --iconv CONVERT_SPEC is untrusted input that drives - the receiver's path decoding: reject a malformed spec or an - unsupported charset name so the run is refused up front instead of - every received file name failing mid-transfer. A NULL spec (iconv - disabled) is always accepted. */ - (!config->iconv_spec || charset_spec_valid(config->iconv_spec)) && - /* --super / --no-super: the received tri-state must be one of the - defined values (AUTO/ON/OFF); anything else is a malformed frame. */ config->super_mode >= SUPER_MODE_AUTO && config->super_mode <= SUPER_MODE_OFF; } @@ -348,6 +329,65 @@ bool config_has_valid_delete_timing(const Config* config) { return timing_count <= 1; } +/* The cross-field invariants FastSync relies on, in one place. Every message + * here was previously duplicated (verbatim) in client_validation.c and/or + * config.c; the client reports the returned string for UX and the server + * enforces the same rules at its trust boundary. Pure: no I/O, no logging. + * The order is deliberate (most specific structural conflicts first). */ +const char* config_invariants_error(const Config* config) { + if (!config) + return "Invalid configuration"; + if (config_has_basis(config) && config->use_chunk_serialization) + return "--compare-dest/--copy-dest/--link-dest require per-file incremental checks and cannot " + "be combined with -s (chunk serialization)"; + if (config->use_sendfile && (config->use_chunk_serialization || config->use_compression)) + return "-f/--sendfile cannot be combined with -c (compression) or -s (chunk serialization)"; + if (config->use_incremental && config->use_chunk_serialization) + return "--incremental is not supported with -s (chunk serialization)"; + if (config->skip_compress_set && config->use_chunk_serialization) + return "--skip-compress cannot be combined with -s (chunk serialization)"; + if (config->use_delta && !config->whole_file && !config->use_incremental) + return "--delta requires --incremental"; + if (config->use_delta && !config->whole_file && config->use_chunk_serialization) + return "--delta cannot be combined with -s (chunk serialization)"; + if (config->use_delta && !config->whole_file && config->use_sendfile) + return "--delta cannot be combined with -f (sendfile)"; + /* --append / --append-verify resume a shorter existing destination by + transmitting only the tail. The resume needs the per-file STATUS_CHECK + handshake (so the dest length is learned), which chunk serialization -s + disables; whole-file is the opposite intent (send everything). */ + if ((config->append || config->append_verify) && config->use_chunk_serialization) + return "--append/--append-verify require the per-file incremental check and cannot be " + "combined with -s (chunk serialization)"; + if ((config->append || config->append_verify) && config->whole_file) + return "--append/--append-verify are incompatible with --whole-file (which forces a full " + "transfer)"; + /* -H transmits each later hard-link group member as a dedicated per-file + STATUS_HARDLINK frame, which -s does not support; and a hard-links sibling + carries no payload, so the tail-resume of --append is meaningless. */ + if (config->preserve_hard_links && config->use_chunk_serialization) + return "--hard-links/-H cannot be combined with -s (chunk serialization)"; + /* -X/-A ride the per-file metadata frame; the chunk-serialization wire format + does not carry the xattr block. */ + if ((config->preserve_xattrs || config->preserve_acls) && config->use_chunk_serialization) + return "--xattrs/-X and --acls/-A cannot be combined with -s (chunk serialization)"; + if (config->preserve_hard_links && (config->append || config->append_verify)) + return "--hard-links/-H cannot be combined with --append/--append-verify"; + if (config->delay_updates && config->inplace) + return "--delay-updates does not work with --inplace"; + if (config->delay_updates && delay_updates_staging_name_conflict(config->backup_dir)) + return "--backup-dir is reserved when --delay-updates is active (used for the internal " + "staging directory)"; + if (!config_has_valid_delete_timing(config)) + return "--delete-before/--delete-during/--delete-delay/--delete-after select the delete " + "timing; at most one may be given and each implies --delete"; + if (config->iconv_spec && !charset_spec_valid(config->iconv_spec)) + return "--iconv requires LOCAL[,REMOTE] charset names supported by iconv"; + if (config->copy_as_set && !config->use_metadata) + return "--copy-as requires metadata preservation and cannot be combined with --no-preserve"; + return NULL; +} + bool config_has_basis(const Config* config) { return config && config->basis_count > 0; } diff --git a/src/shared/config.h b/src/shared/config.h index 29b7baf..c7bdb70 100644 --- a/src/shared/config.h +++ b/src/shared/config.h @@ -754,6 +754,16 @@ bool config_delete_timing_early(const Config* config); * set (none = the default delete-after commit timing); without deletion no * timing flag may be set (each timing flag implies --delete). */ bool config_has_valid_delete_timing(const Config* config); + +/* Single source of truth for the cross-field ("combination") invariants a + * Config must satisfy. Returns NULL when `config` is consistent, or a static, + * human-readable error string (no trailing period) describing the FIRST + * violation found. Pure: performs no I/O, no allocation, no logging and no + * printing, so it is safe to call from every trust boundary. The client calls + * it from validate_config() for up-front UX and the server calls it from + * validate_received_config() so the receiver enforces exactly the same + * invariants it relies on (the server is the trust boundary). */ +const char* config_invariants_error(const Config* config); /* True when at least one --compare-dest/--copy-dest/--link-dest was set. */ bool config_has_basis(const Config* config); /* Append one basis-dir entry. Returns 0 on success, -1 on allocation failure. */ diff --git a/src/shared/metadata.c b/src/shared/metadata.c index d1f14e5..a1e5784 100644 --- a/src/shared/metadata.c +++ b/src/shared/metadata.c @@ -90,63 +90,65 @@ void metadata_to_buf(char** buf, const FileMetadata* m) { *buf += sizeof(crtime_nsec); } -FileMetadata* metadata_from_buf(char** buf) { +FileMetadata* metadata_from_buf(const uint8_t* buf, size_t len) { + if (buf == NULL || len < sizeof(int32_t)) + return NULL; int32_t present; - memcpy(&present, *buf, sizeof(present)); - *buf += sizeof(present); - if (present != 0 && present != 1) + memcpy(&present, buf, sizeof(present)); + if (present != 1) return NULL; - if (!present) + if (len < sizeof(int32_t) + FILE_METADATA_WIRE_SIZE) return NULL; + const uint8_t* cursor = buf + sizeof(int32_t); FileMetadata* m = protocol_alloc(sizeof(FileMetadata)); if (m == NULL) return NULL; int32_t mode; - memcpy(&mode, *buf, sizeof(mode)); - *buf += sizeof(mode); + memcpy(&mode, cursor, sizeof(mode)); + cursor += sizeof(mode); m->mode = (mode_t)mode; int32_t uid; - memcpy(&uid, *buf, sizeof(uid)); - *buf += sizeof(uid); + memcpy(&uid, cursor, sizeof(uid)); + cursor += sizeof(uid); m->uid = (uid_t)uid; int32_t gid; - memcpy(&gid, *buf, sizeof(gid)); - *buf += sizeof(gid); + memcpy(&gid, cursor, sizeof(gid)); + cursor += sizeof(gid); m->gid = (gid_t)gid; int64_t mtime_sec; - memcpy(&mtime_sec, *buf, sizeof(mtime_sec)); - *buf += sizeof(mtime_sec); + memcpy(&mtime_sec, cursor, sizeof(mtime_sec)); + cursor += sizeof(mtime_sec); m->mtime_sec = (time_t)mtime_sec; int64_t mtime_nsec; - memcpy(&mtime_nsec, *buf, sizeof(mtime_nsec)); - *buf += sizeof(mtime_nsec); + memcpy(&mtime_nsec, cursor, sizeof(mtime_nsec)); + cursor += sizeof(mtime_nsec); m->mtime_nsec = (long)mtime_nsec; int32_t atime_valid; - memcpy(&atime_valid, *buf, sizeof(atime_valid)); - *buf += sizeof(atime_valid); + memcpy(&atime_valid, cursor, sizeof(atime_valid)); + cursor += sizeof(atime_valid); int64_t atime_sec; - memcpy(&atime_sec, *buf, sizeof(atime_sec)); - *buf += sizeof(atime_sec); + memcpy(&atime_sec, cursor, sizeof(atime_sec)); + cursor += sizeof(atime_sec); int64_t atime_nsec; - memcpy(&atime_nsec, *buf, sizeof(atime_nsec)); - *buf += sizeof(atime_nsec); + memcpy(&atime_nsec, cursor, sizeof(atime_nsec)); + cursor += sizeof(atime_nsec); int32_t crtime_valid; - memcpy(&crtime_valid, *buf, sizeof(crtime_valid)); - *buf += sizeof(crtime_valid); + memcpy(&crtime_valid, cursor, sizeof(crtime_valid)); + cursor += sizeof(crtime_valid); int64_t crtime_sec; - memcpy(&crtime_sec, *buf, sizeof(crtime_sec)); - *buf += sizeof(crtime_sec); + memcpy(&crtime_sec, cursor, sizeof(crtime_sec)); + cursor += sizeof(crtime_sec); int64_t crtime_nsec; - memcpy(&crtime_nsec, *buf, sizeof(crtime_nsec)); - *buf += sizeof(crtime_nsec); + memcpy(&crtime_nsec, cursor, sizeof(crtime_nsec)); + cursor += sizeof(crtime_nsec); m->atime_valid = atime_valid != 0; m->atime_sec = (time_t)atime_sec; m->atime_nsec = (long)atime_nsec; m->crtime_valid = crtime_valid != 0; m->crtime_sec = (time_t)crtime_sec; m->crtime_nsec = (long)crtime_nsec; - if (present != 1 || mtime_nsec < 0 || mtime_nsec >= 1000000000LL || mode < 0 || uid < 0 || - gid < 0 || atime_valid < 0 || atime_valid > 1 || crtime_valid < 0 || crtime_valid > 1 || + if (mtime_nsec < 0 || mtime_nsec >= 1000000000LL || mode < 0 || uid < 0 || gid < 0 || + atime_valid < 0 || atime_valid > 1 || crtime_valid < 0 || crtime_valid > 1 || (atime_valid && (atime_nsec < 0 || atime_nsec >= 1000000000LL)) || (crtime_valid && (crtime_nsec < 0 || crtime_nsec >= 1000000000LL))) { free(m); @@ -196,8 +198,7 @@ FileMetadata* metadata_receive(int file_descriptor, int* ok) { *ok = 0; return NULL; } - char* cursor = packed; - FileMetadata* m = metadata_from_buf(&cursor); + FileMetadata* m = metadata_from_buf((const uint8_t*)packed, sizeof(packed)); if (m == NULL) { if (ok) *ok = 0; diff --git a/src/shared/metadata.h b/src/shared/metadata.h index 0540037..82f3ecc 100644 --- a/src/shared/metadata.h +++ b/src/shared/metadata.h @@ -3,6 +3,7 @@ #include "file.h" #include +#include #include #include #include @@ -39,7 +40,13 @@ #define FILE_METADATA_WIRE_SIZE (sizeof(int32_t) * 5 + sizeof(int64_t) * 6) void metadata_to_buf(char** buf, const FileMetadata* m); -FileMetadata* metadata_from_buf(char** buf); +/* Decode one packed metadata record (an int32 present flag followed, when + * present, by FILE_METADATA_WIRE_SIZE field bytes) from `buf`, which has `len` + * readable bytes. Every read is bounds-checked against `len`, so the function + * can never over-read the caller's buffer: a too-short record, an absent + * (present == 0) record and a malformed record all return NULL. A successful + * decode returns a heap-allocated FileMetadata owned by the caller. */ +FileMetadata* metadata_from_buf(const uint8_t* buf, size_t len); bool metadata_send(int file_descriptor, const FileMetadata* m); FileMetadata* metadata_receive(int file_descriptor, int* ok); void file_restore_metadata(const char* path, const FileMetadata* metadata, diff --git a/tests/fuzz/fuzz_metadata_from_buf.c b/tests/fuzz/fuzz_metadata_from_buf.c index 00d051d..57b2a92 100644 --- a/tests/fuzz/fuzz_metadata_from_buf.c +++ b/tests/fuzz/fuzz_metadata_from_buf.c @@ -5,19 +5,19 @@ #include int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { - if (size < sizeof(int) + FILE_METADATA_WIRE_SIZE) - return 0; - - char* buf = malloc(size); + /* Exercise the bounds-checked decoder on EVERY input length, including + * records shorter than a full metadata body; the decoder must reject those + * without reading past `size`. */ + char* buf = malloc(size > 0 ? size : 1); if (!buf) return 0; - memcpy(buf, data, size); + if (size > 0) + memcpy(buf, data, size); - char* original_buf = buf; - FileMetadata* m = metadata_from_buf(&buf); + FileMetadata* m = metadata_from_buf((const uint8_t*)buf, size); if (m) free(m); - free(original_buf); + free(buf); return 0; } diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c index 786efe9..e5375b9 100644 --- a/tests/test_client_cli.c +++ b/tests/test_client_cli.c @@ -122,6 +122,52 @@ static void test_validate_config_delta_sendfile_constraints() { config_delete(cfg); } +/* The client must still reject every combination now enforced by the shared + config_invariants_error() predicate (the server trusts the same rules). */ +static void test_validate_config_unified_invariants() { + Config* cfg = valid_client_config(); + cfg->use_incremental = true; + cfg->use_delta = true; + cfg->use_chunk_serialization = true; + EXPECT_FALSE(validate_config(cfg)); /* delta + chunk */ + config_delete(cfg); + + cfg = valid_client_config(); + cfg->use_delta = true; /* whole_file false */ + EXPECT_FALSE(validate_config(cfg)); /* delta without incremental */ + config_delete(cfg); + + cfg = valid_client_config(); + cfg->use_sendfile = true; + cfg->use_chunk_serialization = true; + EXPECT_FALSE(validate_config(cfg)); /* sendfile + chunk */ + config_delete(cfg); + + cfg = valid_client_config(); + cfg->preserve_hard_links = true; + cfg->use_chunk_serialization = true; + EXPECT_FALSE(validate_config(cfg)); /* hard-links + chunk */ + config_delete(cfg); + + cfg = valid_client_config(); + cfg->preserve_hard_links = true; + cfg->append = true; + EXPECT_FALSE(validate_config(cfg)); /* hard-links + append */ + config_delete(cfg); + + cfg = valid_client_config(); + cfg->append = true; + cfg->whole_file = true; + EXPECT_FALSE(validate_config(cfg)); /* append + whole-file */ + config_delete(cfg); + + cfg = valid_client_config(); + cfg->preserve_xattrs = true; + cfg->use_chunk_serialization = true; + EXPECT_FALSE(validate_config(cfg)); /* xattrs + chunk */ + config_delete(cfg); +} + /* Test main() with --help flag (early return path, no server connection needed) */ static void test_cli_help() { /* We can't easily call main() because it calls send_files which needs a server. @@ -3155,6 +3201,7 @@ void test_client_cli() { test_validate_config_tls_requirements(); test_validate_config_credentials_require_tls_or_loopback(); test_validate_config_delta_sendfile_constraints(); + test_validate_config_unified_invariants(); test_cli_help(); test_cli_archive_flags(); test_cli_dry_run(); diff --git a/tests/test_config.c b/tests/test_config.c index 003d768..093cf9a 100644 --- a/tests/test_config.c +++ b/tests/test_config.c @@ -2042,6 +2042,156 @@ static void test_super_does_not_imply_numeric() { config_delete(c); } +/* The single shared predicate must reject every cross-field combination the + client/server enforce and accept a plain valid config. Because both + validate_config() (client) and validate_received_config() (server) call it, + this table documents the whole invariant set in one place. */ +static void test_config_invariants_error_all_combinations() { + Config* c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + EXPECT_NULL(config_invariants_error(c)); + config_delete(c); + + c = config_create(); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_COMPARE, "sub"), 0); + c->use_chunk_serialization = true; + EXPECT_NOT_NULL(config_invariants_error(c)); /* basis + chunk */ + config_delete(c); + + c = config_create(); + c->use_sendfile = true; + c->use_compression = true; + EXPECT_NOT_NULL(config_invariants_error(c)); /* sendfile + compression */ + config_delete(c); + + c = config_create(); + c->use_sendfile = true; + c->use_chunk_serialization = true; + EXPECT_NOT_NULL(config_invariants_error(c)); /* sendfile + chunk */ + config_delete(c); + + c = config_create(); + c->use_incremental = true; + c->use_chunk_serialization = true; + EXPECT_NOT_NULL(config_invariants_error(c)); /* incremental + chunk */ + config_delete(c); + + c = config_create(); + c->skip_compress_set = true; + c->use_chunk_serialization = true; + EXPECT_NOT_NULL(config_invariants_error(c)); /* skip-compress + chunk */ + config_delete(c); + + c = config_create(); + c->use_delta = true; /* whole_file false -> active */ + EXPECT_NOT_NULL(config_invariants_error(c)); /* delta without incremental */ + config_delete(c); + + c = config_create(); + c->use_delta = true; + c->use_incremental = true; + c->use_chunk_serialization = true; + EXPECT_NOT_NULL(config_invariants_error(c)); /* delta + chunk */ + config_delete(c); + + c = config_create(); + c->use_delta = true; + c->use_incremental = true; + c->use_sendfile = true; + EXPECT_NOT_NULL(config_invariants_error(c)); /* delta + sendfile */ + config_delete(c); + + c = config_create(); + c->append = true; + c->use_chunk_serialization = true; + EXPECT_NOT_NULL(config_invariants_error(c)); /* append + chunk */ + config_delete(c); + + c = config_create(); + c->append = true; + c->whole_file = true; + EXPECT_NOT_NULL(config_invariants_error(c)); /* append + whole-file */ + config_delete(c); + + c = config_create(); + c->preserve_hard_links = true; + c->use_chunk_serialization = true; + EXPECT_NOT_NULL(config_invariants_error(c)); /* hard-links + chunk */ + config_delete(c); + + c = config_create(); + c->preserve_xattrs = true; + c->use_chunk_serialization = true; + EXPECT_NOT_NULL(config_invariants_error(c)); /* xattrs + chunk */ + config_delete(c); + + c = config_create(); + c->preserve_hard_links = true; + c->append = true; + EXPECT_NOT_NULL(config_invariants_error(c)); /* hard-links + append */ + config_delete(c); + + c = config_create(); + c->delay_updates = true; + c->inplace = true; + EXPECT_NOT_NULL(config_invariants_error(c)); /* delay-updates + inplace */ + config_delete(c); + + c = config_create(); + c->delay_updates = true; + c->backup_dir = str_dup(".fastsync-stage"); + EXPECT_NOT_NULL(config_invariants_error(c)); /* delay-updates staging conflict */ + config_delete(c); + + c = config_create(); + c->delete_delay = true; /* a timing flag without --delete */ + EXPECT_NOT_NULL(config_invariants_error(c)); /* invalid delete timing */ + config_delete(c); + + c = config_create(); + c->iconv_spec = str_dup("no-such-charset,utf-8"); + EXPECT_NOT_NULL(config_invariants_error(c)); /* malformed iconv spec */ + config_delete(c); + + c = config_create(); + c->copy_as_set = true; + c->use_metadata = false; + EXPECT_NOT_NULL(config_invariants_error(c)); /* copy-as without metadata */ + config_delete(c); +} + +/* The receiver previously missed several of these; a forged frame that sets + the offending serialized fields must now be refused at the config + handshake. (whole_file is client-only, so its rules cannot appear here.) */ +static void test_config_receive_rejects_unified_invariants() { + if (is_running_under_valgrind()) + return; + struct { + bool incremental, delta, chunk, sendfile, compression; + } cases[] = { + {true, false, true, false, false}, /* --incremental + -s */ + {false, true, true, false, false}, /* --delta + -s */ + {false, true, false, false, false}, /* --delta without --incremental */ + {false, false, false, true, true}, /* --sendfile + compression */ + {false, false, true, true, false}, /* --sendfile + -s */ + }; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + Config* c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->use_incremental = cases[i].incremental; + c->use_delta = cases[i].delta; + c->use_chunk_serialization = cases[i].chunk; + c->use_sendfile = cases[i].sendfile; + c->use_compression = cases[i].compression; + EXPECT_FALSE(roundtrip_config_ok(c)); + config_delete(c); + } +} + void test_config() { test_config_lifecycle(); test_config_ssh_dest(); @@ -2095,6 +2245,8 @@ void test_config() { test_config_receive_rejects_copy_as_without_metadata(); test_config_receive_rejects_oversized_string_budget(); test_config_receive_with_validate_rejects(); + test_config_invariants_error_all_combinations(); + test_config_receive_rejects_unified_invariants(); } test_identity_copy_as_refused(); test_identity_ownership_requested(); diff --git a/tests/test_fuzz_smoke.c b/tests/test_fuzz_smoke.c index 50d6c29..22efd50 100644 --- a/tests/test_fuzz_smoke.c +++ b/tests/test_fuzz_smoke.c @@ -128,8 +128,7 @@ static void test_fuzz_metadata_from_buf() { EXPECT_EQ_INT((int)(meta_ptr - meta_buf), (int)meta_buf_size); /* Deserialize from buffer (simulates fuzz_metadata_from_buf) */ - char* buf_copy = meta_buf; - FileMetadata* deserialized = metadata_from_buf(&buf_copy); + FileMetadata* deserialized = metadata_from_buf((const uint8_t*)meta_buf, (size_t)meta_buf_size); EXPECT_NOT_NULL(deserialized); EXPECT_EQ_INT((int)deserialized->mode, (int)meta->mode); EXPECT_EQ_INT((int)deserialized->mtime_sec, (int)meta->mtime_sec); diff --git a/tests/test_metadata.c b/tests/test_metadata.c index b6c7870..68a894f 100644 --- a/tests/test_metadata.c +++ b/tests/test_metadata.c @@ -29,8 +29,8 @@ static void test_metadata_to_from_buf_roundtrip() { char* write_ptr = buf; metadata_to_buf(&write_ptr, &original); - char* read_ptr = buf; - FileMetadata* result = metadata_from_buf(&read_ptr); + FileMetadata* result = + metadata_from_buf((const uint8_t*)buf, FILE_METADATA_WIRE_SIZE + sizeof(int)); EXPECT_NOT_NULL(result); EXPECT_EQ_INT(result->mode, 0755); @@ -45,8 +45,6 @@ static void test_metadata_to_from_buf_roundtrip() { EXPECT_EQ_INT(result->crtime_sec, 1200000000); EXPECT_EQ_INT(result->crtime_nsec, 750000000); - EXPECT_EQ_INT((int)(read_ptr - buf), (int)FILE_METADATA_WIRE_SIZE + (int)sizeof(int)); - free(result); free(buf); } @@ -71,14 +69,46 @@ static void test_metadata_from_buf_null() { int present = 0; memcpy(buf, &present, sizeof(int)); - char* read_ptr = buf; - const FileMetadata* result = metadata_from_buf(&read_ptr); + const FileMetadata* result = + metadata_from_buf((const uint8_t*)buf, FILE_METADATA_WIRE_SIZE + sizeof(int)); EXPECT_NULL(result); free(buf); } +/* The decoder must reject (never over-read) a present record that is even one + * byte shorter than the full int32 flag + FILE_METADATA_WIRE_SIZE body, and + * must reject a buffer too short to even hold the present flag. */ +static void test_metadata_from_buf_bounds() { + char* buf = malloc(FILE_METADATA_WIRE_SIZE + sizeof(int)); + EXPECT_NOT_NULL(buf); + FileMetadata original = {.mode = 0644, + .uid = 1, + .gid = 2, + .mtime_sec = 3, + .mtime_nsec = 4, + .atime_valid = true, + .atime_sec = 5, + .atime_nsec = 6, + .crtime_valid = false}; + char* write_ptr = buf; + metadata_to_buf(&write_ptr, &original); + + EXPECT_NULL(metadata_from_buf((const uint8_t*)buf, 0)); + EXPECT_NULL(metadata_from_buf((const uint8_t*)buf, sizeof(int))); + EXPECT_NULL(metadata_from_buf((const uint8_t*)buf, FILE_METADATA_WIRE_SIZE + sizeof(int) - 1)); + /* A buffer larger than the record decodes using only the record prefix. */ + FileMetadata* decoded = + metadata_from_buf((const uint8_t*)buf, FILE_METADATA_WIRE_SIZE + sizeof(int) + 16); + EXPECT_NOT_NULL(decoded); + EXPECT_EQ_INT(decoded->mode, 0644); + free(decoded); + EXPECT_NULL(metadata_from_buf(NULL, FILE_METADATA_WIRE_SIZE + sizeof(int))); + + free(buf); +} + static void test_metadata_send_receive_roundtrip() { io_set_bwlimit(0); int p[2]; @@ -169,10 +199,8 @@ static void test_metadata_wire_is_one_packed_frame() { EXPECT_EQ_INT(avail, 0); /* The present frame decodes in one shot with the shared codec. */ - char* cursor = (char*)wire; - FileMetadata* decoded = metadata_from_buf(&cursor); + FileMetadata* decoded = metadata_from_buf((const uint8_t*)wire, sizeof(wire)); EXPECT_NOT_NULL(decoded); - EXPECT_EQ_INT((int)(cursor - (char*)wire), (int)sizeof(wire)); EXPECT_EQ_INT(decoded->mode, 0640); EXPECT_EQ_INT(decoded->uid, 42); EXPECT_EQ_INT(decoded->gid, 43); @@ -451,6 +479,7 @@ void test_metadata() { test_metadata_to_from_buf_roundtrip(); test_metadata_to_buf_null(); test_metadata_from_buf_null(); + test_metadata_from_buf_bounds(); test_metadata_send_receive_roundtrip(); test_metadata_send_null(); test_metadata_wire_is_one_packed_frame(); From 99f804510599b507568074d278ff0b301aae6c79 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 05:28:32 +0200 Subject: [PATCH 028/155] refactor(scanner,send): embed scanner options; unify stats and config ownership --- src/client/client_send.c | 67 +++++++----- src/client/client_send.h | 5 +- src/client/scanner.c | 165 ++++++++++------------------- src/client/scanner.h | 54 +--------- src/shared/multiprocessing.c | 4 +- src/shared/multiprocessing.h | 2 + tests/integration/test_features.py | 14 +++ tests/test_config.c | 1 + tests/test_multiprocessing.c | 5 +- 9 files changed, 127 insertions(+), 190 deletions(-) diff --git a/src/client/client_send.c b/src/client/client_send.c index bfec60a..e026e53 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -44,6 +44,10 @@ receiver's RECEIVER_QUEUE_MAX_BYTES). */ #define SENDER_QUEUE_MAX_BYTES (MAX_CONNECTION_MEMORY - 2 * MAX_CHUNK_SIZE) +/* One mebibyte in bytes; the unit used by the --stats/--progress lines. + Always cast to double when dividing so the output stays fractional. */ +#define BYTES_PER_MIB (1024ULL * 1024ULL) + /* Forward declaration for progress-reporting thread used in multithreaded send. */ static int progress_thread_fn(void* arg); @@ -51,10 +55,32 @@ static const char* display_bytes(unsigned long long bytes, bool human_readable, size_t buffer_size) { if (human_readable && format_human_bytes(bytes, buffer, buffer_size)) return buffer; - snprintf(buffer, buffer_size, "%.1f MB", bytes / 1048576.0); + snprintf(buffer, buffer_size, "%.1f MB", (double)bytes / (double)BYTES_PER_MIB); return buffer; } +/* Print the canonical `--stats` line. Shared by the single-threaded and + multithreaded send paths so both honor --stats, --human-readable and --quiet + identically; `start` marks the beginning of the transfer for the rate. */ +static void report_transfer_stats(const Config* config, int total_files, + unsigned long long total_bytes, time_t start) { + if (!config->stats || config->quiet) + return; + double elapsed = difftime(time(NULL), start); + double rate = elapsed > 0.0 ? (double)total_bytes / ((double)BYTES_PER_MIB * elapsed) : 0.0; + if (config->human_readable) { + char total_buffer[32]; + char rate_buffer[32]; + fprintf(stderr, "Stats: %d files, %s, %s/s\n", total_files, + display_bytes(total_bytes, true, total_buffer, sizeof(total_buffer)), + display_bytes((unsigned long long)(rate * (double)BYTES_PER_MIB), true, rate_buffer, + sizeof(rate_buffer))); + } else { + fprintf(stderr, "Stats: %d files, %.1f MB, %.1f MB/s\n", total_files, + (double)total_bytes / (double)BYTES_PER_MIB, rate); + } +} + /* Compiled scanner inputs that are shared read-only across scanner instances * and, in -m mode, across worker threads. `base_filters` owns the compiled * command-line + -C rules; the FileListSet allow-set lives in the Config. @@ -688,7 +714,7 @@ static int send_dry_run_manifest(const Config* config) { printf("Total: %d files, %s\n", file_count, display_bytes(total_bytes, true, size_buffer, sizeof(size_buffer))); else - printf("Total: %d files, %.1f MB\n", file_count, total_bytes / 1048576.0); + printf("Total: %d files, %.1f MB\n", file_count, (double)total_bytes / (double)BYTES_PER_MIB); } return 0; } @@ -1425,12 +1451,9 @@ static int send_chunk_with_removal(Client* client, Chunk* chunk, Config* config, return 0; } -int send_chunk(Client* client, Chunk* chunk, Config* config) { - return send_chunk_with_removal(client, chunk, config, NULL); -} - static int send_chunks_multithreaded(void* pipeline_context) { PipelineContextSender* context = (PipelineContextSender*)pipeline_context; + time_t start = time(NULL); Client* client = connect_transfer_client(context->config); if (!client) { if (context->config->transport == TRANSPORT_TCP) @@ -1578,10 +1601,9 @@ static int send_chunks_multithreaded(void* pipeline_context) { int total_files = context->total_files; unsigned long long total_bytes = context->total_bytes; mtx_unlock(&context->mutex_progress); - if (context->config->stats) - fprintf(stderr, "Stats: %d files, %.1f MB\n", total_files, total_bytes / 1048576.0); + report_transfer_stats(context->config, total_files, total_bytes, start); log_info_message(LOG_INFO_STATS, "Transfer summary: %d files, %.1f MB", total_files, - total_bytes / 1048576.0); + (double)total_bytes / (double)BYTES_PER_MIB); disconnect_transfer_client(client); mark_sender_done(context); protocol_session_unbind(); @@ -1754,17 +1776,18 @@ static int load_files_multithreaded(void* pipeline_context) { static void print_transfer_progress(unsigned long long total_bytes, time_t start, const char* suffix, bool human_readable) { double elapsed = difftime(time(NULL), start); - double rate = elapsed > 0.0 ? total_bytes / (1048576.0 * elapsed) : 0.0; + double rate = elapsed > 0.0 ? (double)total_bytes / ((double)BYTES_PER_MIB * elapsed) : 0.0; if (human_readable) { char total_buffer[32]; char rate_buffer[32]; fprintf(stderr, "\rSent %s (%s/s) %s", display_bytes(total_bytes, true, total_buffer, sizeof(total_buffer)), - display_bytes((unsigned long long)(rate * 1048576.0), true, rate_buffer, + display_bytes((unsigned long long)(rate * (double)BYTES_PER_MIB), true, rate_buffer, sizeof(rate_buffer)), suffix); } else { - fprintf(stderr, "\rSent %.1f MB (%.1f MB/s) %s", total_bytes / 1048576.0, rate, suffix); + fprintf(stderr, "\rSent %.1f MB (%.1f MB/s) %s", (double)total_bytes / (double)BYTES_PER_MIB, + rate, suffix); } fflush(stderr); } @@ -2132,23 +2155,9 @@ int send_files(Config* config) { remove_transferred_sources(config, remove_sources); if (config->show_progress && !config->quiet) print_transfer_progress(total_bytes, start, "Done.\n", config->human_readable); - if (config->stats && !config->quiet) { - double elapsed_total = difftime(time(NULL), start); - double rate = elapsed_total > 0 ? total_bytes / (1048576.0 * elapsed_total) : 0; - if (config->human_readable) { - char total_buffer[32]; - char rate_buffer[32]; - fprintf(stderr, "Stats: %d files, %s, %s/s\n", total_files, - display_bytes(total_bytes, true, total_buffer, sizeof(total_buffer)), - display_bytes((unsigned long long)(rate * 1048576.0), true, rate_buffer, - sizeof(rate_buffer))); - } else { - fprintf(stderr, "Stats: %d files, %.1f MB, %.1f MB/s\n", total_files, total_bytes / 1048576.0, - rate); - } - } + report_transfer_stats(config, total_files, total_bytes, start); log_info_message(LOG_INFO_STATS, "Transfer summary: %d files, %.1f MB", total_files, - total_bytes / 1048576.0); + (double)total_bytes / (double)BYTES_PER_MIB); /* --ignore-errors: an unreadable source directory was skipped but the run still completed (and deleted); report the run as errored like rsync does. */ ret = (ok && !had_scan_io) ? 0 : 1; @@ -2237,7 +2246,7 @@ int send_files_multithreaded(Config** config_ptr) { context->missing_args = missing_args; missing_args = NULL; /* owned by the context from here on */ pipeline_context_sender_set_queue_byte_limit(context, SENDER_QUEUE_MAX_BYTES); - *config_ptr = NULL; /* context now owns config through all remaining paths */ + /* The context borrows `config`; the caller (main) still owns and frees it. */ struct timespec now_mono; if (clock_gettime(CLOCK_MONOTONIC, &now_mono) != 0) { now_mono.tv_sec = 0; diff --git a/src/client/client_send.h b/src/client/client_send.h index 0b12973..8c85106 100644 --- a/src/client/client_send.h +++ b/src/client/client_send.h @@ -5,9 +5,10 @@ #include "config.h" #include "transport_tcp.h" -int send_chunk(Client* client, Chunk* chunk, Config* config); +/* Both sender entry points BORROW `config` for the duration of the call; they + * never free it, and the caller retains ownership (freeing it with + * config_delete() once the call returns). */ int send_files(Config* config); -/* Takes ownership only when *config is set to NULL on return. */ int send_files_multithreaded(Config** config); /* Phase 6 residual-batch (client-only). See client_send.c. */ int write_batch_from_source(const Config* config, const char* batch_path); diff --git a/src/client/scanner.c b/src/client/scanner.c index 7db3f58..3a8722d 100644 --- a/src/client/scanner.c +++ b/src/client/scanner.c @@ -177,7 +177,7 @@ static bool entry_passes_selection(const FileListSet* file_list, const FilterRul /* Best-effort capture of the file's whitelisted xattrs (-X/-A). A failure to * read xattrs is non-fatal: the file is transferred without them. */ static void scanner_capture_xattrs(const DirectoryScanner* scanner, File* file) { - if (!scanner || !file || !(scanner->preserve_xattrs || scanner->preserve_acls)) + if (!scanner || !file || !(scanner->options.preserve_xattrs || scanner->options.preserve_acls)) return; file->xattrs = xattr_capture_path(file->path); } @@ -263,10 +263,10 @@ static bool excluded_sink_append(ArrayList* list, mtx_t* mtx, const char* rel) { how manifest keep entries are stored), so the receiver's walker prefixes match the destination layout. An allocation failure is a fatal scan error. */ static void scanner_record_excluded(DirectoryScanner* scanner, const char* fs_path) { - if (!scanner->excluded_paths || !fs_path) + if (!scanner->options.excluded_paths || !fs_path) return; const char* rel = *fs_path == '/' ? fs_path + 1 : fs_path; - if (!excluded_sink_append(scanner->excluded_paths, scanner->excluded_mutex, rel)) + if (!excluded_sink_append(scanner->options.excluded_paths, scanner->options.excluded_mutex, rel)) scanner->failed = true; } @@ -274,7 +274,7 @@ static void scanner_record_excluded(DirectoryScanner* scanner, const char* fs_pa * context, returning the context used for this directory's entries. On a parse * error the scanner is marked failed. Returns 0 on success, -1 on failure. */ static int open_directory_filter_context(DirectoryScanner* scanner, const FilterNode* inherited) { - if (!scanner->per_dir_filters) { + if (!scanner->options.per_dir_filters) { scanner->current_node = (FilterNode*)inherited; return 0; } @@ -436,6 +436,11 @@ DirectoryScanner* directory_scanner_create_with_options(const char* root_directo DirectoryScanner* scanner = calloc(1, sizeof(DirectoryScanner)); if (scanner == NULL) return NULL; + /* One copy of the scan inputs; normalize chunk_size as the old field-by-field + copy did. */ + scanner->options = *options; + if (scanner->options.chunk_size == 0) + scanner->options.chunk_size = DESIRED_CHUNK_SIZE; scanner->directories = queue_create(100, dir_entry_destroy); if (!scanner->directories) { free(scanner); @@ -443,31 +448,7 @@ DirectoryScanner* directory_scanner_create_with_options(const char* root_directo } scanner->current_dir = NULL; scanner->current_path = NULL; - scanner->use_metadata = options->use_metadata; - scanner->preserve_atimes = options->preserve_atimes; - scanner->preserve_crtimes = options->preserve_crtimes; - scanner->preserve_xattrs = options->preserve_xattrs; - scanner->preserve_acls = options->preserve_acls; - scanner->chunk_size = options->chunk_size > 0 ? options->chunk_size : DESIRED_CHUNK_SIZE; - scanner->exclude_patterns = options->exclude_patterns; - scanner->exclude_count = options->exclude_count; - scanner->include_patterns = options->include_patterns; - scanner->include_count = options->include_count; - scanner->max_size = options->max_size; - scanner->min_size = options->min_size; - scanner->max_depth = options->max_depth; scanner->current_depth = 0; - scanner->follow_symlinks = options->follow_symlinks; - scanner->copy_links = options->copy_links; - scanner->safe_links = options->safe_links; - scanner->copy_unsafe_links = options->copy_unsafe_links; - scanner->copy_dirlinks = options->copy_dirlinks; - scanner->munge_links = options->munge_links; - scanner->checksum = options->checksum; - scanner->one_file_system = options->one_file_system; - scanner->preserve_devices = options->preserve_devices; - scanner->preserve_specials = options->preserve_specials; - scanner->copy_devices = options->copy_devices; scanner->failed = false; scanner->root_path = str_dup(root_directory); if (!scanner->root_path) { @@ -479,28 +460,14 @@ DirectoryScanner* directory_scanner_create_with_options(const char* root_directo scanner->at_seed_dir = true; scanner->seed_node = NULL; scanner->current_node = NULL; - scanner->file_list = options->file_list; - scanner->base_filters = options->base_filters; - scanner->per_dir_filters = options->per_dir_filters; - scanner->excluded_paths = options->excluded_paths; - scanner->excluded_mutex = options->excluded_mutex; - scanner->ignore_io_errors = options->ignore_io_errors; - scanner->ignore_missing_args = options->ignore_missing_args; scanner->io_error = false; - scanner->dirs_mode = options->dirs; scanner->relative_mode = options->relative && options->file_list != NULL; - scanner->hardlinks = options->hardlinks; - scanner->prune_empty_dirs = options->prune_empty_dirs; - scanner->stop_condition = options->stop_condition; - scanner->capture_dir_times = options->capture_dir_times; - scanner->dir_entries = options->dir_entries; - scanner->dir_entries_mutex = options->dir_entries_mutex; scanner->dirs_root_emitted = false; scanner->list_index = 0; scanner->dirs_batch = NULL; scanner->dirs_batch_size = 0; scanner->filter_nodes = NULL; - if (scanner->base_filters || scanner->per_dir_filters) { + if (scanner->options.base_filters || scanner->options.per_dir_filters) { scanner->filter_nodes = array_list_create(filter_node_destroy); if (!scanner->filter_nodes) { free(scanner->root_path); @@ -509,7 +476,7 @@ DirectoryScanner* directory_scanner_create_with_options(const char* root_directo return NULL; } } - if (scanner->one_file_system) { + if (scanner->options.one_file_system) { struct stat root_stats; if (stat(root_directory, &root_stats) != 0) { log_perror("Could not stat source directory"); @@ -715,7 +682,7 @@ static int open_next_directory(DirectoryScanner* scanner) { scanner->current_rel = NULL; free(scanner->current_path); scanner->current_path = NULL; - if (!scanner->ignore_io_errors || is_root_seed) { + if (!scanner->options.ignore_io_errors || is_root_seed) { scanner->failed = true; return -1; } @@ -729,10 +696,11 @@ static int open_next_directory(DirectoryScanner* scanner) { scanner->current_path = NULL; return -1; } - if (scanner->capture_dir_times && - !scanner_capture_dir_time(scanner->dir_entries, scanner->dir_entries_mutex, + if (scanner->options.capture_dir_times && + !scanner_capture_dir_time(scanner->options.dir_entries, scanner->options.dir_entries_mutex, scanner->root_path, scanner->current_path, scanner->relative_mode, - scanner->preserve_atimes, scanner->preserve_crtimes)) { + scanner->options.preserve_atimes, + scanner->options.preserve_crtimes)) { closedir(scanner->current_dir); scanner->current_dir = NULL; free(scanner->current_path); @@ -773,9 +741,9 @@ static File* dirs_root_dir_file(DirectoryScanner* scanner) { return NULL; } file->is_dir = true; - if (scanner->use_metadata) { - file->metadata = file_metadata_create(scanner->root_path, &st, scanner->preserve_atimes, - scanner->preserve_crtimes); + if (scanner->options.use_metadata) { + file->metadata = file_metadata_create(scanner->root_path, &st, scanner->options.preserve_atimes, + scanner->options.preserve_crtimes); if (!file->metadata) { file_destroy(file); scanner->failed = true; @@ -808,7 +776,7 @@ static File* dirs_file_for_entry(DirectoryScanner* scanner, const char* entry) { missing argument and is skipped here, exactly as the recursive scan skips nothing (missing entries never appear there). Without the flags it stays a hard pre-transfer error. */ - if (scanner->ignore_missing_args) { + if (scanner->options.ignore_missing_args) { log_info_message(LOG_INFO_MISC, "skipping missing --files-from entry '%s'", entry); free(abs_path); return NULL; @@ -822,8 +790,8 @@ static File* dirs_file_for_entry(DirectoryScanner* scanner, const char* entry) { if (S_ISLNK(link_stats.st_mode)) { /* A symlink is transferred (following its referent) only when a link resolution option is active, mirroring the regular scanner. */ - bool resolve = scanner->follow_symlinks || scanner->copy_links || scanner->safe_links || - scanner->copy_unsafe_links; + bool resolve = scanner->options.follow_symlinks || scanner->options.copy_links || + scanner->options.safe_links || scanner->options.copy_unsafe_links; if (!resolve || stat(abs_path, &effective) != 0) { free(abs_path); return NULL; @@ -851,9 +819,9 @@ static File* dirs_file_for_entry(DirectoryScanner* scanner, const char* entry) { return NULL; } } - if (scanner->use_metadata) { - file->metadata = file_metadata_create(file->path, &effective, scanner->preserve_atimes, - scanner->preserve_crtimes); + if (scanner->options.use_metadata) { + file->metadata = file_metadata_create(file->path, &effective, scanner->options.preserve_atimes, + scanner->options.preserve_crtimes); if (!file->metadata) { file_destroy(file); scanner->failed = true; @@ -884,18 +852,18 @@ static bool dirs_source_dir_is_empty(const char* path) { /* The next File from the --dirs generator, or NULL when exhausted. */ static File* dirs_next_file(DirectoryScanner* scanner) { - if (!scanner->file_list) { + if (!scanner->options.file_list) { if (scanner->dirs_root_emitted) return NULL; scanner->dirs_root_emitted = true; /* --prune-empty-dirs: a physically empty source directory's explicit entry would only create an empty destination directory, so it is omitted. */ - if (scanner->prune_empty_dirs && dirs_source_dir_is_empty(scanner->root_path)) + if (scanner->options.prune_empty_dirs && dirs_source_dir_is_empty(scanner->root_path)) return NULL; return dirs_root_dir_file(scanner); } - while (scanner->list_index < scanner->file_list->count) { - const char* entry = scanner->file_list->entries[scanner->list_index++]; + while (scanner->list_index < scanner->options.file_list->count) { + const char* entry = scanner->options.file_list->entries[scanner->list_index++]; File* file = dirs_file_for_entry(scanner, entry); if (scanner->failed) return NULL; @@ -922,8 +890,9 @@ static Chunk* dirs_flush_batch(DirectoryScanner* scanner) { } static Chunk* directory_scanner_next_dirs(DirectoryScanner* scanner) { - while (scanner->dirs_batch == NULL || scanner->dirs_batch_size <= scanner->chunk_size) { - if (scanner->stop_condition && stop_condition_reached(scanner->stop_condition)) { + while (scanner->dirs_batch == NULL || scanner->dirs_batch_size <= scanner->options.chunk_size) { + if (scanner->options.stop_condition && + stop_condition_reached(scanner->options.stop_condition)) { Chunk* leftover = dirs_flush_batch(scanner); if (leftover) chunk_destroy(leftover); @@ -962,7 +931,7 @@ static Chunk* directory_scanner_next_dirs(DirectoryScanner* scanner) { } Chunk* directory_scanner_next(DirectoryScanner* scanner) { - if (scanner && scanner->dirs_mode) + if (scanner && scanner->options.dirs) return directory_scanner_next_dirs(scanner); ArrayList* chunk_data = array_list_create(file_destroy); if (!chunk_data) { @@ -972,7 +941,8 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { unsigned long long chunk_data_size = 0; while (1) { - if (scanner->stop_condition && stop_condition_reached(scanner->stop_condition)) { + if (scanner->options.stop_condition && + stop_condition_reached(scanner->options.stop_condition)) { array_list_delete(chunk_data); return NULL; } @@ -996,34 +966,9 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) continue; - ScannerOptions options = { - .use_metadata = scanner->use_metadata, - .chunk_size = scanner->chunk_size, - .exclude_patterns = scanner->exclude_patterns, - .exclude_count = scanner->exclude_count, - .include_patterns = scanner->include_patterns, - .include_count = scanner->include_count, - .max_size = scanner->max_size, - .min_size = scanner->min_size, - .max_depth = scanner->max_depth, - .num_threads = 0, - .follow_symlinks = scanner->follow_symlinks, - .copy_links = scanner->copy_links, - .safe_links = scanner->safe_links, - .copy_unsafe_links = scanner->copy_unsafe_links, - .copy_dirlinks = scanner->copy_dirlinks, - .munge_links = scanner->munge_links, - .checksum = scanner->checksum, - .one_file_system = scanner->one_file_system, - .file_list = scanner->file_list, - .base_filters = scanner->base_filters, - .per_dir_filters = scanner->per_dir_filters, - .dirs = false, - .relative = false, - }; ScannerEntry inspected; - int inspection = scanner_inspect_entry(&options, scanner->current_path, scanner->current_path, - entry->d_name, &inspected); + int inspection = scanner_inspect_entry(&scanner->options, scanner->current_path, + scanner->current_path, entry->d_name, &inspected); if (inspection < 0) { scanner->failed = true; break; @@ -1055,16 +1000,17 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { scanner->failed = true; break; } - bool passes_selection = - entry_passes_selection(scanner->file_list, scanner->base_filters, scanner->current_node, - rel, entry->d_name, is_dir, scanner->per_dir_filters); + bool passes_selection = entry_passes_selection( + scanner->options.file_list, scanner->options.base_filters, scanner->current_node, rel, + entry->d_name, is_dir, scanner->options.per_dir_filters); if (!passes_selection) { /* --files-from subset pruning is not a filter exclusion: its delete semantics stay keep-set-only (an unlisted source path is treated as absent, so its destination mirror is a deletable extra). A rule-based exclusion is recorded as a protected prefix. -R + --files-from bare wire paths are never recorded (see ScannerOptions.excluded_paths). */ - bool files_from_prune = scanner->file_list && !file_list_affects(scanner->file_list, rel); + bool files_from_prune = + scanner->options.file_list && !file_list_affects(scanner->options.file_list, rel); if (!files_from_prune && !scanner->relative_mode) scanner_record_excluded(scanner, cur_path); } @@ -1085,12 +1031,13 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { if (is_dir) { free(rel_copy); - if (!scanner_same_filesystem(scanner->one_file_system, scanner->root_dev, stats.st_dev)) { + if (!scanner_same_filesystem(scanner->options.one_file_system, scanner->root_dev, + stats.st_dev)) { free(cur_path); continue; } int next_depth = scanner->current_depth + 1; - if (scanner->max_depth <= 0 || next_depth < scanner->max_depth) { + if (scanner->options.max_depth <= 0 || next_depth < scanner->options.max_depth) { DirEntry* de = dir_entry_create(cur_path, next_depth, scanner->current_node); if (!de || !queue_enqueue(scanner->directories, de)) { dir_entry_destroy(de); @@ -1099,7 +1046,8 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { } free(cur_path); } else { - if (scanner->max_depth > 0 && scanner->current_depth + 1 > scanner->max_depth) { + if (scanner->options.max_depth > 0 && + scanner->current_depth + 1 > scanner->options.max_depth) { free(rel_copy); free(cur_path); continue; @@ -1126,13 +1074,14 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { } /* --devices/--specials: a device/FIFO/socket entry marked for preservation becomes a node to recreate (is_special, no data, rdev captured). */ - scanner_prepare_special(scanner->preserve_devices, scanner->preserve_specials, file, &stats); - if (scanner->hardlinks && S_ISREG(stats.st_mode)) - scanner_assign_hardlink(scanner, scanner->hardlinks, file, &stats); - if (scanner->use_metadata) - file->metadata = file_metadata_create(file->path, &stats, scanner->preserve_atimes, - scanner->preserve_crtimes); - if (scanner->use_metadata && !file->metadata) { + scanner_prepare_special(scanner->options.preserve_devices, scanner->options.preserve_specials, + file, &stats); + if (scanner->options.hardlinks && S_ISREG(stats.st_mode)) + scanner_assign_hardlink(scanner, scanner->options.hardlinks, file, &stats); + if (scanner->options.use_metadata) + file->metadata = file_metadata_create(file->path, &stats, scanner->options.preserve_atimes, + scanner->options.preserve_crtimes); + if (scanner->options.use_metadata && !file->metadata) { free(rel_copy); file_destroy(file); scanner->failed = true; @@ -1147,7 +1096,7 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { break; } chunk_data_size += file->data->size; - if (chunk_data_size > scanner->chunk_size) { + if (chunk_data_size > scanner->options.chunk_size) { free(rel_copy); Chunk* result = chunk_data_to_chunk(chunk_data); if (!result) @@ -1211,7 +1160,7 @@ static int parallel_worker_thread(void* arg) { free(ds->root_path); ds->root_path = str_dup(wa->root_dir); ds->seed_node = wa->ps->root_filter_node; - ds->excluded_mutex = &wa->ps->result_mutex; + ds->options.excluded_mutex = &wa->ps->result_mutex; Chunk* chunk; while ((chunk = directory_scanner_next(ds)) != NULL) { if (!queue_enqueue_multithreaded_cancel(wa->ps->result_queue, chunk, &wa->ps->result_mutex, diff --git a/src/client/scanner.h b/src/client/scanner.h index 950a14e..3db1ce9 100644 --- a/src/client/scanner.h +++ b/src/client/scanner.h @@ -117,37 +117,16 @@ typedef struct { typedef struct FilterNode FilterNode; typedef struct { + /* Scan inputs, copied once at create time. Everything that is also a + ScannerOptions field lives here (with the normalized chunk_size); only + scanner-owned bookkeeping stays as direct members below. */ + ScannerOptions options; Queue* directories; DIR* current_dir; char* current_path; - bool use_metadata; - bool preserve_atimes; - bool preserve_crtimes; - bool preserve_xattrs; - bool preserve_acls; - unsigned long long chunk_size; - char** exclude_patterns; - int exclude_count; - char** include_patterns; - int include_count; - unsigned long long max_size; - unsigned long long min_size; - int max_depth; int current_depth; - bool follow_symlinks; - bool copy_links; - bool safe_links; - bool copy_unsafe_links; - bool copy_dirlinks; - bool munge_links; - bool checksum; - bool one_file_system; dev_t root_dev; bool failed; - /* Phase 4 special/devices (see ScannerOptions). */ - bool preserve_devices; - bool preserve_specials; - bool copy_devices; /* Phase 2 (files-from / filter layer). */ char* root_path; /* transfer root (fs path) for rel computation */ char* current_rel; /* rel path of the open directory ("" == root) */ @@ -155,40 +134,17 @@ typedef struct { FilterNode* seed_node; /* inherited context of the seed dir, or NULL */ FilterNode* current_node; /* filter context of the open directory */ ArrayList* filter_nodes; /* owned FilterNode arena (may be NULL) */ - const FileListSet* file_list; - const FilterRuleList* base_filters; - bool per_dir_filters; - /* --dirs / -R state for the directory-entry generator (dirs_mode replaces + /* --dirs / -R state for the directory-entry generator (options.dirs replaces the recursive scan). */ - bool dirs_mode; bool relative_mode; /* file_list && relative: send bare relative wire paths */ - bool prune_empty_dirs; bool dirs_root_emitted; int list_index; ArrayList* dirs_batch; /* owned when non-NULL */ unsigned long long dirs_batch_size; - /* Excluded-path sink (see ScannerOptions). `excluded_mutex` is shared across - parallel worker threads. */ - ArrayList* excluded_paths; - mtx_t* excluded_mutex; - /* --ignore-errors: continue past unreadable directories (records io_error). */ - bool ignore_io_errors; - /* --ignore-missing-args: --dirs listed-but-missing entries are skipped, not - fatal (see ScannerOptions.ignore_missing_args). */ - bool ignore_missing_args; /* A directory could not be opened (I/O error, e.g. EACCES). With --ignore-errors the scan continues past it and the caller decides what to do; `failed` is reserved for fatal errors that always abort the scan. */ bool io_error; - /* --hard-links (-H): shared link-group detection table (see ScannerOptions). - NULL when -H is off. */ - HardLinkTable* hardlinks; - /* Phase 6: sender stop deadline (from ScannerOptions). */ - const StopCondition* stop_condition; - /* P7 Wave D directory-time capture (see ScannerOptions). */ - bool capture_dir_times; - ArrayList* dir_entries; - mtx_t* dir_entries_mutex; } DirectoryScanner; typedef struct { diff --git a/src/shared/multiprocessing.c b/src/shared/multiprocessing.c index 697e461..14ad46e 100644 --- a/src/shared/multiprocessing.c +++ b/src/shared/multiprocessing.c @@ -179,6 +179,9 @@ bool pipeline_context_sender_enqueue_chunk(PipelineContextSender* context, Chunk } void pipeline_context_sender_destroy(PipelineContextSender* context) { + /* `config` is borrowed: the caller retains ownership and frees it after the + pipeline has been destroyed (the worker threads are already joined, so no + config access can outlive this call). */ if (context->manifest) { array_list_delete(context->manifest); } @@ -192,7 +195,6 @@ void pipeline_context_sender_destroy(PipelineContextSender* context) { array_list_delete(context->dir_entries); if (context->dir_entries_mutex_init) mtx_destroy(&context->dir_entries_mutex); - config_delete(context->config); queue_destroy(context->queue_scanner); queue_destroy(context->queue_loader); mtx_destroy(&context->mutex_scanner); diff --git a/src/shared/multiprocessing.h b/src/shared/multiprocessing.h index 73162a4..6de2e98 100644 --- a/src/shared/multiprocessing.h +++ b/src/shared/multiprocessing.h @@ -120,6 +120,8 @@ typedef struct PipelineContextReceiver { DirTimeList dir_times; } PipelineContextReceiver; +/* `config` is borrowed and must outlive the context: destroy does NOT free it, + so the caller owns it and frees it with config_delete() afterwards. */ PipelineContextSender* pipeline_context_sender_create(Config* config, Queue* queue_scanner, Queue* queue_loader); void pipeline_context_sender_destroy(PipelineContextSender* context); diff --git a/tests/integration/test_features.py b/tests/integration/test_features.py index bdbec46..7e2ea6e 100644 --- a/tests/integration/test_features.py +++ b/tests/integration/test_features.py @@ -1239,6 +1239,20 @@ class TestProgress: assert "Stats:" in result.stderr assert "KB" in result.stderr + def test_human_readable_stats_multithreaded(self, shared_server): + # The multithreaded sender shares the single-threaded --stats format, + # including --human-readable and the rate suffix. + clean_dir(DEST_DIR) + result, dur = run_client( + SOURCE_DIR, DEST_DIR, + flags=["--threads", "-h", "--stats"], + port=shared_server.port, + ) + assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:100]}" + assert "Stats:" in result.stderr + assert "KB" in result.stderr + assert "/s" in result.stderr + def test_human_readable_progress_multithreaded(self, shared_server): clean_dir(DEST_DIR) result, dur = run_client( diff --git a/tests/test_config.c b/tests/test_config.c index 003d768..66fafc0 100644 --- a/tests/test_config.c +++ b/tests/test_config.c @@ -383,6 +383,7 @@ static void test_pipeline_sender_lifecycle() { EXPECT_EQ_INT((int)pcs->allocation_session.max_alloc, (int)cfg->max_alloc); pipeline_context_sender_destroy(pcs); + config_delete(cfg); /* the context borrows cfg; the caller owns it */ } static void test_pipeline_receiver_lifecycle() { diff --git a/tests/test_multiprocessing.c b/tests/test_multiprocessing.c index 427b5c6..a051897 100644 --- a/tests/test_multiprocessing.c +++ b/tests/test_multiprocessing.c @@ -37,6 +37,7 @@ static void test_sender_create_destroy() { EXPECT_NULL(ctx->manifest); pipeline_context_sender_destroy(ctx); + config_delete(cfg); /* the context borrows cfg; the caller owns it */ } /* Test pipeline_context_receiver_create/destroy with valid arguments */ @@ -80,6 +81,7 @@ static void test_sender_queue_capacities() { EXPECT_EQ_INT(ctx->queue_scanner->capacity, 1); EXPECT_EQ_INT(ctx->queue_loader->capacity, 1); pipeline_context_sender_destroy(ctx); + config_delete(cfg); /* the context borrows cfg; the caller owns it */ } /* Invalid queue capacities must not create unusable pipeline queues. */ @@ -445,8 +447,9 @@ static void test_sender_enqueue_byte_budget() { EXPECT_EQ_INT((int)ctx->queued_bytes, 2000); /* second payload now in flight */ /* pipeline_context_sender_destroy frees the still-queued second chunk and - owns cfg/q_scanner/q_loader from here on. */ + owns q_scanner/q_loader; cfg stays borrowed and is freed by the caller. */ pipeline_context_sender_destroy(ctx); + config_delete(cfg); } void test_multiprocessing() { From dcc78c14c500a68a9429a58e27c0afb51198fd15 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 05:48:27 +0200 Subject: [PATCH 029/155] docs,fuzz: fix ownership/alloc comments; fuzz chunk metadata path --- src/client/client_cli.c | 3 ++- src/shared/config.h | 5 +++-- tests/fuzz/fuzz_chunk_deserialize.c | 6 +++++- 3 files changed, 10 insertions(+), 4 deletions(-) diff --git a/src/client/client_cli.c b/src/client/client_cli.c index 27eb93f..43136af 100644 --- a/src/client/client_cli.c +++ b/src/client/client_cli.c @@ -2113,7 +2113,8 @@ int main(int argc, char* argv[]) { to nor transfers to a server. --write-batch runs the normal live transfer AND then emits the batch FILE from a separate deterministic scan pass. It drives the single-threaded transfer so the config outlives the run for that - second pass (the -m path takes ownership of the config). */ + second pass (main retains ownership of the config; every send path + borrows it). */ if (config->read_batch) { exit_code = apply_batch_to_dest(config, config->read_batch, config->receive_root_directory); goto cleanup; diff --git a/src/shared/config.h b/src/shared/config.h index c7bdb70..b0ae890 100644 --- a/src/shared/config.h +++ b/src/shared/config.h @@ -758,8 +758,9 @@ bool config_has_valid_delete_timing(const Config* config); /* Single source of truth for the cross-field ("combination") invariants a * Config must satisfy. Returns NULL when `config` is consistent, or a static, * human-readable error string (no trailing period) describing the FIRST - * violation found. Pure: performs no I/O, no allocation, no logging and no - * printing, so it is safe to call from every trust boundary. The client calls + * violation found. No I/O, no logging and no printing, so it is safe to call + * from every trust boundary; the iconv rule does invoke charset_spec_valid + * (which parses via str_dup/iconv_open), so it is not allocation-free. The client calls * it from validate_config() for up-front UX and the server calls it from * validate_received_config() so the receiver enforces exactly the same * invariants it relies on (the server is the trust boundary). */ diff --git a/tests/fuzz/fuzz_chunk_deserialize.c b/tests/fuzz/fuzz_chunk_deserialize.c index 444a297..f1c7d2a 100644 --- a/tests/fuzz/fuzz_chunk_deserialize.c +++ b/tests/fuzz/fuzz_chunk_deserialize.c @@ -17,7 +17,11 @@ int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { if (!d) return 0; - Chunk* chunk = chunk_deserialize(d, false); + /* Exercise both the metadata and non-metadata chunk layouts: the + metadata branch (present flag + 4-vs-72 advance) is only reachable with + use_metadata=true, so base the choice on the input rather than hardcoding + false. */ + Chunk* chunk = chunk_deserialize(d, (data[0] & 1) != 0); if (chunk) chunk_destroy(chunk); From 1fa2fbd2669cae18e2087c95073f41b508d66d00 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 06:22:28 +0200 Subject: [PATCH 030/155] feat(client): --port alias, --threads=N, graceful abort, keepalive --- README.md | 10 +-- RSYNC_COMPAT.md | 4 +- src/client/client_cli.c | 105 ++++++++++++++++++++++++++---- src/client/client_send.c | 54 +++++++++++++++- src/client/client_send.h | 9 +++ src/client/scanner.h | 4 ++ src/client/usage.c | 7 +- src/shared/config.c | 7 +- src/shared/config.h | 12 ++-- src/shared/protocol.c | 135 +++++++++++++++++++++++++++++++++++++++ src/shared/protocol.h | 19 ++++++ tests/test_client_cli.c | 78 ++++++++++++++++++++++ tests/test_protocol.c | 93 +++++++++++++++++++++++++++ 13 files changed, 501 insertions(+), 36 deletions(-) diff --git a/README.md b/README.md index 6cd07d0..9820885 100644 --- a/README.md +++ b/README.md @@ -104,7 +104,7 @@ partial, alternate, and planned behavior. | `-c, --checksum` | Verify content by checksum instead of size+mtime | | `-z, --compress [level]` | Enable streaming zstd compression (level 1–22, default 5) | | `-a, --archive` | rsync archive mode (`-rlptgoD`): links, metadata, devices and specials (not compression/multithreading) | -| `-j, --threads` | Multithreading mode | +| `-j, --threads[=N]` | Multithreading mode; `N` (1–256) sets the parallel scanner worker count, bare `-j`/`--threads` uses the default | | `-m` | rsync `--prune-empty-dirs` (short form now rsync-parity) | | `--chunk-serialization` | Chunk serialization (batch all files per chunk; long form only) | | `-s` | rsync `--secluded-args` compatibility no-op (remote SSH argv is already injection-safe) | @@ -202,7 +202,7 @@ transfer is never aborted. 3. **FileMetadata** — `mode`, `uid`, `gid`, `mtime_sec`, `mtime_nsec`; uid / gid are advisory wire fields and are never applied by the receiver; atime is unsupported -4. **Config** — runtime parameters (transported over wire, TLS settings excluded). Includes `timeout`, `contimeout`, `quiet`, `backup`, `backup_dir`, `stats`, `max_depth`, `log_file`, `queue_size`. +4. **Config** — runtime parameters (transported over wire, TLS settings excluded). Includes `timeout`, `contimeout`, `quiet`, `backup`, `backup_dir`, `stats`, `max_depth`, `log_file`. 5. **Queue** — thread-safe bounded queue with condition variables 6. **DirectoryScanner** — recursive BFS traversal with exclude and include pattern support, max-depth enforcement @@ -378,7 +378,7 @@ features without changing the meaning of ordinary compatibility options. | Option | Purpose | |---|---| -| `-j`, `--threads` | Enable the multithreaded scanner/loader/sender pipeline. | +| `-j`, `--threads[=N]` | Enable the multithreaded scanner/loader/sender pipeline. `N` (1–256) sets the parallel scanner worker count; bare `-j`/`--threads` uses the default. | | `-z [level]`, `--compress [level]` | Enable streaming zstd compression, levels 1-22. | | `--compress-level ` | Set the zstd compression level. | | `--zc ` | Alias for `--compress-choice`. FastSync supports `zstd` and `none`. | @@ -392,7 +392,7 @@ features without changing the meaning of ordinary compatibility options. | `--delta-block ` | Set the FastSync delta block size (`--block-size` is an alias). | | `--delta-max ` | Limit files eligible for FastSync delta transfer. | | `--server-host ` | Select the TCP server host. | -| `--server-port ` | Select the TCP server port. | +| `--server-port ` | Select the TCP server port (`--port ` and `--port=` are rsync-friendly aliases). | | `--tls` | Enable TLS for TCP transport. | | `--bwlimit ` | Apply token-bucket bandwidth limiting. | | `--progress` | Show transfer progress and throughput. | @@ -478,7 +478,7 @@ link-target transfer remains incomplete. | | `--dest-dir ` | Set the destination directory explicitly. | | `--save-to-disk` | Enable server-side disk persistence. | | `--server-host ` | TCP server address. | -| `--server-port ` | TCP server port. | +| `--server-port ` | TCP server port. `--port ` / `--port=` is an alias. | | `--tls` | Enable TLS. Requires `--cert` and `--key`. | | `--cert ` | TLS certificate file. | | `--key ` | TLS private key file. | diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 6fb91c8..ea171bc 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -612,7 +612,7 @@ now transmits targets (the prior behavior was broken/partial); its status moved |------|-------------------|-----------------|-------| | `-e`, `--rsh=COMMAND` | Remote shell to use | ✅ Implemented | `-e`/`--rsh` (and `--rsh=COMMAND`) select the remote-shell program used to build the SSH child argv, overriding the default `ssh`. The command is whitespace-split into the leading argv words so rsync's `-e "ssh -p 2222"` works; the standard `-o` family, an optional `-p` port, `user@host` and the quoted remote command (`fastsync-server --stdio`) follow. Stored in the `rsh_command` config field. **Client-only, never crosses the wire** (it is a launch concern, not a handshake property) | | `--rsync-path=PROGRAM` | rsync binary on remote | ✅ Implemented | Alias for `--fastsync-server-path`: both write the `fastsync_server_path` config field used as the remote-side server program (always quoted as one remote-shell word), which CROSSES the wire as before. Kept separate from `--rsh`, which names the local connecting program | -| `--port=PORT` | Alternate daemon port | ✅ Implemented | rsync's daemon-port flag maps to the client-side `server_port` config field: a client connects to a TCP/TLS server (incl. `host::module/path` daemon destinations) with `--server-port`, and the `fastsync-server --daemon` listener's port is taken from its config's `port` key (default 873) or overridden by `--dparam port=` / `-p` | +| `--port=PORT`, `--port PORT` | Alternate daemon port | ✅ Implemented | rsync's daemon-port flag is an alias for `--server-port`: both spellings (and `--server-port=PORT`) map to the client-side `server_port` config field. The client connects to a TCP/TLS server (incl. `host::module/path` daemon destinations) on that port, and the `fastsync-server --daemon` listener's port is taken from its config's `port` key (default 873) or overridden by `--dparam port=` / `-p` | | `--sockopts=OPTIONS` | Custom TCP options | ✅ Implemented | Comma-separated allowlist of `OPT=VAL` applied via `setsockopt` after `socket()` before `connect()`/`bind()`. Only `TCP_NODELAY`, `SO_KEEPALIVE`, `SO_REUSEADDR` (0/1) and `SO_RCVBUF`/`SO_SNDBUF` (byte count) are accepted; an unknown option name or a bad value is rejected up front, never silently ignored. A value is required for every option (`OPT=VAL`; a bare name is an error). Applied to the outgoing TCP and TLS client socket; absent by default. `SockOptEntry`/`sockopts` config fields. Local socket concern: never crosses the wire | | `--blocking-io` | Use blocking I/O for remote shell | ✅ Implemented | With `--blocking-io` the SSH-transport socketpair socket is left without `SO_RCVTIMEO`/`SO_SNDTIMEO`, so the transfer blocks naturally; by default it gets the same read/write timeout as the TCP transport (see `--timeout`). `blocking_io` config bool. **Client-only, never crosses the wire** | | `--outbuf=N\|L\|B` | Set output buffering | ✅ Implemented | `N` (none/unbuffered) → `_IONBF`, `L` (line) → `_IOLBF`, `B` (block, the default) → `_IOFBF` via `setvbuf` on stdout and stderr. Garbage values are rejected. `outbuf` config field (`OutbufMode`). **Client-only, never crosses the wire** | @@ -893,7 +893,7 @@ Ranked by user demand, implementation complexity, and interoperability impact (_ | Feature | Description | |---------|-------------| -| `-j` / `--threads` | Multithreaded pipeline (scanner/loader/sender) (renamed from `-m` in Phase 7 Wave A; `-m` is now rsync `--prune-empty-dirs`) | +| `-j` / `--threads[=N]` | Multithreaded pipeline (scanner/loader/sender); `N` (1–256) sizes the parallel scanner worker pool, bare `-j`/`--threads` uses the built-in default (renamed from `-m` in Phase 7 Wave A; `-m` is now rsync `--prune-empty-dirs`) | | `--chunk-serialization` | Chunk serialization mode (long form only; `-s` is now rsync `--secluded-args`) | | `--sendfile` | Zero-copy sendfile() syscall (TCP only) (long form only; `-f` is now rsync `--filter`) | | `-z [level]` / `--compress` | zstd compression level (1-22) (`-c` is now rsync `--checksum`) | diff --git a/src/client/client_cli.c b/src/client/client_cli.c index 43136af..73812ef 100644 --- a/src/client/client_cli.c +++ b/src/client/client_cli.c @@ -12,6 +12,7 @@ #include "identity.h" #include "log.h" #include "protocol.h" +#include "scanner.h" #include "stop_condition.h" #include "transport_tcp.h" #include "transport_tls.h" @@ -27,6 +28,26 @@ #include #include +/* Async-signal-safe abort flag set by the SIGINT/SIGTERM handler. Exposed via + * client_send.h so the send loops can poll it. Defined here (not in + * client_send.c) so the unit-test binary, which compiles this file but not + * client_send.c, still links the symbol. */ +volatile sig_atomic_t client_abort_requested = 0; + +#ifndef FASTSYNC_TEST_BUILD +/* Signal handler: perform NO work beyond storing the flag. Logging, protocol + * I/O and the STATUS_ABORT frame are all done later on the normal send path, + * which is not async-signal-safe. Only the production client installs it. */ +static void client_signal_handler(int signo) { + (void)signo; + client_abort_requested = 1; +} +#endif + +bool client_abort_pending(void) { + return client_abort_requested != 0; +} + #ifndef FASTSYNC_TEST_BUILD /* Parse environment variables for source/destination directories and save-to-disk flag. */ static void parse_environment(const char** out_env_source, const char** out_env_dest, @@ -141,6 +162,23 @@ static int set_checksum_seed(Config* config, const char* value) { return 0; } +/* --threads=N: enable the -m pipeline and size its parallel scanner pool. + * Rejects a non-positive, non-numeric or oversized value up front. */ +static int set_scanner_threads_option(Config* config, const char* value) { + int threads; + if (!parse_positive_int(value, &threads)) { + log_message(LOG_LEVEL_ERROR, "--threads must be a positive integer"); + return -1; + } + if (threads > MAX_SCANNER_THREADS) { + log_message(LOG_LEVEL_ERROR, "--threads must be between 1 and %d", MAX_SCANNER_THREADS); + return -1; + } + config->use_multithreading = true; + config->scanner_threads = threads; + return 0; +} + static int set_compression_threads_option(int* dest, const char* value) { if (set_positive_int_option(dest, value, "--compress-threads") != 0) return -1; @@ -1288,10 +1326,21 @@ static bool cli_handle_transfer_flags(CliParseCtx* ctx) { return true; } if (opt_is(arg, "-j", "--threads")) { + /* Bare -j/--threads: enable the pipeline with the scanner's built-in + worker default (scanner_threads stays 0). */ config->use_multithreading = true; log_info_message(LOG_INFO_MISC, "Enabled Multithreading"); return true; } + if (strncmp(arg, "--threads=", 10) == 0) { + if (set_scanner_threads_option(config, arg + 10) != 0) { + ctx->exit_code = -1; + } else { + log_info_message(LOG_INFO_MISC, "Enabled Multithreading with %d scanner threads", + config->scanner_threads); + } + return true; + } if (opt_is(arg, "--chunk-serialization", NULL)) { config->use_chunk_serialization = true; log_info_message(LOG_INFO_MISC, "Enabled Chunk Serialization"); @@ -1300,29 +1349,48 @@ static bool cli_handle_transfer_flags(CliParseCtx* ctx) { return false; } -/* Network/IO options: --server-port, --bwlimit, --chunk-size, --log-file and - * --stderr. Returns true when the argument was consumed. */ +/* Parse and validate a TCP server port (--server-port, or its rsync-friendly + * alias --port). Returns 0 on success, -1 (with a message) on a malformed or + * out-of-range value. */ +static int set_server_port_option(Config* config, const char* value, const char* option_name) { + int port; + if (!parse_positive_int(value, &port)) { + char* escaped = output_escape(value, false); + log_message(LOG_LEVEL_ERROR, "invalid %s value: %s", option_name, + escaped ? escaped : ""); + free(escaped); + return -1; + } + if (port > 65535) { + log_message(LOG_LEVEL_ERROR, "server port must be 1-65535"); + return -1; + } + config->server_port = port; + return 0; +} + +/* Network/IO options: --server-port/--port, --bwlimit, --chunk-size, --log-file + * and --stderr. Returns true when the argument was consumed. */ static bool cli_handle_io_options(CliParseCtx* ctx) { Config* config = ctx->config; const char* arg = ctx->argv[ctx->i]; - if (opt_is(arg, "--server-port", NULL)) { + if (opt_is(arg, "--server-port", "--port")) { if (ctx->i + 1 >= ctx->argc) { log_message(LOG_LEVEL_ERROR, "missing argument for %s", arg); ctx->exit_code = -1; return true; } - if (!parse_positive_int(ctx->argv[++ctx->i], &config->server_port)) { - char* escaped = output_escape(ctx->argv[ctx->i], false); - log_message(LOG_LEVEL_ERROR, "invalid --server-port value: %s", - escaped ? escaped : ""); - free(escaped); + if (set_server_port_option(config, ctx->argv[++ctx->i], arg) != 0) ctx->exit_code = -1; - return true; - } - if (config->server_port > 65535) { - log_message(LOG_LEVEL_ERROR, "server port must be 1-65535"); + return true; + } + /* rsync users commonly write --port=NNNN; --server-port=NNNN is accepted too + * so both spellings behave identically. */ + if (strncmp(arg, "--server-port=", 14) == 0 || strncmp(arg, "--port=", 7) == 0) { + const char* option_name = strncmp(arg, "--server-port=", 14) == 0 ? "--server-port" : "--port"; + const char* value = arg + (strncmp(arg, "--server-port=", 14) == 0 ? 14 : 7); + if (set_server_port_option(config, value, option_name) != 0) ctx->exit_code = -1; - } return true; } if (opt_is(arg, "--bwlimit", NULL)) { @@ -1975,6 +2043,17 @@ int main(int argc, char* argv[]) { oversized delta). Ignore SIGPIPE so that a broken TCP connection surfaces as a clean write error instead of killing the client. */ signal(SIGPIPE, SIG_IGN); + /* Ctrl-C / SIGTERM: set the abort flag so the send loops can send + * STATUS_ABORT and let the receiver clean up, instead of dying abruptly. + * No SA_RESTART so an in-flight poll()/read() is interrupted (EINTR), which + * lets the keepalive/abort checks observe the flag promptly. */ + struct sigaction abort_action; + memset(&abort_action, 0, sizeof(abort_action)); + abort_action.sa_handler = client_signal_handler; + sigemptyset(&abort_action.sa_mask); + abort_action.sa_flags = 0; + sigaction(SIGINT, &abort_action, NULL); + sigaction(SIGTERM, &abort_action, NULL); const char* env_source = NULL; const char* env_dest = NULL; bool save_to_disk = false; diff --git a/src/client/client_send.c b/src/client/client_send.c index e026e53..1febec8 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -885,6 +885,10 @@ static int send_delete_manifest(int fd, ArrayList* manifest, ArrayList* protecte unlinks) before replying, so the wait uses a generous explicit deadline instead of the default 60 s receive window. */ #define DELETE_ACK_TIMEOUT_SEC 3600 +/* While waiting for the (potentially slow) receiver-side deletion, send a + * STATUS_KEEPALIVE at most this often so the connection is demonstrably alive + * and neither side's per-message timeout trips. */ +#define DELETE_ACK_KEEPALIVE_SEC 10 static bool send_delete_manifest_early(Client* client, ArrayList* manifest, ArrayList* protected_prefixes, ArrayList* missing_args) { @@ -894,8 +898,21 @@ static bool send_delete_manifest_early(Client* client, ArrayList* manifest, 0) return false; Status ack; - if (!receive_status_timed(client->file_descriptor, &ack, DELETE_ACK_TIMEOUT_SEC)) + /* The wait is long (up to an hour) and runs inline on this thread: a helper + * thread would race the non-thread-safe protocol send path, so keepalives are + * emitted from this wait loop itself. A Ctrl-C/SIGTERM abort flag also ends + * the wait; the caller then best-effort sends STATUS_ABORT. */ + if (!receive_status_keepalive(client->file_descriptor, &ack, DELETE_ACK_TIMEOUT_SEC, + DELETE_ACK_KEEPALIVE_SEC, client_abort_pending)) { + /* A Ctrl-C/SIGTERM abort ends the wait above; tell the receiver before the + caller tears the connection down (best-effort). */ + if (client_abort_pending()) { + log_info_message(LOG_INFO_MISC, + "Abort requested while awaiting delete ack; sending STATUS_ABORT"); + send_status(client->file_descriptor, STATUS_ABORT); + } return false; + } if (ack != STATUS_OK) { log_message(LOG_LEVEL_ERROR, "Server failed to delete files before the transfer"); return false; @@ -1490,6 +1507,19 @@ static int send_chunks_multithreaded(void* pipeline_context) { } while (true) { + /* Graceful abort (Ctrl-C/SIGTERM): tell the receiver to clean up instead of + dying abruptly. Best-effort: a failed send just means the peer is gone. + Only reached while the session is active (config_send already succeeded). */ + if (client_abort_pending()) { + log_info_message(LOG_INFO_MISC, + "Abort requested; sending STATUS_ABORT to server and disconnecting"); + send_status(client->file_descriptor, STATUS_ABORT); + pipeline_cancel(context); + disconnect_transfer_client(client); + mark_sender_done(context); + protocol_session_unbind(); + return thrd_error; + } /* Phase 6: stop-elegantly at the next chunk boundary once the --stop-after / --stop-at deadline has passed. Everything already sent is finalized by the completion tail below; the run still returns success. */ @@ -1624,7 +1654,9 @@ static int scan_directory_multithreaded(void* pipeline_context) { PipelineContextSender* context = (PipelineContextSender*)pipeline_context; protocol_session_bind(&context->allocation_session); PreparedScanner prepared; - if (!prepare_scanner(context->config, 4, &prepared)) { + /* -j/--threads=N sizes the parallel scanner's worker pool; 0 (bare -j) lets + * the scanner apply its built-in default. */ + if (!prepare_scanner(context->config, context->config->scanner_threads, &prepared)) { pipeline_cancel(context); protocol_session_unbind(); return thrd_error; @@ -2036,6 +2068,15 @@ int send_files(Config* config) { only a prefix of the source. */ bool scan_stopped_early = false; while ((current_chunk = directory_scanner_next(scanner)) != NULL) { + /* Graceful abort (Ctrl-C/SIGTERM): notify the receiver and clean up. The + session is active (config_send already succeeded); a send failure here is + fine because the client is exiting anyway. */ + if (client_abort_pending()) { + log_info_message(LOG_INFO_MISC, "Abort requested; sending STATUS_ABORT to server"); + chunk_destroy(current_chunk); + send_status(client->file_descriptor, STATUS_ABORT); + goto send_fail; + } /* Phase 6: stop-elegantly at the next chunk boundary once the deadline has passed. The scanner may also have stopped early itself; either way the completion tail below keeps everything already sent. */ @@ -2099,6 +2140,13 @@ int send_files(Config* config) { goto send_fail; if (directory_scanner_had_io_error(scanner)) had_scan_io = true; + /* An abort that arrived after the last chunk must still stop the completion + tail (manifest/finalize) rather than let it run to success. */ + if (client_abort_pending()) { + log_info_message(LOG_INFO_MISC, "Abort requested; sending STATUS_ABORT to server"); + send_status(client->file_descriptor, STATUS_ABORT); + goto send_fail; + } /* Phase 6: the scanner may have stopped early (returning NULL without a failure) as soon as the deadline passed, so reflect that here too. A deadline that cut the scan short leaves an incomplete keep-set; transmitting @@ -2276,7 +2324,7 @@ int send_files_multithreaded(Config** config_ptr) { fills the protected excluded prefixes. */ PreparedScanner prepared; memset(&prepared, 0, sizeof(prepared)); - bool prepared_ok = prepare_scanner(config, 4, &prepared); + bool prepared_ok = prepare_scanner(config, config->scanner_threads, &prepared); if (prepared_ok && context->excluded_paths) prepared.options.excluded_paths = context->excluded_paths; bool prebuilt = prepared_ok && scan_paths_only(config, &prepared.options, context->manifest, diff --git a/src/client/client_send.h b/src/client/client_send.h index 8c85106..06d6c6d 100644 --- a/src/client/client_send.h +++ b/src/client/client_send.h @@ -4,6 +4,15 @@ #include "chunk.h" #include "config.h" #include "transport_tcp.h" +#include +#include + +/* Set ONLY by the client's SIGINT/SIGTERM handler (async-signal-safe: the + * handler stores 1 and does nothing else). The send loops poll it via + * client_abort_pending() and, when set, best-effort send STATUS_ABORT so the + * receiver can clean up before the client exits. */ +extern volatile sig_atomic_t client_abort_requested; +bool client_abort_pending(void); /* Both sender entry points BORROW `config` for the duration of the call; they * never free it, and the caller retains ownership (freeing it with diff --git a/src/client/scanner.h b/src/client/scanner.h index 3db1ce9..51200e2 100644 --- a/src/client/scanner.h +++ b/src/client/scanner.h @@ -14,6 +14,10 @@ #include #include +/* Upper bound on the configurable parallel scanner worker count (--threads=N): + * keeps one transfer from spawning an unbounded pool on a very large machine. */ +#define MAX_SCANNER_THREADS 256 + typedef struct { bool use_metadata; /* Phase 4 metadata capture: -U/--atimes and -N/--crtimes tell the scanner to diff --git a/src/client/usage.c b/src/client/usage.c index f63a5c3..0045d26 100644 --- a/src/client/usage.c +++ b/src/client/usage.c @@ -2,6 +2,7 @@ #include #include #include +#include "scanner.h" void print_usage(void) { printf("Usage:\n"); @@ -144,7 +145,10 @@ void print_usage(void) { printf(" Delta block size in bytes (default: %d)\n", DELTA_BLOCK_SIZE_DEFAULT); printf(" --delta-max Max file size for delta transfer (default: %llu)\n", DELTA_MAX_FILE_SIZE); - printf(" -j, --threads Enable multithreading\n"); + printf(" -j, --threads[=N] Enable the multithreaded scanner/loader/sender\n"); + printf(" pipeline; N (1-%d) sets the parallel scanner worker\n", + MAX_SCANNER_THREADS); + printf(" count (bare -j/--threads uses the default)\n"); printf(" --chunk-serialization Enable chunk serialization (long form only)\n"); printf(" -s, --secluded-args Protect-args compatibility option (no effect; remote\n"); printf(" SSH argv is already built injection-safe)\n"); @@ -203,6 +207,7 @@ void print_usage(void) { printf(" --save-to-disk Write received files to disk\n"); printf(" --server-host Server IP address (default: 127.0.0.1)\n"); printf(" --server-port Server port (default: 8080)\n"); + printf(" --port Alias for --server-port\n"); printf(" --password-file Authenticate a host::module/path daemon destination.\n"); printf(" The file's first user:password line supplies the\n"); printf(" username and password (only a SHA-256 digest of the\n"); diff --git a/src/shared/config.c b/src/shared/config.c index c0f1412..64adf10 100644 --- a/src/shared/config.c +++ b/src/shared/config.c @@ -23,6 +23,7 @@ static void config_set_defaults(Config* config) { config->receive_root_directory = NULL; config->save_to_disk = false; config->use_multithreading = false; + config->scanner_threads = 0; config->use_chunk_serialization = false; config->use_compression = false; config->use_metadata = false; @@ -79,7 +80,6 @@ static void config_set_defaults(Config* config) { config->stats = false; config->max_depth = 0; config->log_file = NULL; - config->queue_size = 100; config->follow_symlinks = false; config->partial = false; config->copy_links = false; @@ -147,14 +147,11 @@ static void config_set_defaults(Config* config) { config->delete_during = false; config->delete_delay = false; config->address = NULL; - config->bind_address = NULL; config->ipv6 = false; config->ipv4 = false; config->sockopts = NULL; config->sockopt_count = 0; config->daemon = false; - config->daemon_config = NULL; - config->server_mode = false; config->no_motd = false; config->checksum = false; config->checksum_algo = CHECKSUM_ALGO_XXH64; @@ -789,9 +786,7 @@ void config_delete(Config* config) { free(config->partial_dir); free(config->suffix); free(config->address); - free(config->bind_address); free(config->sockopts); - free(config->daemon_config); free(config->compress_choice); free(config->chmod_spec); if (config->skip_compress_suffixes) { diff --git a/src/shared/config.h b/src/shared/config.h index b0ae890..0aba9d6 100644 --- a/src/shared/config.h +++ b/src/shared/config.h @@ -81,6 +81,11 @@ typedef struct Config { char* receive_root_directory; bool save_to_disk; bool use_multithreading; + /* -j/--threads=N: number of parallel scanner worker threads for the -m + * pipeline. 0 (the default, also set by bare -j/--threads) means "use the + * scanner's built-in default" (4). CLIENT-ONLY: it is a local scheduling + * concern and is NEVER serialized into the wire config frame. */ + int scanner_threads; bool use_chunk_serialization; bool use_compression; bool use_sendfile; @@ -170,7 +175,6 @@ typedef struct Config { bool stats; int max_depth; FILE* log_file; - int queue_size; bool follow_symlinks; bool partial; @@ -348,21 +352,17 @@ typedef struct Config { // PR #181: IPv6 and bind address char* address; - char* bind_address; bool ipv6; bool ipv4; /* --sockopts=OPTIONS (Phase 5, Wave B): strict allowlist of TCP/socket * options applied via setsockopt after socket() and before connect()/bind(). * These are LOCAL socket concerns: they never cross the wire config frame. - * .address is the outgoing/source bind address (--address); .bind_address is - * reserved for daemon-side binding and is not wired yet. */ + * .address is the outgoing/source bind address (--address). */ SockOptEntry* sockopts; int sockopt_count; // PR #182: Daemon/server mode bool daemon; - char* daemon_config; - bool server_mode; /* --no-motd (Wave C): CLIENT-ONLY, never crosses the wire. Suppresses * DISPLAY of the daemon's MOTD; the daemon still sends the MOTD frame, so * the client reads and discards it to keep the stream in sync. rsync's diff --git a/src/shared/protocol.c b/src/shared/protocol.c index f3dd81b..0a8dac9 100644 --- a/src/shared/protocol.c +++ b/src/shared/protocol.c @@ -609,6 +609,136 @@ bool protocol_receive_status_timed(ProtocolSession* session, Status* status, int return true; } +/* Read exactly one Status frame within `deadline` (CLOCK_MONOTONIC). Unlike + * protocol_receive_status_keepalive this never emits a keepalive: it is used + * to consume the first byte(s) of an already-signalled frame and to drain the + * peer's outstanding keepalive replies, where injecting a write could split a + * reply across a frame boundary. Returns false on timeout/EOF/error. */ +static bool protocol_read_status_until(ProtocolSession* session, Status* status, + const struct timespec* deadline) { + Status received = STATUS_ERROR; + size_t got = 0; + while (got < sizeof(Status)) { + if (!session->ssl || SSL_pending(session->ssl) == 0) { + int remaining_ms = deadline_remaining_ms(deadline); + if (remaining_ms <= 0) { + log_message(LOG_LEVEL_ERROR, "Receive timeout while reading status"); + return false; + } + struct pollfd pfd = {.fd = session->read_fd, .events = POLLIN}; + int poll_result = poll(&pfd, 1, remaining_ms); + if (poll_result == 0) { + log_message(LOG_LEVEL_ERROR, "Receive timeout while reading status"); + return false; + } + if (poll_result < 0) { + if (errno == EINTR) + continue; + return false; + } + if (pfd.revents & (POLLERR | POLLNVAL)) + return false; + } + ssize_t bytes_received; + if (session->ssl) + bytes_received = SSL_read(session->ssl, (char*)&received + got, sizeof(Status) - got); + else + bytes_received = read(session->read_fd, (char*)&received + got, sizeof(Status) - got); + if (bytes_received <= 0) { + if (session->ssl) { + int ssl_err = SSL_get_error(session->ssl, (int)bytes_received); + if (ssl_err == SSL_ERROR_WANT_READ || ssl_err == SSL_ERROR_WANT_WRITE) + continue; + } + if (bytes_received < 0 && errno == EINTR) + continue; + log_message(LOG_LEVEL_ERROR, "Connection closed while receiving status"); + return false; + } + got += (size_t)bytes_received; + } + *status = received; + return true; +} + +bool protocol_receive_status_keepalive(ProtocolSession* session, Status* status, int timeout_sec, + int keepalive_interval_sec, ProtocolWaitAbort abort_check) { + if (!session || !status) + return false; + if (timeout_sec <= 0) + timeout_sec = RECEIVE_TIMEOUT_SEC; + if (keepalive_interval_sec <= 0) + keepalive_interval_sec = timeout_sec; + + struct timespec deadline; + clock_gettime(CLOCK_MONOTONIC, &deadline); + deadline.tv_sec += timeout_sec; + + unsigned long keepalives_sent = 0; + unsigned long replies_seen = 0; + Status final = STATUS_ERROR; + while (true) { + if (abort_check && abort_check()) + return false; + if (!session->ssl || SSL_pending(session->ssl) == 0) { + int remaining_ms = deadline_remaining_ms(&deadline); + if (remaining_ms <= 0) { + log_message(LOG_LEVEL_ERROR, "Receive timeout after %ds", timeout_sec); + return false; + } + /* Only interleave a keepalive while waiting for the FIRST byte of a + * frame; once part of a frame is buffered a write could race the peer's + * reply into the middle of it. */ + int interval_ms = keepalive_interval_sec * 1000; + int wait_ms = interval_ms < remaining_ms ? interval_ms : remaining_ms; + struct pollfd pfd = {.fd = session->read_fd, .events = POLLIN}; + int poll_result = poll(&pfd, 1, wait_ms); + if (poll_result == 0) { + if (abort_check && abort_check()) + return false; + if (!protocol_send_status(session, STATUS_KEEPALIVE)) + return false; + keepalives_sent++; + continue; + } + if (poll_result < 0) { + if (errno == EINTR) + continue; + return false; + } + if (pfd.revents & (POLLERR | POLLNVAL)) + return false; + } + Status received; + if (!protocol_read_status_until(session, &received, &deadline)) + return false; + if (received == STATUS_KEEPALIVE) { + /* The receiver's answer to one of our keepalives. */ + replies_seen++; + continue; + } + final = received; + break; + } + /* Drain the replies the receiver still owes for keepalives we sent while it + * was busy. It answers them only after the real status, so leaving them + * unread would put stale KEEPALIVE frames ahead of the next exchange and + * desynchronize the protocol. */ + while (replies_seen < keepalives_sent) { + Status drained; + if (!protocol_read_status_until(session, &drained, &deadline)) + return false; + if (drained != STATUS_KEEPALIVE) { + log_message(LOG_LEVEL_ERROR, "Unexpected status while draining keepalive replies"); + return false; + } + replies_seen++; + } + *status = final; + log_debug_message(LOG_DEBUG_PROTO, "Received Status: %s", status_to_string(*status)); + return true; +} + bool send_str(int fd, const char* data) { return protocol_send_str(legacy_session(-1, fd), data); } @@ -648,3 +778,8 @@ bool receive_status(int fd, Status* status) { bool receive_status_timed(int fd, Status* status, int timeout_sec) { return protocol_receive_status_timed(legacy_session(fd, -1), status, timeout_sec); } +bool receive_status_keepalive(int fd, Status* status, int timeout_sec, int keepalive_interval_sec, + ProtocolWaitAbort abort_check) { + return protocol_receive_status_keepalive(legacy_session(fd, -1), status, timeout_sec, + keepalive_interval_sec, abort_check); +} diff --git a/src/shared/protocol.h b/src/shared/protocol.h index b27c2d1..e55a742 100644 --- a/src/shared/protocol.h +++ b/src/shared/protocol.h @@ -198,4 +198,23 @@ bool receive_status(int file_descriptor, Status* status); this so the sender does not abort after the deletion already committed. */ bool receive_status_timed(int file_descriptor, Status* status, int timeout_sec); +/* Callback polled by protocol_receive_status_keepalive once per keepalive + interval. Return true to stop waiting (e.g. a SIGINT/SIGTERM abort flag was + set). Kept as a function pointer so the protocol layer does not depend on + client signal state. */ +typedef bool (*ProtocolWaitAbort)(void); + +/* Like receive_status_timed, but while the peer is silent it emits + STATUS_KEEPALIVE every keepalive_interval_sec (the receiver answers each with + STATUS_KEEPALIVE, which this function consumes and skips) so a long + server-side operation does not look like a dead connection. The total wait + is still bounded by timeout_sec; abort_check (may be NULL) is polled every + interval and, when it returns true, ends the wait immediately with false. + Runs entirely on the calling thread: the protocol send path is NOT safe for + concurrent writers, so this must not be paired with a helper thread. */ +bool receive_status_keepalive(int file_descriptor, Status* status, int timeout_sec, + int keepalive_interval_sec, ProtocolWaitAbort abort_check); +bool protocol_receive_status_keepalive(ProtocolSession* session, Status* status, int timeout_sec, + int keepalive_interval_sec, ProtocolWaitAbort abort_check); + #endif diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c index e5375b9..b2a7021 100644 --- a/tests/test_client_cli.c +++ b/tests/test_client_cli.c @@ -1,5 +1,6 @@ #include "test_client_cli.h" #include "checksum.h" +#include "client_send.h" #include "client_validation.h" #include "chmod.h" #include "config.h" @@ -579,6 +580,80 @@ static void test_parse_args_invalid_server_port() { config_delete(cfg); } +/* --port is a documented rsync-style alias for --server-port; both the + * two-argument and the inline "=" spellings must work. */ +static void test_parse_args_port_alias() { + Config* cfg = config_create(); + char* argv_space[] = {"fastsync", "--port", "9000", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv_space, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->server_port, 9000); + config_delete(cfg); + + cfg = config_create(); + char* argv_inline[] = {"fastsync", "--port=9001", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_inline, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->server_port, 9001); + config_delete(cfg); + + cfg = config_create(); + char* argv_long[] = {"fastsync", "--server-port=9002", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_long, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->server_port, 9002); + config_delete(cfg); +} + +/* --threads=N sizes the pipeline scanner; bare -j/--threads keeps the default + * (scanner_threads == 0), and invalid values are rejected. */ +static void test_parse_args_threads() { + Config* cfg = config_create(); + char* argv_eq[] = {"fastsync", "--threads=8", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_eq, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_multithreading); + EXPECT_EQ_INT(cfg->scanner_threads, 8); + config_delete(cfg); + + cfg = config_create(); + char* argv_short[] = {"fastsync", "-j", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_short, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_multithreading); + EXPECT_EQ_INT(cfg->scanner_threads, 0); + config_delete(cfg); + + cfg = config_create(); + char* argv_long[] = {"fastsync", "--threads", "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_long, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->use_multithreading); + EXPECT_EQ_INT(cfg->scanner_threads, 0); + config_delete(cfg); + + const char* bad[] = {"--threads=0", "--threads=-3", "--threads=abc", "--threads=257"}; + for (size_t i = 0; i < sizeof(bad) / sizeof(bad[0]); i++) { + cfg = config_create(); + char* argv_bad[] = {"fastsync", (char*)bad[i], "/src", "/dst"}; + positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_bad, positional_args, &positional_count), -1); + config_delete(cfg); + } +} + +/* The graceful-abort flag is a plain sig_atomic_t toggled by the handler. */ +static void test_client_abort_flag() { + client_abort_requested = 0; + EXPECT_FALSE(client_abort_pending()); + client_abort_requested = 1; + EXPECT_TRUE(client_abort_pending()); + client_abort_requested = 0; + EXPECT_FALSE(client_abort_pending()); +} + /* Test parse_args rejects invalid compression level (-z/--compress) */ static void test_parse_args_invalid_compression_level() { Config* cfg = config_create(); @@ -3224,6 +3299,9 @@ void test_client_cli() { test_parse_args_invalid_port(); test_parse_args_non_numeric_port(); test_parse_args_invalid_server_port(); + test_parse_args_port_alias(); + test_parse_args_threads(); + test_client_abort_flag(); test_parse_args_invalid_compression_level(); test_parse_args_valid_compression_level(); test_parse_args_debug_flags(); diff --git a/tests/test_protocol.c b/tests/test_protocol.c index 273861e..3a63125 100644 --- a/tests/test_protocol.c +++ b/tests/test_protocol.c @@ -460,6 +460,96 @@ static void test_send_receive_status_timed() { close(p[0]); } +static bool keepalive_always_abort(void) { + return true; +} + +/* A pre-buffered KEEPALIVE reply from the peer must be consumed transparently, + leaving the first real status visible to the caller. */ +static void test_receive_status_keepalive_skips_reply() { + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + ProtocolSession session; + protocol_session_init(&session, p[0], p[1]); + + EXPECT_TRUE(protocol_send_status(&session, STATUS_KEEPALIVE)); + EXPECT_TRUE(protocol_send_status(&session, STATUS_OK)); + + Status received = STATUS_ERROR; + EXPECT_TRUE(protocol_receive_status_keepalive(&session, &received, 5, 1, NULL)); + EXPECT_EQ_INT((int)received, (int)STATUS_OK); + + close(p[0]); + close(p[1]); +} + +/* The abort callback ends the wait immediately, before any keepalive traffic. */ +static void test_receive_status_keepalive_aborts() { + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + ProtocolSession session; + protocol_session_init(&session, p[0], p[1]); + + Status received = STATUS_ERROR; + EXPECT_FALSE( + protocol_receive_status_keepalive(&session, &received, 5, 1, keepalive_always_abort)); + + close(p[0]); + close(p[1]); +} + +typedef struct { + int peer_read_fd; + int peer_write_fd; + bool replied; +} KeepalivePeerArg; + +static int keepalive_peer(void* arg) { + KeepalivePeerArg* peer = arg; + ProtocolSession session; + protocol_session_init(&session, peer->peer_read_fd, peer->peer_write_fd); + Status status = STATUS_ERROR; + if (protocol_receive_status(&session, &status) && status == STATUS_KEEPALIVE) { + /* Model the busy receiver: it sends the real ack first, then the keepalive + reply it owes for the queued keepalive (which the client must drain so it + does not desynchronize the stream). */ + peer->replied = protocol_send_status(&session, STATUS_OK) && + protocol_send_status(&session, STATUS_KEEPALIVE); + } + return thrd_success; +} + +/* While the peer is silent the helper must emit STATUS_KEEPALIVE, then consume + the peer's ack and drain the keepalive reply that follows it -- proving the + inline keepalive loop works without a second writer racing the send path. */ +static void test_receive_status_keepalive_emits() { + int to_client[2]; + int to_peer[2]; + EXPECT_EQ_INT(pipe(to_client), 0); + EXPECT_EQ_INT(pipe(to_peer), 0); + + ProtocolSession session; + protocol_session_init(&session, to_client[0], to_peer[1]); + + KeepalivePeerArg peer = {.peer_read_fd = to_peer[0], .peer_write_fd = to_client[1]}; + thrd_t thread; + EXPECT_EQ_INT(thrd_create(&thread, keepalive_peer, &peer), thrd_success); + + Status received = STATUS_ERROR; + EXPECT_TRUE(protocol_receive_status_keepalive(&session, &received, 10, 1, NULL)); + EXPECT_EQ_INT((int)received, (int)STATUS_OK); + + int result = 0; + EXPECT_EQ_INT(thrd_join(thread, &result), thrd_success); + EXPECT_EQ_INT(result, thrd_success); + EXPECT_TRUE(peer.replied); + + close(to_client[0]); + close(to_client[1]); + close(to_peer[0]); + close(to_peer[1]); +} + void test_protocol() { test_send_receive_n_data(); test_send_receive_n_data_zero(); @@ -471,6 +561,9 @@ void test_protocol() { test_send_receive_status(); test_protocol_session_io_timeout(); test_send_receive_status_timed(); + test_receive_status_keepalive_skips_reply(); + test_receive_status_keepalive_aborts(); + test_receive_status_keepalive_emits(); test_receive_n_data_truncated(); test_receive_str_truncated(); test_max_alloc_rejects_single_buffer(); From 2854a9d149be8269e1f76dfb25f7a47c20387a04 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 06:24:46 +0200 Subject: [PATCH 031/155] test: fuzz manifest/protocol/xattr, hardlink unit, fault injection --- tests/fuzz/fuzz_manifest.c | 126 ++++++++ tests/fuzz/fuzz_protocol_framing.c | 127 ++++++++ tests/fuzz/fuzz_xattr_block.c | 115 +++++++ tests/integration/test_fault_injection.py | 367 ++++++++++++++++++++++ tests/runner.c | 2 + tests/test_hardlink.c | 203 ++++++++++++ tests/test_hardlink.h | 6 + 7 files changed, 946 insertions(+) create mode 100644 tests/fuzz/fuzz_manifest.c create mode 100644 tests/fuzz/fuzz_protocol_framing.c create mode 100644 tests/fuzz/fuzz_xattr_block.c create mode 100644 tests/integration/test_fault_injection.py create mode 100644 tests/test_hardlink.c create mode 100644 tests/test_hardlink.h diff --git a/tests/fuzz/fuzz_manifest.c b/tests/fuzz/fuzz_manifest.c new file mode 100644 index 0000000..57a9f3e --- /dev/null +++ b/tests/fuzz/fuzz_manifest.c @@ -0,0 +1,126 @@ +/* + * Fuzz the delete-manifest parser: receive_manifest_entries(int fd). + * + * The parser reads three length-delimited sections (keeps, protected prefixes, + * missing-args paths) from the connection. Feeding raw bytes alone exercises + * the "reject the first malformed count/string" fast paths, but because each + * section is self-delimiting a single bad value hides every later section. + * + * To reach the protected-prefix and missing-args parsers (the paths that drive + * actual destination deletion) we build one canonical, fully-valid manifest + * with hand-written wire framing and then feed the receiver several shapes: + * + * 1. raw : the raw fuzz bytes as the whole manifest. + * 2. keeps : the valid keep count only + the fuzz bytes, so the fuzzer + * drives the keep count and entries directly. + * 3. prot : the valid keeps section + the fuzz bytes, so the fuzzer drives + * the protected count and prefixes. + * 4. missing: the valid keeps+protected sections + the fuzz bytes, so the + * fuzzer drives the trailing missing-args section, including the + * aggregate MAX_MANIFEST_BYTES budget. + * + * The wire encoding matches receive_int (native int) and receive_wire_str + * (native size_t length prefix + body); no charset conversion is configured in + * the fuzz process, so receive_wire_str is receive_str. + */ +#include "file_receive.h" +#include "protocol.h" +#include +#include +#include +#include +#include +#include +#include + +static unsigned char g_manifest[512]; +static size_t g_len_after_count; /* offset of the first keep entry */ +static size_t g_len_after_keeps; /* offset of the protected count */ +static size_t g_len_after_protected; /* offset of the missing count */ +static int g_manifest_ready; + +static void append_int32(unsigned char* buf, size_t* off, int32_t value) { + memcpy(buf + *off, &value, sizeof(value)); + *off += sizeof(value); +} + +static void append_wire_str(unsigned char* buf, size_t* off, const char* s) { + size_t n = strlen(s); + memcpy(buf + *off, &n, sizeof(n)); + *off += sizeof(n); + memcpy(buf + *off, s, n); + *off += n; +} + +static void build_canonical_manifest(void) { + g_manifest_ready = 1; + size_t off = 0; + append_int32(g_manifest, &off, 2); + g_len_after_count = off; + append_wire_str(g_manifest, &off, "keep/a"); + append_wire_str(g_manifest, &off, "keep/b"); + g_len_after_keeps = off; + append_int32(g_manifest, &off, 1); + append_wire_str(g_manifest, &off, "excluded/prefix"); + g_len_after_protected = off; + append_int32(g_manifest, &off, 1); + append_wire_str(g_manifest, &off, "missing/path"); +} + +/* Best-effort non-blocking write: an oversized fuzz input is truncated rather + * than stalling the harness. */ +static void write_best_effort(int fd, const void* data, size_t size) { + const unsigned char* p = data; + size_t off = 0; + while (off < size) { + ssize_t n = write(fd, p + off, size - off); + if (n > 0) { + off += (size_t)n; + continue; + } + if (n < 0 && errno == EINTR) + continue; + break; + } +} + +/* Build prefix ++ data as a stream and drive receive_manifest_entries over it. + * The write half is shut down first so the parser always sees EOF instead of + * blocking on a missing frame tail. */ +static void receive_stream(const unsigned char* prefix, size_t prefix_len, const uint8_t* data, + size_t size) { + int sv[2]; + if (socketpair(AF_UNIX, SOCK_STREAM, 0, sv) != 0) + return; + + int flags = fcntl(sv[0], F_GETFL, 0); + if (flags != -1) + (void)fcntl(sv[0], F_SETFL, flags | O_NONBLOCK); + + if (prefix_len > 0) + write_best_effort(sv[0], prefix, prefix_len); + if (size > 0) + write_best_effort(sv[0], data, size); + shutdown(sv[0], SHUT_WR); + + DeleteManifest* manifest = receive_manifest_entries(sv[1]); + delete_manifest_free(manifest); + + close(sv[0]); + close(sv[1]); +} + +int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { + if (!g_manifest_ready) + build_canonical_manifest(); + + /* Raw bytes as the whole manifest. */ + receive_stream(NULL, 0, data, size); + + /* Keep the valid framing so the fuzzer reaches each later section. */ + receive_stream(g_manifest, g_len_after_protected, data, size); + receive_stream(g_manifest, g_len_after_keeps, data, size); + receive_stream(g_manifest, g_len_after_count, data, size); + + return 0; +} diff --git a/tests/fuzz/fuzz_protocol_framing.c b/tests/fuzz/fuzz_protocol_framing.c new file mode 100644 index 0000000..6bb0359 --- /dev/null +++ b/tests/fuzz/fuzz_protocol_framing.c @@ -0,0 +1,127 @@ +/* + * Fuzz the base protocol framing: receive_str / receive_data / receive_status + * (plus the redacted string, the size-limited data and the timed-status + * variants) fed arbitrary bytes over an in-memory socketpair. + * + * Every receive primitive reads a fixed-width header (a size_t string length, + * an unsigned long long data length, an int status/int value) and then a body. + * The fuzzer attacks: + * - oversized length headers (the MAX_STRING_SIZE / MAX_DATA_PAYLOAD_SIZE + * gates must reject before allocating), + * - truncated bodies (a declared body larger than the stream must fail + * cleanly at EOF, never read uninitialised memory or leak), + * - embedded NUL bytes in strings (must be refused), + * - out-of-range status enum values (status_to_string must stay in bounds). + * + * Each entry point gets its own socketpair because a single receive consumes a + * variable number of bytes from the stream; reusing one would make the later + * calls meaningless. The write half is shut down first so a truncated frame + * always terminates at EOF instead of blocking. + */ +#include "data.h" +#include "protocol.h" +#include +#include +#include +#include +#include +#include +#include + +static void write_best_effort(int fd, const void* data, size_t size) { + const unsigned char* p = data; + size_t off = 0; + while (off < size) { + ssize_t n = write(fd, p + off, size - off); + if (n > 0) { + off += (size_t)n; + continue; + } + if (n < 0 && errno == EINTR) + continue; + break; + } +} + +/* Create a socketpair pre-loaded with `data`, shut down the write half and + * return the read end (which the receiver reads from). `*write_end` is also + * returned so the caller can close it. */ +static int make_stream(const uint8_t* data, size_t size, int* write_end) { + int sv[2]; + if (socketpair(AF_UNIX, SOCK_STREAM, 0, sv) != 0) { + *write_end = -1; + return -1; + } + int flags = fcntl(sv[0], F_GETFL, 0); + if (flags != -1) + (void)fcntl(sv[0], F_SETFL, flags | O_NONBLOCK); + if (size > 0) + write_best_effort(sv[0], data, size); + shutdown(sv[0], SHUT_WR); + *write_end = sv[0]; + return sv[1]; +} + +int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { + int w; + + int rd = make_stream(data, size, &w); + if (rd >= 0) { + char* s = receive_str(rd); + free(s); + close(rd); + close(w); + } + + rd = make_stream(data, size, &w); + if (rd >= 0) { + char* s = receive_str_redacted(rd); + free(s); + close(rd); + close(w); + } + + rd = make_stream(data, size, &w); + if (rd >= 0) { + Data* d = receive_data(rd); + data_destroy(d); + close(rd); + close(w); + } + + /* The size-limited variant must reject anything beyond its explicit bound + * before allocating the body buffer. */ + rd = make_stream(data, size, &w); + if (rd >= 0) { + Data* d = receive_data_limited(rd, 256); + data_destroy(d); + close(rd); + close(w); + } + + rd = make_stream(data, size, &w); + if (rd >= 0) { + Status status = STATUS_OK; + (void)receive_status(rd, &status); + close(rd); + close(w); + } + + rd = make_stream(data, size, &w); + if (rd >= 0) { + Status status = STATUS_OK; + (void)receive_status_timed(rd, &status, 1); + close(rd); + close(w); + } + + rd = make_stream(data, size, &w); + if (rd >= 0) { + int value = 0; + (void)receive_int(rd, &value); + close(rd); + close(w); + } + + return 0; +} diff --git a/tests/fuzz/fuzz_xattr_block.c b/tests/fuzz/fuzz_xattr_block.c new file mode 100644 index 0000000..26e3775 --- /dev/null +++ b/tests/fuzz/fuzz_xattr_block.c @@ -0,0 +1,115 @@ +/* + * Fuzz the xattr wire block parser: xattr_receive(int fd, int* ok). + * + * The block is a count followed by that many (name_len, name, value_len, value) + * records. The receiver must reject an invalid count, an out-of-range or + * negative name/value length, an embedded NUL or non-whitelisted namespace in + * the name, an oversized value, and an aggregate payload beyond + * XATTR_TOTAL_MAX -- all without over-allocating or leaking. + * + * Raw bytes mostly stop at the first invalid count/length, so we also build a + * canonical, fully-valid two-entry block by hand and feed the receiver valid + * prefixes of it followed by the fuzz bytes. That drives the deep value- + * parsing and per-entry namespace/budget checks with attacker-controlled input. + */ +#include "protocol.h" +#include "xattr.h" +#include +#include +#include +#include +#include +#include +#include + +static unsigned char g_block[512]; +static size_t g_off_after_count; /* start of entry 0 */ +static size_t g_off_after_entry0; /* start of entry 1 */ +static size_t g_off_value0; /* start of the first value length */ +static int g_block_ready; + +static void append_int32(unsigned char* buf, size_t* off, int32_t value) { + memcpy(buf + *off, &value, sizeof(value)); + *off += sizeof(value); +} + +static void append_bytes(unsigned char* buf, size_t* off, const void* p, size_t n) { + if (n > 0) + memcpy(buf + *off, p, n); + *off += n; +} + +static void build_canonical_block(void) { + g_block_ready = 1; + size_t off = 0; + append_int32(g_block, &off, 2); + g_off_after_count = off; + + int32_t name0_len = (int32_t)strlen("user.foo"); + append_int32(g_block, &off, name0_len); + append_bytes(g_block, &off, "user.foo", (size_t)name0_len); + g_off_value0 = off; + append_int32(g_block, &off, 3); + append_bytes(g_block, &off, "bar", 3); + + g_off_after_entry0 = off; + int32_t name1_len = (int32_t)strlen("user.empty"); + append_int32(g_block, &off, name1_len); + append_bytes(g_block, &off, "user.empty", (size_t)name1_len); + append_int32(g_block, &off, 0); +} + +static void write_best_effort(int fd, const void* data, size_t size) { + const unsigned char* p = data; + size_t off = 0; + while (off < size) { + ssize_t n = write(fd, p + off, size - off); + if (n > 0) { + off += (size_t)n; + continue; + } + if (n < 0 && errno == EINTR) + continue; + break; + } +} + +static void receive_stream(const unsigned char* prefix, size_t prefix_len, const uint8_t* data, + size_t size) { + int sv[2]; + if (socketpair(AF_UNIX, SOCK_STREAM, 0, sv) != 0) + return; + + int flags = fcntl(sv[0], F_GETFL, 0); + if (flags != -1) + (void)fcntl(sv[0], F_SETFL, flags | O_NONBLOCK); + + if (prefix_len > 0) + write_best_effort(sv[0], prefix, prefix_len); + if (size > 0) + write_best_effort(sv[0], data, size); + shutdown(sv[0], SHUT_WR); + + int ok = 0; + FileXattrList* list = xattr_receive(sv[1], &ok); + xattr_list_free(list); + + close(sv[0]); + close(sv[1]); +} + +int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { + if (!g_block_ready) + build_canonical_block(); + + /* Raw bytes as the whole block. */ + receive_stream(NULL, 0, data, size); + + /* Valid framing so the fuzzer mutates the entry list, the first value and + * the second entry respectively instead of stopping at the count. */ + receive_stream(g_block, g_off_after_entry0, data, size); + receive_stream(g_block, g_off_value0, data, size); + receive_stream(g_block, g_off_after_count, data, size); + + return 0; +} diff --git a/tests/integration/test_fault_injection.py b/tests/integration/test_fault_injection.py new file mode 100644 index 0000000..8978daf --- /dev/null +++ b/tests/integration/test_fault_injection.py @@ -0,0 +1,367 @@ +"""Fault injection: the server must survive truncated / corrupted protocol +frames and abrupt mid-frame disconnects, and keep serving later connections. + +These tests deliberately speak raw bytes to a real server process: + + * malformed frames before/inside the config handshake (oversized length + headers, truncated string bodies, outright garbage), + * a captured *valid* config frame replayed so the connection reaches the + operation loop, followed by a partial ``STATUS_MANIFEST`` frame that is cut + mid-body and dropped, and + * a real client run relayed through a proxy that truncates the stream at a + range of byte offsets and resets both ends. + +After every fault the server process is asserted alive and a subsequent +ordinary transfer must complete and verify, proving the accept loop and +per-connection children recovered cleanly. All interactions are bounded by +short socket timeouts (no sleeps). +""" +import os +import select +import shutil +import socket +import struct +import subprocess +import sys +import threading + +import pytest + +sys.path.insert(0, os.path.dirname(__file__)) +from common import ( # noqa: E402 + ServerManager, + TEST_DATA_DIR, + get_dest_received_dir, + run_client, + verify_transfer, +) + +PROTOCOL_VERSION = b"2.20.0" +STATUS_MANIFEST = 5 +STATUS_OK = 0 + +SOURCE_DIR = os.path.join(TEST_DATA_DIR, "fault_src") +DEST_DIR = os.path.join(TEST_DATA_DIR, "fault_dst") + + +@pytest.fixture(scope="module") +def fault_server(): + """A dedicated server so the aliveness assertions observe exactly the + process these faults were sent to.""" + server = ServerManager() + server.start() + yield server + server.stop() + + +@pytest.fixture(scope="module", autouse=True) +def _seed_source(): + if os.path.exists(SOURCE_DIR): + shutil.rmtree(SOURCE_DIR) + os.makedirs(os.path.join(SOURCE_DIR, "nested")) + with open(os.path.join(SOURCE_DIR, "hello.txt"), "wb") as fh: + fh.write(b"fault injection payload\n" * 64) + with open(os.path.join(SOURCE_DIR, "nested", "deep.bin"), "wb") as fh: + fh.write(bytes(range(256)) * 16) + yield + shutil.rmtree(SOURCE_DIR, ignore_errors=True) + shutil.rmtree(DEST_DIR, ignore_errors=True) + + +def _assert_alive(server): + assert server._proc is not None, "server process missing" + assert server._proc.poll() is None, ( + f"server exited with {server._proc.returncode} after fault injection" + ) + + +def _recover(server, label): + """Run one ordinary transfer and verify it end-to-end.""" + shutil.rmtree(DEST_DIR, ignore_errors=True) + os.makedirs(DEST_DIR) + result, _ = run_client(SOURCE_DIR, DEST_DIR, flags=["--preserve"], port=server.port) + assert result.returncode == 0, ( + f"{label}: recovery transfer failed rc={result.returncode}: " + f"{(result.stderr or result.stdout)[:200]}" + ) + received = get_dest_received_dir(DEST_DIR, SOURCE_DIR) + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"{label}: recovery missing {missing}" + assert not mismatches, f"{label}: recovery mismatch {mismatches}" + + +def _abrupt_close(sock): + """Force an RST instead of a graceful FIN, the nastier mid-frame drop.""" + try: + sock.setsockopt(socket.SOL_SOCKET, socket.SO_LINGER, struct.pack("ii", 1, 0)) + except OSError: + pass + try: + sock.close() + except OSError: + pass + + +def _raw_connect(server): + sock = socket.create_connection(("127.0.0.1", server.port), timeout=5) + sock.settimeout(5) + return sock + + +def _recv_exact(sock, n): + buf = b"" + while len(buf) < n: + chunk = sock.recv(n - len(buf)) + if not chunk: + return None + buf += chunk + return buf + + +# --- faults before/inside the config handshake ----------------------------- + +CONFIG_HANDSHAKE_FAULTS = { + "empty": b"", + # Length header claims a 1 EiB string body that never arrives. + "oversized_length": struct.pack("server connection and record the client's config + frame (all client bytes forwarded before the server's first reply).""" + + def __init__(self, target_port): + self.target = ("127.0.0.1", target_port) + self.listener = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + self.listener.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + self.listener.bind(("127.0.0.1", 0)) + self.listener.listen(1) + self.listener.settimeout(20) + self.port = self.listener.getsockname()[1] + self.config_frame = None + + def run(self, cmd): + def serve(): + try: + client, _ = self.listener.accept() + except OSError: + return + try: + backend = socket.create_connection(self.target, timeout=10) + except OSError: + client.close() + return + client.settimeout(20) + backend.settimeout(20) + buf_c = bytearray() + seen_server = False + try: + while True: + ready, _, _ = select.select([client, backend], [], [], 20) + if not ready: + break + done = False + for sock in ready: + data = sock.recv(65536) + if not data: + done = True + continue + if sock is client: + buf_c += data + backend.sendall(data) + else: + if not seen_server: + seen_server = True + self.config_frame = bytes(buf_c) + client.sendall(data) + if done: + break + except OSError: + pass + finally: + client.close() + backend.close() + + thread = threading.Thread(target=serve) + thread.start() + result = subprocess.run(cmd, capture_output=True, text=True, timeout=60) + thread.join(20) + return result + + def close(self): + try: + self.listener.close() + except OSError: + pass + + +@pytest.fixture(scope="module") +def captured_config(fault_server): + """Capture the config frame of one real client run through a relay.""" + proxy = _CaptureProxy(fault_server.port) + cmd = [ + os.path.join(os.path.dirname(__file__), "..", "..", "build", "client"), + "--source-dir", + SOURCE_DIR, + "--dest-dir", + DEST_DIR, + "--save-to-disk", + "--server-port", + str(proxy.port), + ] + try: + result = proxy.run(cmd) + assert result.returncode == 0, ( + f"capture run failed rc={result.returncode}: " + f"{(result.stderr or result.stdout)[:200]}" + ) + assert proxy.config_frame, "failed to capture the client config frame" + yield proxy.config_frame + finally: + proxy.close() + + +class TestTruncatedStatusFrame: + def test_partial_manifest_frame_then_drop(self, fault_server, captured_config): + sock = _raw_connect(fault_server) + sock.sendall(captured_config) + ack = _recv_exact(sock, 4) + assert ack is not None, "server closed before the config ack" + (status,) = struct.unpack("= self.max_client_bytes: + break + except (OSError, socket.timeout): + pass + for sock in (client, backend): + try: + sock.setsockopt(socket.SOL_SOCKET, socket.SO_LINGER, struct.pack("ii", 1, 0)) + except OSError: + pass + try: + sock.close() + except OSError: + pass + + thread = threading.Thread(target=serve) + thread.start() + try: + subprocess.run(cmd, capture_output=True, text=True, timeout=30) + finally: + thread.join(20) + self.listener.close() + + +class TestAbruptMidTransferDisconnect: + def test_client_stream_cut_at_offsets(self, fault_server, captured_config): + """Cut the real client stream at offsets anchored to the config frame's + actual size: mid-config, right after the config, and into the operation + stream -- each followed by an RST of both ends.""" + config_len = len(captured_config) + cuts = sorted({max(1, config_len // 2), max(1, config_len - 1), config_len + 8, + config_len + 256}) + for cut in cuts: + proxy = _TruncatingProxy(fault_server.port, cut) + cmd = [ + os.path.join(os.path.dirname(__file__), "..", "..", "build", "client"), + "--source-dir", + SOURCE_DIR, + "--dest-dir", + DEST_DIR, + "--save-to-disk", + "--server-port", + str(proxy.port), + ] + # The client is expected to fail; what matters is the server survives. + proxy.run(cmd) + _assert_alive(fault_server) + _recover(fault_server, "abrupt mid-transfer disconnects") diff --git a/tests/runner.c b/tests/runner.c index 51938e1..9e12323 100644 --- a/tests/runner.c +++ b/tests/runner.c @@ -16,6 +16,7 @@ #include "test_file_sendfile.h" #include "test_fuzz_smoke.h" #include "test_glob.h" +#include "test_hardlink.h" #include "test_iconv.h" #include "test_log.h" #include "test_metadata.h" @@ -88,6 +89,7 @@ int main() { RUN_TEST(test_server_cli); RUN_TEST(test_fuzz_smoke); RUN_TEST(test_xattr); + RUN_TEST(test_hardlink); printf("\n\033[1;36m=== TEST SUMMARY ===\033[0m\n"); printf("Total Tests Run: %d\n", tests_run); diff --git a/tests/test_hardlink.c b/tests/test_hardlink.c new file mode 100644 index 0000000..054e285 --- /dev/null +++ b/tests/test_hardlink.c @@ -0,0 +1,203 @@ +#include "test_hardlink.h" +#include "hardlink.h" +#include "test_utils.h" +#include +#include +#include +#include + +/* A fresh table starts empty and destroy accepts NULL / a fresh table. */ +static void test_hardlink_create_destroy() { + hardlink_table_destroy(NULL); + + HardLinkTable* table = hardlink_table_create(); + EXPECT_NOT_NULL(table); + EXPECT_EQ_INT((int)table->count, 0); + EXPECT_EQ_INT((int)table->capacity, 0); + EXPECT_EQ_INT(table->next_gid, 1); + hardlink_table_destroy(table); +} + +/* The first member of an (dev, ino) group is data-carrying and owns the group; + * every later member gets the SAME gid, is not first, and points back at the + * first member's wire path. */ +static void test_hardlink_grouping() { + HardLinkTable* table = hardlink_table_create(); + EXPECT_NOT_NULL(table); + + int gid_a = -1, gid_b = -1; + bool first_a = false, first_b = false; + char* first_path_a = NULL; + char* first_path_b = NULL; + + EXPECT_TRUE( + hardlink_table_assign(table, "dir/first.txt", 7, 42, &gid_a, &first_a, &first_path_a)); + EXPECT_TRUE(first_a); + EXPECT_EQ_INT(gid_a, 1); + EXPECT_NOT_NULL(first_path_a); + EXPECT_EQ_STR(first_path_a, "dir/first.txt"); + + EXPECT_TRUE( + hardlink_table_assign(table, "dir/second.txt", 7, 42, &gid_b, &first_b, &first_path_b)); + EXPECT_FALSE(first_b); + EXPECT_EQ_INT(gid_b, gid_a); + EXPECT_NOT_NULL(first_path_b); + EXPECT_EQ_STR(first_path_b, "dir/first.txt"); + + /* Two members map onto a single stored group. */ + EXPECT_EQ_INT((int)table->count, 1); + EXPECT_EQ_INT(table->next_gid, 2); + + free(first_path_a); + free(first_path_b); + hardlink_table_destroy(table); +} + +/* A different inode on the same device is a distinct group with a fresh gid. */ +static void test_hardlink_distinct_inode() { + HardLinkTable* table = hardlink_table_create(); + EXPECT_NOT_NULL(table); + + int gid1 = -1, gid2 = -1; + bool first1 = false, first2 = false; + char* path1 = NULL; + char* path2 = NULL; + + EXPECT_TRUE(hardlink_table_assign(table, "a", 7, 100, &gid1, &first1, &path1)); + EXPECT_TRUE(first1); + EXPECT_TRUE(hardlink_table_assign(table, "b", 7, 101, &gid2, &first2, &path2)); + EXPECT_TRUE(first2); + EXPECT_TRUE(gid1 != gid2); + EXPECT_EQ_INT(gid1, 1); + EXPECT_EQ_INT(gid2, 2); + EXPECT_EQ_STR(path1, "a"); + EXPECT_EQ_STR(path2, "b"); + + free(path1); + free(path2); + hardlink_table_destroy(table); +} + +/* Identical (dev, ino) on a DIFFERENT device must never be conflated: inode + * numbers are only unique per filesystem, so grouping is scoped by st_dev. */ +static void test_hardlink_distinct_device() { + HardLinkTable* table = hardlink_table_create(); + EXPECT_NOT_NULL(table); + + int gid1 = -1, gid2 = -1; + bool first1 = false, first2 = false; + char* path1 = NULL; + char* path2 = NULL; + + EXPECT_TRUE(hardlink_table_assign(table, "dev_a/one", 1, 55, &gid1, &first1, &path1)); + EXPECT_TRUE(hardlink_table_assign(table, "dev_b/one", 2, 55, &gid2, &first2, &path2)); + EXPECT_TRUE(first1); + EXPECT_TRUE(first2); + EXPECT_TRUE(gid1 != gid2); + EXPECT_EQ_INT((int)table->count, 2); + + free(path1); + free(path2); + hardlink_table_destroy(table); +} + +/* The table owns deep copies of every path: mutating (or freeing) the caller's + * buffer after assign must not affect the stored / returned paths. */ +static void test_hardlink_path_ownership() { + HardLinkTable* table = hardlink_table_create(); + EXPECT_NOT_NULL(table); + + char caller[] = "owned/path"; + int gid = -1; + bool is_first = false; + char* out = NULL; + + EXPECT_TRUE(hardlink_table_assign(table, caller, 3, 9, &gid, &is_first, &out)); + EXPECT_TRUE(is_first); + /* The returned pointer is a distinct allocation, not the caller's buffer. */ + EXPECT_TRUE(out != caller); + EXPECT_TRUE(table->items[0].first_path != caller); + + memset(caller, 'X', sizeof(caller) - 1); + EXPECT_EQ_STR(out, "owned/path"); + EXPECT_EQ_STR(table->items[0].first_path, "owned/path"); + + /* Later members get their own independent copy of the first path. */ + char second_caller[] = "owned/second"; + int gid2 = -1; + bool first2 = true; + char* out2 = NULL; + EXPECT_TRUE(hardlink_table_assign(table, second_caller, 3, 9, &gid2, &first2, &out2)); + EXPECT_FALSE(first2); + EXPECT_EQ_STR(out2, "owned/path"); + EXPECT_TRUE(out2 != table->items[0].first_path); + EXPECT_TRUE(out2 != out); + + free(out); + free(out2); + hardlink_table_destroy(table); +} + +/* Bad arguments must be rejected without touching the table. */ +static void test_hardlink_reject_bad_args() { + HardLinkTable* table = hardlink_table_create(); + EXPECT_NOT_NULL(table); + + int gid = 0; + bool is_first = false; + char* out = NULL; + + EXPECT_FALSE(hardlink_table_assign(NULL, "x", 1, 1, &gid, &is_first, &out)); + EXPECT_FALSE(hardlink_table_assign(table, NULL, 1, 1, &gid, &is_first, &out)); + EXPECT_FALSE(hardlink_table_assign(table, "x", 1, 1, NULL, &is_first, &out)); + EXPECT_FALSE(hardlink_table_assign(table, "x", 1, 1, &gid, NULL, &out)); + EXPECT_FALSE(hardlink_table_assign(table, "x", 1, 1, &gid, &is_first, NULL)); + EXPECT_EQ_INT((int)table->count, 0); + + hardlink_table_destroy(table); +} + +/* Many distinct groups grow the item array through its realloc path and keep + * gid assignment stable and monotonic. */ +static void test_hardlink_many_groups() { + HardLinkTable* table = hardlink_table_create(); + EXPECT_NOT_NULL(table); + + const int n = 200; + for (int i = 0; i < n; i++) { + int gid = -1; + bool is_first = false; + char* out = NULL; + char path[32]; + snprintf(path, sizeof(path), "file_%d", i); + EXPECT_TRUE(hardlink_table_assign(table, path, 1, (ino_t)(1000 + i), &gid, &is_first, &out)); + EXPECT_TRUE(is_first); + EXPECT_EQ_INT(gid, i + 1); + EXPECT_EQ_STR(out, path); + free(out); + } + EXPECT_EQ_INT((int)table->count, n); + EXPECT_EQ_INT(table->next_gid, n + 1); + + /* Re-querying an existing inode still reports the original gid. */ + int gid = -1; + bool is_first = true; + char* out = NULL; + EXPECT_TRUE(hardlink_table_assign(table, "file_7_again", 1, 1007, &gid, &is_first, &out)); + EXPECT_FALSE(is_first); + EXPECT_EQ_INT(gid, 8); + EXPECT_EQ_STR(out, "file_7"); + free(out); + + hardlink_table_destroy(table); +} + +void test_hardlink() { + test_hardlink_create_destroy(); + test_hardlink_grouping(); + test_hardlink_distinct_inode(); + test_hardlink_distinct_device(); + test_hardlink_path_ownership(); + test_hardlink_reject_bad_args(); + test_hardlink_many_groups(); +} diff --git a/tests/test_hardlink.h b/tests/test_hardlink.h new file mode 100644 index 0000000..c271154 --- /dev/null +++ b/tests/test_hardlink.h @@ -0,0 +1,6 @@ +#ifndef TEST_HARDLINK_H +#define TEST_HARDLINK_H + +void test_hardlink(void); + +#endif From c2df0347ef6423acef7a911e337aef3c52d36f0e Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 06:59:02 +0200 Subject: [PATCH 032/155] fix(client,protocol): EINTR-safe sends, armed abort, keepalive drain grace, TLS WANT_WRITE --- src/client/client_cli.c | 31 +++++++++++++++------ src/client/client_send.c | 8 ++++++ src/client/client_send.h | 4 +++ src/shared/protocol.c | 43 ++++++++++++++++++++++-------- tests/fuzz/fuzz_protocol_framing.c | 5 +++- 5 files changed, 71 insertions(+), 20 deletions(-) diff --git a/src/client/client_cli.c b/src/client/client_cli.c index 73812ef..dd644c4 100644 --- a/src/client/client_cli.c +++ b/src/client/client_cli.c @@ -34,20 +34,35 @@ * client_send.c, still links the symbol. */ volatile sig_atomic_t client_abort_requested = 0; -#ifndef FASTSYNC_TEST_BUILD -/* Signal handler: perform NO work beyond storing the flag. Logging, protocol - * I/O and the STATUS_ABORT frame are all done later on the normal send path, - * which is not async-signal-safe. Only the production client installs it. */ -static void client_signal_handler(int signo) { - (void)signo; - client_abort_requested = 1; +/* Only armed while a network transfer is in flight. Outside that window the + * handler restores the default disposition and re-raises, so purely local modes + * (--list-only/--dry-run/--read-batch/--only-write-batch and the batch-emission + * pass) keep terminating on Ctrl-C/SIGTERM instead of silently swallowing it. */ +volatile sig_atomic_t client_abort_armed = 0; + +void client_set_abort_armed(bool armed) { + client_abort_armed = armed ? 1 : 0; } -#endif bool client_abort_pending(void) { return client_abort_requested != 0; } +#ifndef FASTSYNC_TEST_BUILD +/* Signal handler: perform NO work beyond storing the flag. Logging, protocol + * I/O and the STATUS_ABORT frame are all done later on the normal send path, + * which is not async-signal-safe. When no transfer is armed, fall back to the + * default action so local-only modes remain interruptible. */ +static void client_signal_handler(int signo) { + if (!client_abort_armed) { + signal(signo, SIG_DFL); + raise(signo); + return; + } + client_abort_requested = 1; +} +#endif + #ifndef FASTSYNC_TEST_BUILD /* Parse environment variables for source/destination directories and save-to-disk flag. */ static void parse_environment(const char** out_env_source, const char** out_env_dest, diff --git a/src/client/client_send.c b/src/client/client_send.c index 1febec8..db7a560 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -1949,6 +1949,9 @@ int send_files(Config* config) { return 1; } + /* From here on a server session may be live, so Ctrl-C/SIGTERM should set the + abort flag (and be forwarded as STATUS_ABORT) instead of terminating. */ + client_set_abort_armed(true); Client* client = connect_transfer_client(config); if (!client) { if (config->transport == TRANSPORT_TCP) @@ -2228,6 +2231,7 @@ send_fail: prepared_scanner_destroy(&prepared); disconnect_transfer_client(client); protocol_session_unbind(); + client_set_abort_armed(false); return ret; } @@ -2257,6 +2261,9 @@ int send_files_multithreaded(Config** config_ptr) { return 1; } + /* Armed only once a session may go live (see send_files). */ + client_set_abort_armed(true); + long pages = sysconf(_SC_AVPHYS_PAGES); long page_size = sysconf(_SC_PAGE_SIZE); unsigned long long available_memory = @@ -2411,5 +2418,6 @@ int send_files_multithreaded(Config** config_ptr) { /* --ignore-errors: the run completed (and deleted) past an unreadable source directory; report it as errored like rsync does. */ pipeline_context_sender_destroy(context); + client_set_abort_armed(false); return sender_ok && !scan_io ? 0 : 1; } diff --git a/src/client/client_send.h b/src/client/client_send.h index 06d6c6d..2605899 100644 --- a/src/client/client_send.h +++ b/src/client/client_send.h @@ -13,6 +13,10 @@ * receiver can clean up before the client exits. */ extern volatile sig_atomic_t client_abort_requested; bool client_abort_pending(void); +/* Arm/disarm abort handling around the network phase. While disarmed, a + * SIGINT/SIGTERM takes the default action (immediate termination) so local-only + * modes are not left unresponsive. Defined in client_cli.c. */ +void client_set_abort_armed(bool armed); /* Both sender entry points BORROW `config` for the duration of the call; they * never free it, and the caller retains ownership (freeing it with diff --git a/src/shared/protocol.c b/src/shared/protocol.c index 0a8dac9..57fd2e2 100644 --- a/src/shared/protocol.c +++ b/src/shared/protocol.c @@ -303,6 +303,12 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat wait_events = ssl_err == SSL_ERROR_WANT_WRITE ? POLLOUT : POLLIN; continue; } + /* A signal (e.g. Ctrl-C) interrupts the blocking TLS write: retry so + the send loop can observe the abort flag at the next checkpoint. */ + if (ssl_err == SSL_ERROR_SYSCALL && errno == EINTR) + continue; + } else if (errno == EINTR) { + continue; } log_message(LOG_LEVEL_ERROR, "Could not send data"); return false; @@ -618,6 +624,7 @@ static bool protocol_read_status_until(ProtocolSession* session, Status* status, const struct timespec* deadline) { Status received = STATUS_ERROR; size_t got = 0; + short wait_events = POLLIN; while (got < sizeof(Status)) { if (!session->ssl || SSL_pending(session->ssl) == 0) { int remaining_ms = deadline_remaining_ms(deadline); @@ -625,7 +632,7 @@ static bool protocol_read_status_until(ProtocolSession* session, Status* status, log_message(LOG_LEVEL_ERROR, "Receive timeout while reading status"); return false; } - struct pollfd pfd = {.fd = session->read_fd, .events = POLLIN}; + struct pollfd pfd = {.fd = session->read_fd, .events = wait_events}; int poll_result = poll(&pfd, 1, remaining_ms); if (poll_result == 0) { log_message(LOG_LEVEL_ERROR, "Receive timeout while reading status"); @@ -647,8 +654,10 @@ static bool protocol_read_status_until(ProtocolSession* session, Status* status, if (bytes_received <= 0) { if (session->ssl) { int ssl_err = SSL_get_error(session->ssl, (int)bytes_received); - if (ssl_err == SSL_ERROR_WANT_READ || ssl_err == SSL_ERROR_WANT_WRITE) + if (ssl_err == SSL_ERROR_WANT_READ || ssl_err == SSL_ERROR_WANT_WRITE) { + wait_events = ssl_err == SSL_ERROR_WANT_WRITE ? POLLOUT : POLLIN; continue; + } } if (bytes_received < 0 && errno == EINTR) continue; @@ -689,7 +698,8 @@ bool protocol_receive_status_keepalive(ProtocolSession* session, Status* status, /* Only interleave a keepalive while waiting for the FIRST byte of a * frame; once part of a frame is buffered a write could race the peer's * reply into the middle of it. */ - int interval_ms = keepalive_interval_sec * 1000; + long long interval_ms_ll = (long long)keepalive_interval_sec * 1000LL; + int interval_ms = interval_ms_ll > INT_MAX ? INT_MAX : (int)interval_ms_ll; int wait_ms = interval_ms < remaining_ms ? interval_ms : remaining_ms; struct pollfd pfd = {.fd = session->read_fd, .events = POLLIN}; int poll_result = poll(&pfd, 1, wait_ms); @@ -724,15 +734,26 @@ bool protocol_receive_status_keepalive(ProtocolSession* session, Status* status, * was busy. It answers them only after the real status, so leaving them * unread would put stale KEEPALIVE frames ahead of the next exchange and * desynchronize the protocol. */ - while (replies_seen < keepalives_sent) { - Status drained; - if (!protocol_read_status_until(session, &drained, &deadline)) - return false; - if (drained != STATUS_KEEPALIVE) { - log_message(LOG_LEVEL_ERROR, "Unexpected status while draining keepalive replies"); - return false; + if (replies_seen < keepalives_sent) { + /* A short separate grace, not the (possibly exhausted) main deadline: the + terminal status already arrived, so a peer that never answers its owed + keepalives must not turn a successful ack into a reported failure. */ + struct timespec drain_deadline; + clock_gettime(CLOCK_MONOTONIC, &drain_deadline); + drain_deadline.tv_sec += 1; + while (replies_seen < keepalives_sent) { + Status drained; + if (!protocol_read_status_until(session, &drained, &drain_deadline)) { + log_message(LOG_LEVEL_WARNING, "peer did not answer %lu keepalive(s); continuing", + keepalives_sent - replies_seen); + break; + } + if (drained != STATUS_KEEPALIVE) { + log_message(LOG_LEVEL_ERROR, "Unexpected status while draining keepalive replies"); + return false; + } + replies_seen++; } - replies_seen++; } *status = final; log_debug_message(LOG_DEBUG_PROTO, "Received Status: %s", status_to_string(*status)); diff --git a/tests/fuzz/fuzz_protocol_framing.c b/tests/fuzz/fuzz_protocol_framing.c index 6bb0359..a80350a 100644 --- a/tests/fuzz/fuzz_protocol_framing.c +++ b/tests/fuzz/fuzz_protocol_framing.c @@ -81,9 +81,12 @@ int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { close(w); } + /* Bounded so a crafted 256 MiB length header cannot make each iteration + allocate the full MAX_DATA_PAYLOAD_SIZE under ASan; the framing logic is + identical to receive_data(), which delegates to the limited variant. */ rd = make_stream(data, size, &w); if (rd >= 0) { - Data* d = receive_data(rd); + Data* d = receive_data_limited(rd, 1u << 20); data_destroy(d); close(rd); close(w); From 3499baf80b475fa9a1ad23b099af44867dcd5257 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 07:20:28 +0200 Subject: [PATCH 033/155] build: explicit CMake targets; move receiver pipeline out of shared --- CMakeLists.txt | 184 ++++++++++++++++++++---- src/server/receiver_pipeline.c | 254 +++++++++++++++++++++++++++++++++ src/server/receiver_pipeline.h | 68 +++++++++ src/server/server.c | 2 +- src/shared/multiprocessing.c | 246 ------------------------------- src/shared/multiprocessing.h | 52 ------- tests/test_config.c | 1 + tests/test_multiprocessing.c | 1 + 8 files changed, 482 insertions(+), 326 deletions(-) create mode 100644 src/server/receiver_pipeline.c create mode 100644 src/server/receiver_pipeline.h diff --git a/CMakeLists.txt b/CMakeLists.txt index f384f62..e61b0fa 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -70,36 +70,105 @@ endif() find_package(OpenSSL REQUIRED) -file(GLOB SHARED_SRCS "src/shared/*.c") -set(FILE_STORE_SRCS "${CMAKE_CURRENT_SOURCE_DIR}/src/shared/file_store.c") -list(REMOVE_ITEM SHARED_SRCS ${FILE_STORE_SRCS}) -file(GLOB SERVER_SRCS "src/server/*.c") -set(SERVER_RECEIVER_SRCS src/server/receiver.c) -file(GLOB CLIENT_SRCS "src/client/*.c") +# --- Explicit source lists --- +# The shared library is self-contained: it must never depend on the client or +# server modules. In particular, the receiver pipeline (receive_thread / +# write_thread) lives under src/server, not here, so the client executable can +# link the shared library without pulling in any server code. +set(SHARED_SRCS + src/shared/array_list.c + src/shared/batch.c + src/shared/charset.c + src/shared/checksum.c + src/shared/chmod.c + src/shared/chunk.c + src/shared/compression.c + src/shared/config.c + src/shared/credentials.c + src/shared/daemon_conf.c + src/shared/data.c + src/shared/delay_updates.c + src/shared/delta.c + src/shared/file.c + src/shared/file_list.c + src/shared/file_receive.c + src/shared/file_send.c + src/shared/file_store.c + src/shared/filter.c + src/shared/hardlink.c + src/shared/identity.c + src/shared/log.c + src/shared/metadata.c + src/shared/motd.c + src/shared/multiprocessing.c + src/shared/protocol.c + src/shared/queue.c + src/shared/stop_condition.c + src/shared/transport_ssh.c + src/shared/transport_tcp.c + src/shared/transport_tls.c + src/shared/utils.c + src/shared/xattr.c +) + +# Server implementation (no main): the receiver read/write pipeline plus the +# CLI parser. The server executable adds its own main (server.c). +set(SERVER_CORE_SRCS + src/server/receiver.c + src/server/receiver_pipeline.c + src/server/server_cli.c +) +set(SERVER_MAIN_SRCS src/server/server.c) + +# Client implementation (no main): everything except the CLI entry point. +set(CLIENT_CORE_SRCS + src/client/change_list.c + src/client/client_send.c + src/client/client_validation.c + src/client/scanner.c + src/client/usage.c +) +set(CLIENT_MAIN_SRCS src/client/client_cli.c) + +# --- Library targets --- +add_library(fastsync_shared STATIC ${SHARED_SRCS}) +target_include_directories(fastsync_shared PUBLIC src/shared) +target_link_libraries(fastsync_shared PUBLIC Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL + OpenSSL::Crypto xxhash) + +add_library(fastsync_client_core STATIC ${CLIENT_CORE_SRCS}) +target_include_directories(fastsync_client_core PUBLIC src/client) +target_link_libraries(fastsync_client_core PUBLIC fastsync_shared) + +add_library(fastsync_server_core STATIC ${SERVER_CORE_SRCS}) +target_include_directories(fastsync_server_core PUBLIC src/server) +target_link_libraries(fastsync_server_core PUBLIC fastsync_shared) # --- Main executables --- -add_executable(server ${SERVER_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS}) -target_include_directories(server PRIVATE src/shared src/server src/client) -target_link_libraries(server PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash) +# The client links only the shared library and its own core; it deliberately +# does NOT get src/server on its include path nor compile receiver.c. +add_executable(server ${SERVER_MAIN_SRCS}) +target_link_libraries(server PRIVATE fastsync_server_core) -add_executable(client ${CLIENT_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS}) -target_include_directories(client PRIVATE src/shared src/server src/client) -target_link_libraries(client PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash) +add_executable(client ${CLIENT_MAIN_SRCS}) +target_link_libraries(client PRIVATE fastsync_client_core) # --- Production hardening --- # Each compile flag is probed so a compiler/architecture that lacks it still # configures cleanly. _FORTIFY_SOURCE is guarded separately because it only # works in an optimising build. xxHash is a static archive built by -# FetchContent, so it must be position-independent for the -pie link. +# FetchContent, so it must be position-independent for the -pie link; the same +# applies to the first-party static libraries linked into the -pie binaries. if(HARDENING_ACTIVE) - set_target_properties(xxhash PROPERTIES POSITION_INDEPENDENT_CODE ON) + set_target_properties(xxhash fastsync_shared fastsync_server_core fastsync_client_core + PROPERTIES POSITION_INDEPENDENT_CODE ON) include(CheckCCompilerFlag) foreach(flag -fstack-protector-strong -fstack-clash-protection -fPIE) string(MAKE_C_IDENTIFIER "HARDEN_${flag}" _harden_var) check_c_compiler_flag("${flag}" ${_harden_var}) endforeach() check_c_compiler_flag("-D_FORTIFY_SOURCE=2" HARDEN_FORTIFY_SOURCE) - foreach(target server client) + foreach(target fastsync_shared fastsync_server_core fastsync_client_core server client) foreach(flag -fstack-protector-strong -fstack-clash-protection -fPIE) string(MAKE_C_IDENTIFIER "HARDEN_${flag}" _harden_var) if(${_harden_var}) @@ -109,6 +178,8 @@ if(HARDENING_ACTIVE) if(HARDEN_FORTIFY_SOURCE) target_compile_options(${target} PRIVATE -D_FORTIFY_SOURCE=2) endif() + endforeach() + foreach(target server client) target_link_options(${target} PRIVATE -pie -Wl,-z,relro -Wl,-z,now -Wl,-z,noexecstack) endforeach() endif() @@ -116,16 +187,59 @@ endif() # --- Testing --- enable_testing() -# Common test libraries -set(TEST_LIBS Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash) -set(TEST_INCLUDES tests src/shared src/server src/client) +# --- Unit tests --- +# The monolithic test binary exercises both client and server code, so it is +# the one place that legitimately sees both include directories and links both +# core libraries. client_cli.c is compiled here directly (with the test build +# define) rather than linked from fastsync_client_core so its test-only shims +# and the absence of main() are preserved. +set(TEST_SRCS + tests/runner.c + tests/test_array_list.c + tests/test_batch.c + tests/test_change_list.c + tests/test_checksum.c + tests/test_chunk.c + tests/test_client_cli.c + tests/test_compression.c + tests/test_config.c + tests/test_credentials.c + tests/test_daemon_conf.c + tests/test_data.c + tests/test_delay_updates.c + tests/test_delta.c + tests/test_file.c + tests/test_file_list.c + tests/test_file_sendfile.c + tests/test_fuzz_smoke.c + tests/test_glob.c + tests/test_hardlink.c + tests/test_iconv.c + tests/test_log.c + tests/test_metadata.c + tests/test_motd.c + tests/test_multiprocessing.c + tests/test_property.c + tests/test_protocol.c + tests/test_queue.c + tests/test_receiver_timeout.c + tests/test_robustness.c + tests/test_scanner.c + tests/test_server.c + tests/test_server_cli.c + tests/test_shared_utils.c + tests/test_stop.c + tests/test_stress.c + tests/test_transport_ssh.c + tests/test_transport_tcp.c + tests/test_transport_tls.c + tests/test_xattr.c +) -# Monolithic test binary (backward compatible) -file(GLOB TEST_SRCS "tests/test_*.c" "tests/runner.c") -add_executable(tests ${TEST_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS} src/client/scanner.c src/client/change_list.c src/client/client_cli.c src/client/client_validation.c src/client/usage.c src/server/server_cli.c) -target_include_directories(tests PRIVATE ${TEST_INCLUDES}) +add_executable(tests ${TEST_SRCS} src/client/client_cli.c) +target_include_directories(tests PRIVATE tests) target_compile_definitions(tests PRIVATE FASTSYNC_TEST_BUILD) -target_link_libraries(tests PRIVATE ${TEST_LIBS}) +target_link_libraries(tests PRIVATE fastsync_server_core fastsync_client_core) add_test(NAME unit_all COMMAND tests) # --- Fuzz targets (requires clang) --- @@ -134,13 +248,29 @@ if(ENABLE_FUZZ) if(NOT CMAKE_C_COMPILER_ID MATCHES "Clang") message(FATAL_ERROR "ENABLE_FUZZ requires Clang (compiler is ${CMAKE_C_COMPILER_ID})") endif() - file(GLOB FUZZ_SRCS "tests/fuzz/*.c") + set(FUZZ_SRCS + tests/fuzz/fuzz_chunk_deserialize.c + tests/fuzz/fuzz_compress_decompress.c + tests/fuzz/fuzz_config_receive.c + tests/fuzz/fuzz_delta_deserialize.c + tests/fuzz/fuzz_delta_signature_deserialize.c + tests/fuzz/fuzz_glob_match.c + tests/fuzz/fuzz_identity_parse.c + tests/fuzz/fuzz_manifest.c + tests/fuzz/fuzz_metadata_from_buf.c + tests/fuzz/fuzz_protocol_framing.c + tests/fuzz/fuzz_xattr_block.c + ) + # Compile the sources under test directly so libFuzzer's coverage + # instrumentation sees them (static libraries would be uninstrumented). + set(FUZZ_CORE_SRCS ${SHARED_SRCS} src/server/receiver.c src/server/receiver_pipeline.c) foreach(FUZZ_SRC ${FUZZ_SRCS}) get_filename_component(FUZZ_NAME ${FUZZ_SRC} NAME_WE) - add_executable(${FUZZ_NAME} ${FUZZ_SRC} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS}) - target_include_directories(${FUZZ_NAME} PRIVATE ${TEST_INCLUDES}) + add_executable(${FUZZ_NAME} ${FUZZ_SRC} ${FUZZ_CORE_SRCS}) + target_include_directories(${FUZZ_NAME} PRIVATE tests src/shared src/server) target_compile_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined -fno-omit-frame-pointer) target_link_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined) - target_link_libraries(${FUZZ_NAME} PRIVATE ${TEST_LIBS}) + target_link_libraries(${FUZZ_NAME} PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL + OpenSSL::Crypto xxhash) endforeach() endif() diff --git a/src/server/receiver_pipeline.c b/src/server/receiver_pipeline.c new file mode 100644 index 0000000..abe8719 --- /dev/null +++ b/src/server/receiver_pipeline.c @@ -0,0 +1,254 @@ +#include "receiver_pipeline.h" + +#include "log.h" +#include "protocol.h" +#include "queue.h" +#include "utils.h" +#include +#include +#include + +PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue, + int file_descriptor, SSL* ssl) { + PipelineContextReceiver* context = malloc(sizeof(PipelineContextReceiver)); + if (context == NULL) + return NULL; + context->config = config; + context->queue = queue; + context->file_descriptor = file_descriptor; + context->ssl = ssl; + context->outcomes.entries = NULL; + context->outcomes.count = 0; + context->outcomes.capacity = 0; + dir_time_list_init(&context->dir_times); + protocol_session_init(&context->session, file_descriptor, file_descriptor); + protocol_session_set_ssl(&context->session, ssl); + context->receiver_done = false; + context->queued_bytes = 0; + context->max_queue_bytes = 0; + context->deferred_manifest = NULL; + atomic_init(&context->cancelled, false); + int init = 0; + if (mtx_init(&context->mutex, mtx_plain) != thrd_success) + goto fail; + init++; + if (cnd_init(&context->condition_not_full) != thrd_success) + goto fail; + init++; + if (cnd_init(&context->condition_not_empty) != thrd_success) + goto fail; + // cppcheck-suppress unreadVariable + init++; + return context; + +fail: + log_perror("Error initializing synchronization objects"); + if (init >= 3) + cnd_destroy(&context->condition_not_empty); + if (init >= 2) + cnd_destroy(&context->condition_not_full); + if (init >= 1) + mtx_destroy(&context->mutex); + free(context); + return NULL; +} + +void pipeline_context_receiver_destroy(PipelineContextReceiver* context) { + config_delete(context->config); + if (context->deferred_manifest) + delete_manifest_free(context->deferred_manifest); + queue_destroy(context->queue); + receiver_outcomes_destroy(&context->outcomes); + dir_time_list_free(&context->dir_times); + mtx_destroy(&context->mutex); + cnd_destroy(&context->condition_not_full); + cnd_destroy(&context->condition_not_empty); + free(context); +} + +void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context, + size_t max_bytes) { + if (context == NULL) + return; + mtx_lock(&context->mutex); + context->max_queue_bytes = max_bytes; + context->queued_bytes = 0; + cnd_broadcast(&context->condition_not_full); + mtx_unlock(&context->mutex); +} + +void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context, + size_t released_bytes) { + if (context == NULL || context->max_queue_bytes == 0 || released_bytes == 0) + return; + mtx_lock(&context->mutex); + if (released_bytes >= context->queued_bytes) + context->queued_bytes = 0; + else + context->queued_bytes -= released_bytes; + cnd_signal(&context->condition_not_full); + mtx_unlock(&context->mutex); +} + +bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file) { + if (context == NULL || file == NULL) + return false; + size_t file_bytes = file->data ? file->data->size : 0; + mtx_lock(&context->mutex); + while (!atomic_load(&context->cancelled)) { + bool blocked_by_count = queue_is_full(context->queue); + bool blocked_by_budget = false; + if (context->max_queue_bytes > 0) { + size_t budget = context->max_queue_bytes; + size_t used = context->queued_bytes; + if (used >= budget) { + blocked_by_budget = true; + } else if (file_bytes > budget - used) { + /* A single payload larger than the whole budget (not possible with + the per-file receive cap) is only admitted to an empty pipeline so + the wait can never deadlock. */ + blocked_by_budget = used != 0; + } + } + if (!blocked_by_count && !blocked_by_budget) + break; + cnd_wait(&context->condition_not_full, &context->mutex); + } + if (atomic_load(&context->cancelled)) { + mtx_unlock(&context->mutex); + file_destroy(file); + return false; + } + if (!queue_enqueue(context->queue, file)) { + mtx_unlock(&context->mutex); + file_destroy(file); + return false; + } + context->queued_bytes += file_bytes; + cnd_signal(&context->condition_not_empty); + mtx_unlock(&context->mutex); + return true; +} + +static bool receiver_enqueue_file(File* file, void* context_pointer) { + PipelineContextReceiver* context = (PipelineContextReceiver*)context_pointer; + return pipeline_context_receiver_enqueue_file(context, file); +} + +static void receiver_thread_fail(PipelineContextReceiver* context) { + mtx_lock(&context->mutex); + atomic_store(&context->cancelled, true); + context->receiver_done = true; + cnd_broadcast(&context->condition_not_empty); + cnd_broadcast(&context->condition_not_full); + mtx_unlock(&context->mutex); +} + +int receive_thread(void* pipeline_context) { + PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context; + protocol_session_bind(&context->session); + mtx_lock(&context->mutex); + int file_descriptor = context->file_descriptor; + const Config* config = context->config; + mtx_unlock(&context->mutex); + + ReceiverSink sink = {receiver_enqueue_file, context, false, false, NULL}; + if (receiver_process_pending((Config*)config, file_descriptor, &sink, + &context->deferred_manifest) != 0) { + receiver_thread_fail(context); + protocol_session_unbind(); + return thrd_error; + } + mtx_lock(&context->mutex); + context->receiver_done = true; + cnd_signal(&context->condition_not_empty); + mtx_unlock(&context->mutex); + protocol_session_unbind(); + return thrd_success; +} + +int write_thread(void* pipeline_context) { + PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context; + protocol_session_bind(&context->session); + mtx_lock(&context->mutex); + bool save_to_disk = context->config->save_to_disk; + char* root_directory = str_dup(context->config->receive_root_directory); + mtx_unlock(&context->mutex); + if (save_to_disk && !root_directory) { + mtx_lock(&context->mutex); + atomic_store(&context->cancelled, true); + context->receiver_done = true; + cnd_broadcast(&context->condition_not_full); + cnd_broadcast(&context->condition_not_empty); + mtx_unlock(&context->mutex); + protocol_session_unbind(); + return thrd_error; + } + + while (true) { + File* file = + queue_dequeue_multithreaded(context->queue, &context->mutex, &context->condition_not_empty, + &context->condition_not_full, &context->receiver_done); + if (file == NULL) { + free(root_directory); + protocol_session_unbind(); + return thrd_success; + } + size_t file_bytes = file->data ? file->data->size : 0; + FileSaveResult result = FILE_SAVE_SKIPPED; + if (save_to_disk) { + result = file_save_to_disk_full(root_directory, file, context->config); + if (result == FILE_SAVE_ERROR) { + file_destroy(file); + pipeline_context_receiver_note_bytes_released(context, file_bytes); + mtx_lock(&context->mutex); + atomic_store(&context->cancelled, true); + context->receiver_done = true; + cnd_broadcast(&context->condition_not_full); + cnd_broadcast(&context->condition_not_empty); + mtx_unlock(&context->mutex); + free(root_directory); + protocol_session_unbind(); + return thrd_error; + } + } + /* P7 Wave D: a directory's times are never applied inline (a later child + write would clobber them); accumulate the metadata here and let the + caller apply it once every writer has drained. */ + if (result != FILE_SAVE_ERROR && file->is_dir && file->metadata && + dir_times_should_capture(context->config) && + !dir_time_list_add(&context->dir_times, file->path, file->metadata)) { + file_destroy(file); + pipeline_context_receiver_note_bytes_released(context, file_bytes); + mtx_lock(&context->mutex); + atomic_store(&context->cancelled, true); + context->receiver_done = true; + cnd_broadcast(&context->condition_not_full); + cnd_broadcast(&context->condition_not_empty); + mtx_unlock(&context->mutex); + free(root_directory); + protocol_session_unbind(); + return thrd_error; + } + /* Record the per-file outcome so a --remove-source-files sender learns + which sources were actually written versus skipped on the receiver. + Explicit directory entries and recreated device/special nodes have no + source and are never acknowledged (mirrors receiver.c). */ + if (context->config->remove_source_files && !file->is_dir && !file->is_special && !file->skip && + !receiver_outcomes_append(&context->outcomes, (unsigned char)result)) { + file_destroy(file); + pipeline_context_receiver_note_bytes_released(context, file_bytes); + mtx_lock(&context->mutex); + atomic_store(&context->cancelled, true); + context->receiver_done = true; + cnd_broadcast(&context->condition_not_full); + cnd_broadcast(&context->condition_not_empty); + mtx_unlock(&context->mutex); + free(root_directory); + protocol_session_unbind(); + return thrd_error; + } + file_destroy(file); + pipeline_context_receiver_note_bytes_released(context, file_bytes); + } +} diff --git a/src/server/receiver_pipeline.h b/src/server/receiver_pipeline.h new file mode 100644 index 0000000..7aef579 --- /dev/null +++ b/src/server/receiver_pipeline.h @@ -0,0 +1,68 @@ +#ifndef RECEIVER_PIPELINE_H +#define RECEIVER_PIPELINE_H + +#include +#include +#include + +#include "config.h" +#include "file.h" +#include "file_receive.h" +#include "protocol.h" +#include "queue.h" +#include "receiver.h" +#include + +typedef struct PipelineContextReceiver { + Queue* queue; + Config* config; + int file_descriptor; + SSL* ssl; + ProtocolSession session; + ReceiverOutcomes outcomes; + mtx_t mutex; + cnd_t condition_not_full; + cnd_t condition_not_empty; + bool receiver_done; + atomic_bool cancelled; + /* Aggregate payload bytes that have been received but not yet released by + the disk writer (queued or in the writer's hand). Guarded by `mutex`. + When `max_queue_bytes` is non-zero the receiver blocks before enqueuing + once this total would exceed it, so decompressed/copied file payloads + buffered ahead of a slow disk writer respect the per-connection memory + budget instead of growing without bound. */ + size_t queued_bytes; + size_t max_queue_bytes; + /* Keep-set manifest for the commit-style (late) deletion + (--delete/--delete-after/--delete-delay). receive_thread parses the whole + protocol stream but hands the manifest here instead of deleting while the + disk writer may still be draining; the caller (server.c) commits the + deletion after both threads have joined, so no extra is removed unless the + transfer truly succeeded. NULL in the early delete modes (which delete at + the manifest). */ + DeleteManifest* deferred_manifest; + /* P7 Wave D: directory metadata collected by write_thread from received + directory entries. Only write_thread mutates it (before it joins); the + caller (server.c) applies it after the delete/delay-updates phase. */ + DirTimeList dir_times; +} PipelineContextReceiver; + +PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue_receiver, + int file_descriptor, SSL* ssl); +void pipeline_context_receiver_destroy(PipelineContextReceiver* context); +/* Bound the bytes buffered ahead of the disk writer (see max_queue_bytes). */ +void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context, + size_t max_bytes); +/* Blocking enqueue used by the receive pipeline sink. Blocks while the queue + is full by element count or when adding `file` would push queued_bytes over + the configured byte limit; waits until the disk writer releases bytes. + Takes ownership of `file` on success and destroys it on failure/cancel. */ +bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file); +/* Account for `released_bytes` of payload memory that has been freed by the + disk writer, unblocking a receiver that is waiting on the byte limit. */ +void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context, + size_t released_bytes); +int receive_thread(void* pipeline_context); +int write_thread(void* pipeline_context); + +#endif diff --git a/src/server/server.c b/src/server/server.c index da3e65d..494a840 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -7,10 +7,10 @@ #include "identity.h" #include "log.h" #include "motd.h" -#include "multiprocessing.h" #include "protocol.h" #include "queue.h" #include "receiver.h" +#include "receiver_pipeline.h" #include "server_cli.h" #include "transport_tcp.h" #include "transport_tls.h" diff --git a/src/shared/multiprocessing.c b/src/shared/multiprocessing.c index 14ad46e..2155837 100644 --- a/src/shared/multiprocessing.c +++ b/src/shared/multiprocessing.c @@ -1,5 +1,4 @@ #include "multiprocessing.h" -#include "receiver.h" #include "array_list.h" #include "chunk.h" @@ -206,248 +205,3 @@ void pipeline_context_sender_destroy(PipelineContextSender* context) { mtx_destroy(&context->mutex_progress); free(context); } - -PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue, - int file_descriptor, SSL* ssl) { - PipelineContextReceiver* context = malloc(sizeof(PipelineContextReceiver)); - if (context == NULL) - return NULL; - context->config = config; - context->queue = queue; - context->file_descriptor = file_descriptor; - context->ssl = ssl; - context->outcomes.entries = NULL; - context->outcomes.count = 0; - context->outcomes.capacity = 0; - dir_time_list_init(&context->dir_times); - protocol_session_init(&context->session, file_descriptor, file_descriptor); - protocol_session_set_ssl(&context->session, ssl); - context->receiver_done = false; - context->queued_bytes = 0; - context->max_queue_bytes = 0; - context->deferred_manifest = NULL; - atomic_init(&context->cancelled, false); - int init = 0; - if (mtx_init(&context->mutex, mtx_plain) != thrd_success) - goto fail; - init++; - if (cnd_init(&context->condition_not_full) != thrd_success) - goto fail; - init++; - if (cnd_init(&context->condition_not_empty) != thrd_success) - goto fail; - // cppcheck-suppress unreadVariable - init++; - return context; - -fail: - log_perror("Error initializing synchronization objects"); - if (init >= 3) - cnd_destroy(&context->condition_not_empty); - if (init >= 2) - cnd_destroy(&context->condition_not_full); - if (init >= 1) - mtx_destroy(&context->mutex); - free(context); - return NULL; -} - -void pipeline_context_receiver_destroy(PipelineContextReceiver* context) { - config_delete(context->config); - if (context->deferred_manifest) - delete_manifest_free(context->deferred_manifest); - queue_destroy(context->queue); - receiver_outcomes_destroy(&context->outcomes); - dir_time_list_free(&context->dir_times); - mtx_destroy(&context->mutex); - cnd_destroy(&context->condition_not_full); - cnd_destroy(&context->condition_not_empty); - free(context); -} - -void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context, - size_t max_bytes) { - if (context == NULL) - return; - mtx_lock(&context->mutex); - context->max_queue_bytes = max_bytes; - context->queued_bytes = 0; - cnd_broadcast(&context->condition_not_full); - mtx_unlock(&context->mutex); -} - -void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context, - size_t released_bytes) { - if (context == NULL || context->max_queue_bytes == 0 || released_bytes == 0) - return; - mtx_lock(&context->mutex); - if (released_bytes >= context->queued_bytes) - context->queued_bytes = 0; - else - context->queued_bytes -= released_bytes; - cnd_signal(&context->condition_not_full); - mtx_unlock(&context->mutex); -} - -bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file) { - if (context == NULL || file == NULL) - return false; - size_t file_bytes = file->data ? file->data->size : 0; - mtx_lock(&context->mutex); - while (!atomic_load(&context->cancelled)) { - bool blocked_by_count = queue_is_full(context->queue); - bool blocked_by_budget = false; - if (context->max_queue_bytes > 0) { - size_t budget = context->max_queue_bytes; - size_t used = context->queued_bytes; - if (used >= budget) { - blocked_by_budget = true; - } else if (file_bytes > budget - used) { - /* A single payload larger than the whole budget (not possible with - the per-file receive cap) is only admitted to an empty pipeline so - the wait can never deadlock. */ - blocked_by_budget = used != 0; - } - } - if (!blocked_by_count && !blocked_by_budget) - break; - cnd_wait(&context->condition_not_full, &context->mutex); - } - if (atomic_load(&context->cancelled)) { - mtx_unlock(&context->mutex); - file_destroy(file); - return false; - } - if (!queue_enqueue(context->queue, file)) { - mtx_unlock(&context->mutex); - file_destroy(file); - return false; - } - context->queued_bytes += file_bytes; - cnd_signal(&context->condition_not_empty); - mtx_unlock(&context->mutex); - return true; -} - -static bool receiver_enqueue_file(File* file, void* context_pointer) { - PipelineContextReceiver* context = (PipelineContextReceiver*)context_pointer; - return pipeline_context_receiver_enqueue_file(context, file); -} - -static void receiver_thread_fail(PipelineContextReceiver* context) { - mtx_lock(&context->mutex); - atomic_store(&context->cancelled, true); - context->receiver_done = true; - cnd_broadcast(&context->condition_not_empty); - cnd_broadcast(&context->condition_not_full); - mtx_unlock(&context->mutex); -} - -int receive_thread(void* pipeline_context) { - PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context; - protocol_session_bind(&context->session); - mtx_lock(&context->mutex); - int file_descriptor = context->file_descriptor; - const Config* config = context->config; - mtx_unlock(&context->mutex); - - ReceiverSink sink = {receiver_enqueue_file, context, false, false, NULL}; - if (receiver_process_pending((Config*)config, file_descriptor, &sink, - &context->deferred_manifest) != 0) { - receiver_thread_fail(context); - protocol_session_unbind(); - return thrd_error; - } - mtx_lock(&context->mutex); - context->receiver_done = true; - cnd_signal(&context->condition_not_empty); - mtx_unlock(&context->mutex); - protocol_session_unbind(); - return thrd_success; -} - -int write_thread(void* pipeline_context) { - PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context; - protocol_session_bind(&context->session); - mtx_lock(&context->mutex); - bool save_to_disk = context->config->save_to_disk; - char* root_directory = str_dup(context->config->receive_root_directory); - mtx_unlock(&context->mutex); - if (save_to_disk && !root_directory) { - mtx_lock(&context->mutex); - atomic_store(&context->cancelled, true); - context->receiver_done = true; - cnd_broadcast(&context->condition_not_full); - cnd_broadcast(&context->condition_not_empty); - mtx_unlock(&context->mutex); - protocol_session_unbind(); - return thrd_error; - } - - while (true) { - File* file = - queue_dequeue_multithreaded(context->queue, &context->mutex, &context->condition_not_empty, - &context->condition_not_full, &context->receiver_done); - if (file == NULL) { - free(root_directory); - protocol_session_unbind(); - return thrd_success; - } - size_t file_bytes = file->data ? file->data->size : 0; - FileSaveResult result = FILE_SAVE_SKIPPED; - if (save_to_disk) { - result = file_save_to_disk_full(root_directory, file, context->config); - if (result == FILE_SAVE_ERROR) { - file_destroy(file); - pipeline_context_receiver_note_bytes_released(context, file_bytes); - mtx_lock(&context->mutex); - atomic_store(&context->cancelled, true); - context->receiver_done = true; - cnd_broadcast(&context->condition_not_full); - cnd_broadcast(&context->condition_not_empty); - mtx_unlock(&context->mutex); - free(root_directory); - protocol_session_unbind(); - return thrd_error; - } - } - /* P7 Wave D: a directory's times are never applied inline (a later child - write would clobber them); accumulate the metadata here and let the - caller apply it once every writer has drained. */ - if (result != FILE_SAVE_ERROR && file->is_dir && file->metadata && - dir_times_should_capture(context->config) && - !dir_time_list_add(&context->dir_times, file->path, file->metadata)) { - file_destroy(file); - pipeline_context_receiver_note_bytes_released(context, file_bytes); - mtx_lock(&context->mutex); - atomic_store(&context->cancelled, true); - context->receiver_done = true; - cnd_broadcast(&context->condition_not_full); - cnd_broadcast(&context->condition_not_empty); - mtx_unlock(&context->mutex); - free(root_directory); - protocol_session_unbind(); - return thrd_error; - } - /* Record the per-file outcome so a --remove-source-files sender learns - which sources were actually written versus skipped on the receiver. - Explicit directory entries and recreated device/special nodes have no - source and are never acknowledged (mirrors receiver.c). */ - if (context->config->remove_source_files && !file->is_dir && !file->is_special && !file->skip && - !receiver_outcomes_append(&context->outcomes, (unsigned char)result)) { - file_destroy(file); - pipeline_context_receiver_note_bytes_released(context, file_bytes); - mtx_lock(&context->mutex); - atomic_store(&context->cancelled, true); - context->receiver_done = true; - cnd_broadcast(&context->condition_not_full); - cnd_broadcast(&context->condition_not_empty); - mtx_unlock(&context->mutex); - free(root_directory); - protocol_session_unbind(); - return thrd_error; - } - file_destroy(file); - pipeline_context_receiver_note_bytes_released(context, file_bytes); - } -} diff --git a/src/shared/multiprocessing.h b/src/shared/multiprocessing.h index 6de2e98..4bf1fac 100644 --- a/src/shared/multiprocessing.h +++ b/src/shared/multiprocessing.h @@ -10,7 +10,6 @@ #include "file.h" #include "protocol.h" #include "queue.h" -#include "receiver.h" #include "stop_condition.h" #include @@ -86,40 +85,6 @@ typedef struct { bool dir_entries_mutex_init; } PipelineContextSender; -typedef struct PipelineContextReceiver { - Queue* queue; - Config* config; - int file_descriptor; - SSL* ssl; - ProtocolSession session; - ReceiverOutcomes outcomes; - mtx_t mutex; - cnd_t condition_not_full; - cnd_t condition_not_empty; - bool receiver_done; - atomic_bool cancelled; - /* Aggregate payload bytes that have been received but not yet released by - the disk writer (queued or in the writer's hand). Guarded by `mutex`. - When `max_queue_bytes` is non-zero the receiver blocks before enqueuing - once this total would exceed it, so decompressed/copied file payloads - buffered ahead of a slow disk writer respect the per-connection memory - budget instead of growing without bound. */ - size_t queued_bytes; - size_t max_queue_bytes; - /* Keep-set manifest for the commit-style (late) deletion - (--delete/--delete-after/--delete-delay). receive_thread parses the whole - protocol stream but hands the manifest here instead of deleting while the - disk writer may still be draining; the caller (server.c) commits the - deletion after both threads have joined, so no extra is removed unless the - transfer truly succeeded. NULL in the early delete modes (which delete at - the manifest). */ - DeleteManifest* deferred_manifest; - /* P7 Wave D: directory metadata collected by write_thread from received - directory entries. Only write_thread mutates it (before it joins); the - caller (server.c) applies it after the delete/delay-updates phase. */ - DirTimeList dir_times; -} PipelineContextReceiver; - /* `config` is borrowed and must outlive the context: destroy does NOT free it, so the caller owns it and frees it with config_delete() afterwards. */ PipelineContextSender* pipeline_context_sender_create(Config* config, Queue* queue_scanner, @@ -141,21 +106,4 @@ bool pipeline_context_sender_enqueue_chunk(PipelineContextSender* context, Chunk destroying a chunk, unblocking a loader waiting on the byte limit. */ void pipeline_context_sender_note_bytes_released(PipelineContextSender* context, size_t released_bytes); -PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue_receiver, - int file_descriptor, SSL* ssl); -void pipeline_context_receiver_destroy(PipelineContextReceiver* context); -/* Bound the bytes buffered ahead of the disk writer (see max_queue_bytes). */ -void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context, - size_t max_bytes); -/* Blocking enqueue used by the receive pipeline sink. Blocks while the queue - is full by element count or when adding `file` would push queued_bytes over - the configured byte limit; waits until the disk writer releases bytes. - Takes ownership of `file` on success and destroys it on failure/cancel. */ -bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file); -/* Account for `released_bytes` of payload memory that has been freed by the - disk writer, unblocking a receiver that is waiting on the byte limit. */ -void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context, - size_t released_bytes); -int receive_thread(void* pipeline_context); -int write_thread(void* pipeline_context); #endif diff --git a/tests/test_config.c b/tests/test_config.c index c9e672e..7689cf5 100644 --- a/tests/test_config.c +++ b/tests/test_config.c @@ -4,6 +4,7 @@ #include "multiprocessing.h" #include "protocol.h" #include "queue.h" +#include "receiver_pipeline.h" #include "test_utils.h" #include "utils.h" #include diff --git a/tests/test_multiprocessing.c b/tests/test_multiprocessing.c index a051897..1a0e6f8 100644 --- a/tests/test_multiprocessing.c +++ b/tests/test_multiprocessing.c @@ -3,6 +3,7 @@ #include "config.h" #include "protocol.h" #include "queue.h" +#include "receiver_pipeline.h" #include "utils.h" #include "test_utils.h" #include From 0155902d95b03ab5778357ce0101b4e644e2b0ae Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 07:30:28 +0200 Subject: [PATCH 034/155] docs: update stale PipelineContextReceiver reference --- src/shared/delay_updates.h | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/src/shared/delay_updates.h b/src/shared/delay_updates.h index 225c19b..9e03311 100644 --- a/src/shared/delay_updates.h +++ b/src/shared/delay_updates.h @@ -18,8 +18,9 @@ typedef struct { /* Receiver-side --delay-updates staging registry. All successfully written files land under a private staging directory inside the receive root and are atomically renamed into their final destination only at the very end of the - transfer. A single PipelineContextReceiver has exactly one writer thread, - but the registry is still mutex-protected so the same object can be safely + transfer. A single receiver pipeline (see src/server/receiver_pipeline.h) + has exactly one writer thread, but the registry is still mutex-protected so + the same object can be safely shared with the publish/cleanup phase that runs after the threads join. */ typedef struct DelayUpdatesContext { char* root_directory; /* receive root the staging dir lives under */ From 5d3c43305e5d680784fe194af23a66c3fcee147d Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:05:38 +0200 Subject: [PATCH 035/155] fix(protocol): release Data charge to its owning session Data charged against a ProtocolSession kept only the charge amount, so data_destroy released it from whatever session was thread-locally bound at destroy time. Destroying a received Data on another thread, after the session was unbound, or while a different session was bound leaked the originating session's budget and underflowed the other's. Add Data.owner, set it whenever protocol_receive_data_limited charges a session, and have data_destroy release against that owner directly via the newly-exported protocol_release_memory_for_session. Uncharged Data (owner NULL) keeps the previous bound-session fallback. Add a unit test proving a Data acquired on session A is released to A even when unrelated session B is bound at destroy time. --- src/client/client_send.c | 1 + src/shared/data.c | 10 ++++++-- src/shared/data.h | 11 +++++++++ src/shared/protocol.c | 3 ++- tests/test_protocol.c | 49 ++++++++++++++++++++++++++++++++++++++++ 5 files changed, 71 insertions(+), 3 deletions(-) diff --git a/src/client/client_send.c b/src/client/client_send.c index db7a560..e220fd4 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -1157,6 +1157,7 @@ static int send_append(const Client* client, File* file, Config* config, tail_view.data = (char*)file->data->data + off; tail_view.size = tail_len; tail_view.protocol_charge = 0; + tail_view.owner = NULL; ok = send_data(fd, &tail_view); } return ok ? 0 : -1; diff --git a/src/shared/data.c b/src/shared/data.c index 55af503..2ec189c 100644 --- a/src/shared/data.c +++ b/src/shared/data.c @@ -23,6 +23,7 @@ Data* data_create_reserve(size_t size) { d->data = NULL; d->size = size; d->protocol_charge = 0; + d->owner = NULL; return d; } @@ -36,14 +37,19 @@ Data* data_create(void* data, size_t data_size) { new_data->data = data; new_data->size = data_size; new_data->protocol_charge = 0; + new_data->owner = NULL; return new_data; } void data_destroy(Data* data) { if (data == NULL) return; - if (data->protocol_charge != 0) - protocol_release_memory(data->protocol_charge); + if (data->protocol_charge != 0) { + if (data->owner != NULL) + protocol_release_memory_for_session(data->owner, data->protocol_charge); + else + protocol_release_memory(data->protocol_charge); + } free(data->data); free(data); } diff --git a/src/shared/data.h b/src/shared/data.h index b65ae29..8112976 100644 --- a/src/shared/data.h +++ b/src/shared/data.h @@ -3,11 +3,19 @@ #include +/* Forward declaration for the connection budget a received Data is charged + * against; defined in protocol.h (which includes this header). */ +typedef struct ProtocolSession ProtocolSession; + typedef struct { void* data; size_t size; /* Non-zero only for a buffer charged to the protocol connection budget. */ size_t protocol_charge; + /* Session whose budget `protocol_charge` was reserved from. The charge must + * always be returned to this session, regardless of which session (if any) is + * bound to the destroying thread. NULL for uncharged Data. */ + ProtocolSession* owner; } Data; Data* data_create_empty(size_t data_size); @@ -15,5 +23,8 @@ Data* data_create_reserve(size_t size); Data* data_create(void* data, size_t data_size); void data_destroy(Data* data); void protocol_release_memory(size_t charge); +/* Release `charge` against `session` directly instead of the thread-local bound + * session. Used by data_destroy to honor Data.owner. */ +void protocol_release_memory_for_session(ProtocolSession* session, size_t charge); #endif diff --git a/src/shared/protocol.c b/src/shared/protocol.c index 57fd2e2..4527296 100644 --- a/src/shared/protocol.c +++ b/src/shared/protocol.c @@ -40,7 +40,7 @@ static bool protocol_reserve_memory(ProtocolSession* session, size_t charge) { } } -static void protocol_release_memory_for_session(ProtocolSession* session, size_t charge) { +void protocol_release_memory_for_session(ProtocolSession* session, size_t charge) { unsigned long long allocated = atomic_load(&session->total_allocated_bytes); while (true) { unsigned long long remaining = (unsigned long long)charge >= allocated ? 0 : allocated - charge; @@ -573,6 +573,7 @@ Data* protocol_receive_data_limited(ProtocolSession* session, unsigned long long return NULL; } result->protocol_charge = allocation_size; + result->owner = session; return result; } diff --git a/tests/test_protocol.c b/tests/test_protocol.c index 3a63125..d84f293 100644 --- a/tests/test_protocol.c +++ b/tests/test_protocol.c @@ -412,6 +412,54 @@ static void test_protocol_accounting_release_does_not_underflow() { protocol_session_unbind(); } +/* A Data acquired on session A must return its connection-memory charge to A + even when a different session B is bound at destroy time: releasing against + the thread-local bound session would leak A's budget and drain B's. */ +static void test_receive_data_charge_follows_owning_session() { + int pipe_a[2]; + int pipe_b[2]; + EXPECT_EQ_INT(pipe(pipe_a), 0); + EXPECT_EQ_INT(pipe(pipe_b), 0); + + ProtocolSession session_a; + ProtocolSession session_b; + protocol_session_init(&session_a, pipe_a[0], pipe_a[1]); + protocol_session_init(&session_b, pipe_b[0], pipe_b[1]); + protocol_session_set_max_alloc(&session_a, 64); + protocol_session_set_max_alloc(&session_b, 64); + + unsigned long long size = 8; + EXPECT_EQ_INT((int)write(pipe_a[1], &size, sizeof(size)), (int)sizeof(size)); + EXPECT_EQ_INT((int)write(pipe_a[1], "12345678", 8), 8); + EXPECT_EQ_INT((int)write(pipe_b[1], &size, sizeof(size)), (int)sizeof(size)); + EXPECT_EQ_INT((int)write(pipe_b[1], "abcdefgh", 8), 8); + + Data* data_a = protocol_receive_data_limited(&session_a, 8); + Data* data_b = protocol_receive_data_limited(&session_b, 8); + EXPECT_NOT_NULL(data_a); + EXPECT_NOT_NULL(data_b); + EXPECT_TRUE(data_a->owner == &session_a); + EXPECT_TRUE(data_b->owner == &session_b); + EXPECT_EQ_INT((int)atomic_load(&session_a.total_allocated_bytes), 8); + EXPECT_EQ_INT((int)atomic_load(&session_b.total_allocated_bytes), 8); + + /* Destroy A's Data while the unrelated session B is the bound session. */ + protocol_session_bind(&session_b); + data_destroy(data_a); + protocol_session_unbind(); + + EXPECT_EQ_INT((int)atomic_load(&session_a.total_allocated_bytes), 0); + EXPECT_EQ_INT((int)atomic_load(&session_b.total_allocated_bytes), 8); + + data_destroy(data_b); + EXPECT_EQ_INT((int)atomic_load(&session_b.total_allocated_bytes), 0); + + close(pipe_a[0]); + close(pipe_a[1]); + close(pipe_b[0]); + close(pipe_b[1]); +} + static void test_protocol_session_io_timeout() { /* Default is the built-in 60 s window; the setter stores exactly what it is * given (<= 0 means "fall back to the default") so callers can propagate @@ -574,4 +622,5 @@ void test_protocol() { test_protocol_accounting_reservation_is_atomic(); test_protocol_string_accounting_is_transient(); test_protocol_accounting_release_does_not_underflow(); + test_receive_data_charge_follows_owning_session(); } From 3260a39ab454dabebe0ecee4318f44ff5f40c9da Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:06:04 +0200 Subject: [PATCH 036/155] refactor(shared): single owner for authorized_root state --- src/server/server.c | 34 +++++++-------------------------- src/shared/file.c | 46 +++++++++++++++++---------------------------- src/shared/file.h | 3 --- src/shared/utils.c | 24 ++++++++++++++++------- src/shared/utils.h | 7 +++++++ tests/test_file.c | 16 ++++++++-------- 6 files changed, 56 insertions(+), 74 deletions(-) diff --git a/src/server/server.c b/src/server/server.c index 494a840..1eb1e8d 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -29,8 +29,6 @@ #include #include -static char* authorized_root; -static int authorized_root_fd = -1; static bool allow_delete; static bool trust_sender; static bool allow_unauthenticated; @@ -187,13 +185,10 @@ static bool tls_client_identity_allowed(SSL* ssl) { } static void release_authorization(void) { - file_set_authorized_root(-1, NULL); - utils_set_authorized_root_fd(-1); - if (authorized_root_fd >= 0) - close(authorized_root_fd); - authorized_root_fd = -1; - free(authorized_root); - authorized_root = NULL; + int root_fd = utils_get_authorized_root_fd(); + utils_set_authorized_root(-1, NULL); + if (root_fd >= 0) + close(root_fd); } static bool path_is_within(const char* root, const char* path) { @@ -219,13 +214,11 @@ static bool ensure_receive_root(const Config* config) { static bool configure_authorization(const char* root) { char resolved[PATH_MAX]; if (!root) { - file_set_authorized_root(-1, NULL); utils_set_authorized_root(-1, NULL); return false; } int root_fd = open(root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); if (root_fd < 0) { - file_set_authorized_root(-1, NULL); utils_set_authorized_root(-1, NULL); return false; } @@ -234,26 +227,12 @@ static bool configure_authorization(const char* root) { if (fd_path_length < 0 || (size_t)fd_path_length >= sizeof(fd_path) || !realpath(fd_path, resolved)) { close(root_fd); - file_set_authorized_root(-1, NULL); utils_set_authorized_root(-1, NULL); return false; } - authorized_root = str_dup(resolved); - if (!authorized_root) { + if (!utils_set_authorized_root(root_fd, resolved)) { + utils_set_authorized_root(-1, NULL); close(root_fd); - file_set_authorized_root(-1, NULL); - utils_set_authorized_root(-1, NULL); - return false; - } - authorized_root_fd = root_fd; - if (!file_set_authorized_root(authorized_root_fd, authorized_root) || - !utils_set_authorized_root(authorized_root_fd, authorized_root)) { - file_set_authorized_root(-1, NULL); - utils_set_authorized_root(-1, NULL); - close(authorized_root_fd); - authorized_root_fd = -1; - free(authorized_root); - authorized_root = NULL; return false; } return true; @@ -615,6 +594,7 @@ void handler(int file_descriptor) { * in effect. A client's --timeout tightens only that client's own protocol * I/O and the server's socket read/write timeout is the transport default. */ protocol_session_set_io_timeout(&session, config->timeout); + const char* authorized_root = utils_get_authorized_root_path(); if (!authorized_root) { log_message(LOG_LEVEL_ERROR, "No server-side destination root configured"); goto done; diff --git a/src/shared/file.c b/src/shared/file.c index b85e988..9d5aa23 100644 --- a/src/shared/file.c +++ b/src/shared/file.c @@ -284,23 +284,6 @@ size_t file_content_to_buffer(File* file) { /* ---- Secure filesystem primitives ---- */ -static int authorized_root_fd = -1; -static char* authorized_root_path; - -bool file_set_authorized_root(int fd, const char* canonical_path) { - char* path_copy = canonical_path ? str_dup(canonical_path) : NULL; - if (canonical_path && !path_copy) { - authorized_root_fd = -1; - free(authorized_root_path); - authorized_root_path = NULL; - return false; - } - authorized_root_fd = fd; - free(authorized_root_path); - authorized_root_path = path_copy; - return true; -} - bool file_path_exists_secure(const char* path) { if (!path) return false; @@ -483,7 +466,10 @@ static int open_dir_beneath_root(const char* resolved, const char* root) { rel++; if (*rel == '\0') return -1; - int fd = dup(authorized_root_fd); + int root_fd = utils_get_authorized_root_fd(); + if (root_fd < 0) + return -1; + int fd = dup(root_fd); if (fd < 0) return -1; char* copy = str_dup(rel); @@ -525,20 +511,21 @@ int file_open_secure_parent(const char* path, char** leaf_out, bool create_dirs) return -1; } int fd; - if (authorized_root_fd >= 0) { - if (!authorized_root_path || path[0] != '/' || - !path_is_within_root(authorized_root_path, path)) { + int root_fd = utils_get_authorized_root_fd(); + const char* root_path = utils_get_authorized_root_path(); + if (root_fd >= 0) { + if (!root_path || path[0] != '/' || !path_is_within_root(root_path, path)) { free(copy); free(leaf); return -1; } - fd = dup(authorized_root_fd); + fd = dup(root_fd); if (fd < 0) { free(copy); free(leaf); return -1; } - size_t root_len = strlen(authorized_root_path); + size_t root_len = strlen(root_path); char* relative = str_dup(path + root_len); if (!relative) { free(copy); @@ -602,15 +589,14 @@ int file_open_secure_parent(const char* path, char** leaf_out, bool create_dirs) O_NOFOLLOW walk. Only honoured when the symlink resolves to a directory that stays beneath the authorized root, so a malicious link can never redirect the write outside it. */ - if (next < 0 && file_keep_dirlinks && authorized_root_path != NULL && + if (next < 0 && file_keep_dirlinks && root_path != NULL && (errno == ELOOP || errno == ENOTDIR || errno == EACCES)) { struct stat lst; if (fstatat(fd, component, &lst, AT_SYMLINK_NOFOLLOW) == 0 && S_ISLNK(lst.st_mode)) { char candidate[PATH_MAX]; char root[PATH_MAX]; - if (realpath(authorized_root_path, root) && - snprintf(candidate, sizeof(candidate), "%s%s/%s", root, rel_buf, component) < - (int)sizeof(candidate)) { + if (realpath(root_path, root) && snprintf(candidate, sizeof(candidate), "%s%s/%s", root, + rel_buf, component) < (int)sizeof(candidate)) { char resolved[PATH_MAX]; if (realpath(candidate, resolved) && strcmp(resolved, root) != 0 && strncmp(root, resolved, strlen(root)) == 0 && @@ -685,8 +671,9 @@ bool file_ensure_directory_secure(const char* path) { return false; /* The authorized root is already an open directory, and the filesystem root is always present: there is no final component left to create for them. */ + const char* root_path = utils_get_authorized_root_path(); bool root_is_open = - authorized_root_fd >= 0 && authorized_root_path && strcmp(norm, authorized_root_path) == 0; + utils_get_authorized_root_fd() >= 0 && root_path && strcmp(norm, root_path) == 0; if (root_is_open || strcmp(norm, "/") == 0) { free(norm); return true; @@ -733,8 +720,9 @@ bool file_directory_exists_secure(const char* path) { char* norm = normalize_directory_path(path); if (!norm) return false; + const char* root_path = utils_get_authorized_root_path(); bool root_is_open = - authorized_root_fd >= 0 && authorized_root_path && strcmp(norm, authorized_root_path) == 0; + utils_get_authorized_root_fd() >= 0 && root_path && strcmp(norm, root_path) == 0; if (root_is_open || strcmp(norm, "/") == 0) { free(norm); return true; diff --git a/src/shared/file.h b/src/shared/file.h index 570d703..3ccc2aa 100644 --- a/src/shared/file.h +++ b/src/shared/file.h @@ -62,9 +62,6 @@ void file_set_keep_dirlinks(bool enable); void file_set_trust_sender(bool enable); bool file_get_trust_sender(void); -/* A configured fd without a canonical identity deliberately rejects paths. */ -bool file_set_authorized_root(int fd, const char* canonical_path); - /* Secure path/filesystem primitives (symlink-safe, O_NOFOLLOW, root-confined). */ bool file_path_exists_secure(const char* path); bool file_stat_secure(const char* path, struct stat* st); diff --git a/src/shared/utils.c b/src/shared/utils.c index d790172..6ab725a 100644 --- a/src/shared/utils.c +++ b/src/shared/utils.c @@ -36,6 +36,14 @@ void utils_set_authorized_root_fd(int fd) { (void)utils_set_authorized_root(fd, NULL); } +int utils_get_authorized_root_fd(void) { + return authorized_root_fd; +} + +const char* utils_get_authorized_root_path(void) { + return authorized_root_path; +} + bool path_is_within_root(const char* root, const char* path) { size_t root_len = strlen(root); return strncmp(root, path, root_len) == 0 && (path[root_len] == '\0' || path[root_len] == '/'); @@ -50,15 +58,16 @@ bool path_is_within_root(const char* root, const char* path) { * in the extra receiver policies they apply, so they are intentionally kept * separate. Both rely on the shared lexical path_is_within_root check. */ static int open_authorized_destination(const char* dest_root) { - if (authorized_root_fd < 0 || !authorized_root_path || !dest_root || - !path_is_within_root(authorized_root_path, dest_root)) + int root_fd = utils_get_authorized_root_fd(); + const char* root_path = utils_get_authorized_root_path(); + if (root_fd < 0 || !root_path || !dest_root || !path_is_within_root(root_path, dest_root)) return -1; - int dirfd = dup(authorized_root_fd); + int dirfd = dup(root_fd); if (dirfd < 0) return -1; - const char* relative_path = dest_root + strlen(authorized_root_path); + const char* relative_path = dest_root + strlen(root_path); while (*relative_path == '/') relative_path++; char* relative = str_dup(*relative_path ? relative_path : "."); @@ -689,11 +698,12 @@ DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* m if (!build_keep_index(manifest, &keep)) return DELETE_WALK_ERROR; int rootfd; - if (authorized_root_fd >= 0) { - if (authorized_root_path) + int root_fd = utils_get_authorized_root_fd(); + if (root_fd >= 0) { + if (utils_get_authorized_root_path()) rootfd = open_authorized_destination(dest_root); else if (dest_root == NULL) - rootfd = dup(authorized_root_fd); + rootfd = dup(root_fd); else rootfd = -1; } else { diff --git a/src/shared/utils.h b/src/shared/utils.h index f4a5bd6..7ac9059 100644 --- a/src/shared/utils.h +++ b/src/shared/utils.h @@ -127,6 +127,13 @@ bool utils_set_authorized_root(int fd, const char* canonical_path); /* The fd-only compatibility form is fail-closed for path-based operations; * callers should use utils_set_authorized_root with the canonical identity. */ void utils_set_authorized_root_fd(int fd); +/* Read accessors for the process-wide authorized root, so every secure-walk + * site consumes the single shared state instead of keeping its own copy. The + * fd is caller-owned (see the setters): it is returned verbatim, never dup'd, + * and the caller that opened it is responsible for closing it. With no root + * configured the fd accessor returns -1 and the path accessor returns NULL. */ +int utils_get_authorized_root_fd(void); +const char* utils_get_authorized_root_path(void); /* True when `path` is `root` itself or lies directly beneath it: a lexical * prefix test requiring the byte after `root` to be '\0' or '/'. Both `root` * and `path` must be absolute canonical paths free of "."/".." components (the diff --git a/tests/test_file.c b/tests/test_file.c index 4561f1a..73c3431 100644 --- a/tests/test_file.c +++ b/tests/test_file.c @@ -1180,7 +1180,7 @@ static void test_trust_sender_authorized_root_confinement() { rmdir(sibling); return; } - EXPECT_TRUE(file_set_authorized_root(root_fd, root_abs)); + EXPECT_TRUE(utils_set_authorized_root(root_fd, root_abs)); file_set_trust_sender(true); struct stat st; @@ -1204,7 +1204,7 @@ static void test_trust_sender_authorized_root_confinement() { free(outside_link); free(inside_link); - file_set_authorized_root(-1, NULL); + utils_set_authorized_root(-1, NULL); close(root_fd); unlink("test_trust_sender_outside_link"); rmdir(sibling); @@ -1220,7 +1220,7 @@ void test_trust_sender() { test_trust_sender_confines_hostile_paths(); test_trust_sender_authorized_root_confinement(); file_set_trust_sender(false); - file_set_authorized_root(-1, NULL); + utils_set_authorized_root(-1, NULL); } /* --sparse/-S hole preservation: a buffer with a long zero run written via @@ -1341,7 +1341,7 @@ static void test_file_write_to_disk_partial_retention() { static void test_dir_time_list() { const char* root = "test_dir_time_root"; const char* sub = "test_dir_time_root/sub"; - file_set_authorized_root(-1, NULL); + utils_set_authorized_root(-1, NULL); rmdir(sub); rmdir(root); EXPECT_EQ_INT(mkdir(root, 0755), 0); @@ -1485,7 +1485,7 @@ static void test_keep_dirlinks_secure_open_impl() { rmdir(outside); return; } - EXPECT_TRUE(file_set_authorized_root(root_fd, root_abs)); + EXPECT_TRUE(utils_set_authorized_root(root_fd, root_abs)); file_set_keep_dirlinks(true); struct stat real_st; @@ -1538,7 +1538,7 @@ static void test_keep_dirlinks_secure_open_impl() { free(leaf); file_set_keep_dirlinks(false); - file_set_authorized_root(-1, NULL); + utils_set_authorized_root(-1, NULL); close(root_fd); unlink(link); unlink(abslink); @@ -1552,10 +1552,10 @@ static void test_keep_dirlinks_secure_open_impl() { * cleared even when an EXPECT inside the body returns early (a failing EXPECT * returns from its own function, so the body's trailing resets may be skipped). */ static void test_keep_dirlinks_secure_open() { - file_set_authorized_root(-1, NULL); + utils_set_authorized_root(-1, NULL); file_set_keep_dirlinks(false); test_keep_dirlinks_secure_open_impl(); - file_set_authorized_root(-1, NULL); + utils_set_authorized_root(-1, NULL); file_set_keep_dirlinks(false); } From 5aca91ab22eaa4124f7705eb6b4a624d070e4014 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:14:37 +0200 Subject: [PATCH 037/155] fix(benchmark): use real FastSync flags and Release builds The benchmark tool used stale rsync-style spellings that map to different FastSync options, so it never enabled the features it claimed to measure: -c -> --checksum (not compression) -m -> --prune-empty-dirs (not multithreading) -s -> --secluded-args, a no-op (not chunk serialization) -f -> --filter, needs an argument (not sendfile) Replace them with the real flags (-z, -j, --chunk-serialization, --sendfile), force CMAKE_BUILD_TYPE=Release, route informational output to stderr so --output json emits valid JSON, surface client/rsync failures instead of silently dropping them, and widen the results table for the longer config names. Update the benchmark skill to match (correct flags, server invocation, and replace the nonexistent test.py --full with benchmark/bench.py). --- .opencode/skills/benchmark/SKILL.md | 55 ++++++++++++++++++----------- benchmark/bench.py | 50 +++++++++++++++----------- 2 files changed, 63 insertions(+), 42 deletions(-) diff --git a/.opencode/skills/benchmark/SKILL.md b/.opencode/skills/benchmark/SKILL.md index aed3533..fb91be0 100644 --- a/.opencode/skills/benchmark/SKILL.md +++ b/.opencode/skills/benchmark/SKILL.md @@ -38,50 +38,62 @@ dd if=/dev/urandom of=/tmp/fastsync_bench/src/large.bin bs=1M count=10 2>/dev/nu Test each configuration 3 times, record median: ```bash +# Real FastSync flags: -z=compression, -j=multithreading, +# --chunk-serialization, --sendfile (long form only). The old rsync-style +# spellings -c/-m/-s/-f are NOT the same options (-c=--checksum, +# -m=--prune-empty-dirs, -s=--secluded-args, -f=--filter) and must not be used. CONFIGS=( "Standard|" - "Compression|-c" - "Multithreading|-m" - "MT+Compression|-m -c" - "Chunk Serialization|-s" - "MT+Compression+Chunk|-m -c -s" - "Sendfile|-f" + "Compression|-z" + "Multithreading|-j" + "MT+Compression|-j -z" + "Chunk Serialization|-j -z --chunk-serialization" + "Sendfile|--sendfile" ) +PORT=18080 for config in "${CONFIGS[@]}"; do IFS='|' read -r name flags <<< "$config" echo "=== $name ===" for run in 1 2 3; do rm -rf /tmp/fastsync_bench/dst mkdir -p /tmp/fastsync_bench/dst - - ./build/server & + + ./build/server -p "$PORT" --allow-unauthenticated & SERVER_PID=$! sleep 0.5 - + START=$(date +%s%N) ./build/client --source-dir /tmp/fastsync_bench/src \ --dest-dir /tmp/fastsync_bench/dst \ + --server-port "$PORT" \ --save-to-disk $flags END=$(date +%s%N) - + ELAPSED=$(( (END - START) / 1000000 )) echo " Run $run: ${ELAPSED}ms" - + kill $SERVER_PID 2>/dev/null wait $SERVER_PID 2>/dev/null done done ``` -### Step 4: Full Integration Benchmark (Optional) +### Step 4: Full Benchmark Tool (Preferred) + +The maintained benchmark tool is `benchmark/bench.py`. It handles building, +data generation, network shaping (LAN/WAN profiles or custom `--delay`/`--jitter`/ +`--throughput`/`--loss`), rsync comparison, and JSON/table reporting: -For comprehensive benchmarking with network shaping: ```bash -python3 test.py --full +python3 benchmark/bench.py --help +python3 benchmark/bench.py --runs 5 --profiles unlimited +python3 benchmark/bench.py --size-mb 100 --random-ratio 0.5 --output json +python3 benchmark/bench.py --delay 50ms --jitter 10ms --throughput 100mbit ``` -This tests LAN/WAN profiles, SSH, TLS, and compares against rsync. +Network shaping needs root (`tc`/`netem` on `lo`). SSH and TLS coverage lives in +the pytest integration suite, not the benchmark tool. ### Step 5: Report Results @@ -92,13 +104,14 @@ Platform: Configuration | Run 1 | Run 2 | Run 3 | Median -----------------------|---------|---------|---------|-------- -Standard | 0.12s | 0.11s | 0.12s | 0.12s -Compression (-c) | 0.09s | 0.08s | 0.09s | 0.09s -Multithreading (-m) | 0.07s | 0.07s | 0.08s | 0.07s -MT+Compression (-m -c) | 0.05s | 0.05s | 0.06s | 0.05s -Sendfile (-f) | 0.04s | 0.04s | 0.04s | 0.04s +Standard | 0.12s | 0.11s | 0.12s | 0.12s +Compression (-z) | 0.09s | 0.08s | 0.09s | 0.09s +Multithreading (-j) | 0.07s | 0.07s | 0.08s | 0.07s +MT+Compression (-j -z) | 0.05s | 0.05s | 0.06s | 0.05s +Chunk Serialization (--chunk-serialization) | 0.05s | 0.04s | 0.05s | 0.05s +Sendfile (--sendfile) | 0.04s | 0.04s | 0.04s | 0.04s -Best configuration: MT+Compression (-m -c) +Best configuration: Sendfile (--sendfile) Throughput: MB/s ``` diff --git a/benchmark/bench.py b/benchmark/bench.py index abb1db8..d4b59b1 100644 --- a/benchmark/bench.py +++ b/benchmark/bench.py @@ -26,7 +26,7 @@ import time PROJECT_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..")) BUILD_DIR = os.path.join(PROJECT_ROOT, "build") -SERVER_CMD = [os.path.join(BUILD_DIR, "server")] +SERVER_CMD = [os.path.join(BUILD_DIR, "server"), "--allow-unauthenticated"] CLIENT_CMD = [os.path.join(BUILD_DIR, "client")] BENCH_DIR = os.path.join(PROJECT_ROOT, "bench_data") @@ -43,11 +43,12 @@ NETWORK_PROFILES = { } FASTSYNC_CONFIGS = [ - {"name": "fastsync", "flags": [], "tool": "fastsync"}, - {"name": "fastsync -c", "flags": ["-c"], "tool": "fastsync"}, - {"name": "fastsync -m", "flags": ["-m"], "tool": "fastsync"}, - {"name": "fastsync -m -c", "flags": ["-m", "-c"], "tool": "fastsync"}, - {"name": "fastsync -m -c -s", "flags": ["-m", "-c", "-s"], "tool": "fastsync"}, + {"name": "fastsync", "flags": [], "tool": "fastsync"}, + {"name": "fastsync -z", "flags": ["-z"], "tool": "fastsync"}, + {"name": "fastsync -j", "flags": ["-j"], "tool": "fastsync"}, + {"name": "fastsync -j -z", "flags": ["-j", "-z"], "tool": "fastsync"}, + {"name": "fastsync -j -z --chunk-serialization", "flags": ["-j", "-z", "--chunk-serialization"], "tool": "fastsync"}, + {"name": "fastsync --sendfile", "flags": ["--sendfile"], "tool": "fastsync"}, ] RSYNC_CONFIGS = [ @@ -252,8 +253,10 @@ def run_fastsync(source_dir, dest_dir, flags, port): duration = time.monotonic() - start if result.returncode == 0: return duration + sys.stderr.write(f" fastsync failed (exit {result.returncode}): " + f"{result.stderr.strip()[:500]}\n") except subprocess.TimeoutExpired: - pass + sys.stderr.write(" fastsync timed out after 120s\n") return None @@ -269,8 +272,10 @@ def run_rsync(source_dir, dest_dir, flags, rsync_daemon=None): duration = time.monotonic() - start if result.returncode == 0: return duration + sys.stderr.write(f" rsync failed (exit {result.returncode}): " + f"{result.stderr.strip()[:500]}\n") except subprocess.TimeoutExpired: - pass + sys.stderr.write(" rsync timed out after 120s\n") return None @@ -367,15 +372,15 @@ def print_table(results, total_bytes, random_ratio): if fs_entries: print(f"\n FastSync:") - print(f" {'Config':<25} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} {'stdev':>8} {'runs':>5}") - print(f" {'-' * 25} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 5}") + print(f" {'Config':<38} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} {'stdev':>8} {'runs':>5}") + print(f" {'-' * 38} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 5}") for e in sorted(fs_entries, key=lambda x: x.get("p50", 999)): _print_entry(e) if rsync_entries: print(f"\n rsync:") - print(f" {'Config':<25} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} {'stdev':>8} {'runs':>5}") - print(f" {'-' * 25} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 5}") + print(f" {'Config':<38} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} {'stdev':>8} {'runs':>5}") + print(f" {'-' * 38} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 5}") for e in sorted(rsync_entries, key=lambda x: x.get("p50", 999)): _print_entry(e) @@ -392,10 +397,10 @@ def print_table(results, total_bytes, random_ratio): def _print_entry(e): if "p50" in e: - print(f" {e['config']:<25} {e['p50']:>7.4f}s {e['p95']:>7.4f}s " + print(f" {e['config']:<38} {e['p50']:>7.4f}s {e['p95']:>7.4f}s " f"{e['min']:>7.4f}s {e['max']:>7.4f}s {e['stdev']:>7.4f} {e['runs']:>5}") else: - print(f" {e['config']:<25} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} {e['runs']:>5}") + print(f" {e['config']:<38} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} {e['runs']:>5}") def main(): @@ -448,12 +453,14 @@ Examples: help="Don't clean up test data") args = parser.parse_args() - # Build - print("Building...") - if os.system(f"cmake -B {BUILD_DIR} -S {PROJECT_ROOT} > /dev/null 2>&1") != 0: - print("CMake configure failed"); sys.exit(1) + # Build (Release: benchmarking a debug build is meaningless) + print("Building (Release)...", file=sys.stderr) + configure = (f"cmake -B {BUILD_DIR} -S {PROJECT_ROOT} " + f"-DCMAKE_BUILD_TYPE=Release > /dev/null 2>&1") + if os.system(configure) != 0: + print("CMake configure failed", file=sys.stderr); sys.exit(1) if os.system(f"cmake --build {BUILD_DIR} -j$(nproc) > /dev/null 2>&1") != 0: - print("Build failed"); sys.exit(1) + print("Build failed", file=sys.stderr); sys.exit(1) # Determine active profile for display has_custom_net = args.delay or args.jitter or args.throughput or args.loss @@ -480,7 +487,8 @@ Examples: compressible_pct = (1 - args.random_ratio) * 100 random_pct = args.random_ratio * 100 print(f"Generated {total_bytes / (1024*1024):.1f} MB " - f"({random_pct:.0f}% random, {compressible_pct:.0f}% compressible)") + f"({random_pct:.0f}% random, {compressible_pct:.0f}% compressible)", + file=sys.stderr) # Build config list if args.configs: @@ -496,7 +504,7 @@ Examples: total_runs = len(configs) * args.runs * len(profiles_to_run) progress = Progress(total_runs, "Benchmarking") if args.progress else None if progress: - print(f"Running {total_runs} transfers...") + print(f"Running {total_runs} transfers...", file=sys.stderr) all_results = [] try: From 5334397b817babe93c78c0a2a3bb011f4d4b60fb Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:23:58 +0200 Subject: [PATCH 038/155] feat(daemon): add shared cross-process connection registry The daemon forks one child per accepted connection, so per-module and per-source accounting must live in state shared across the children. Add a fixed-size registry carved from an anonymous shared mapping (mmap(MAP_SHARED|MAP_ANONYMOUS)) created before the accept loop: a slot lifecycle (FREE/CLAIMED/REGISTERED) with parent claim/reclaim and a lock-free, open-addressed per-source table for the per-host occupancy and the shared auth-failure counter. C11 atomics only; no pthread locks across fork. Unit tests cover slot exhaustion, the module/host caps, pid reclaim and fork-shared visibility. --- CMakeLists.txt | 2 + src/shared/daemon_limits.c | 356 +++++++++++++++++++++++++++++++++++++ src/shared/daemon_limits.h | 102 +++++++++++ tests/runner.c | 2 + tests/test_daemon_limits.c | 194 ++++++++++++++++++++ tests/test_daemon_limits.h | 6 + 6 files changed, 662 insertions(+) create mode 100644 src/shared/daemon_limits.c create mode 100644 src/shared/daemon_limits.h create mode 100644 tests/test_daemon_limits.c create mode 100644 tests/test_daemon_limits.h diff --git a/CMakeLists.txt b/CMakeLists.txt index e61b0fa..479cbca 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -86,6 +86,7 @@ set(SHARED_SRCS src/shared/config.c src/shared/credentials.c src/shared/daemon_conf.c + src/shared/daemon_limits.c src/shared/data.c src/shared/delay_updates.c src/shared/delta.c @@ -205,6 +206,7 @@ set(TEST_SRCS tests/test_config.c tests/test_credentials.c tests/test_daemon_conf.c + tests/test_daemon_limits.c tests/test_data.c tests/test_delay_updates.c tests/test_delta.c diff --git a/src/shared/daemon_limits.c b/src/shared/daemon_limits.c new file mode 100644 index 0000000..2ba7e4e --- /dev/null +++ b/src/shared/daemon_limits.c @@ -0,0 +1,356 @@ +#include "daemon_limits.h" +#include +#include +#include +#include +#include +#include +#include +#include + +/* Slot lifecycle states (stored in slot_state). */ +enum { + SLOT_FREE = 0, + SLOT_CLAIMED = 1, + SLOT_REGISTERED = 2, +}; + +/* The registry header lives at the base of the shared mapping; the pointer + * fields point at the arrays carved out of the same mapping. Absolute pointers + * remain valid in a forked child because fork() clones the address space and + * mapping, so parent and child observe the same virtual addresses. */ +struct DaemonLimitRegistry { + int max_slots; + int module_count; + int host_slots; /* power of two; 1 when no per-source tracking is needed */ + int per_host_cap; + int lockout_threshold; + int lockout_duration_sec; + size_t map_size; + _Atomic int* slot_state; + _Atomic int* slot_pid; + _Atomic int* slot_module; + _Atomic int* slot_host; /* per-source table bucket, or -1 */ + _Atomic int* module_active; + _Atomic uint64_t* host_key; /* 0 == empty bucket */ + _Atomic int* host_active; + _Atomic int* host_fail; + _Atomic long long* host_until; /* epoch seconds the lockout expires */ +}; + +static size_t round_up(size_t n, size_t align) { + return (n + align - 1) & ~(align - 1); +} + +static size_t next_pow2(size_t n) { + size_t p = 1; + while (p < n) + p <<= 1; + return p; +} + +/* Parse a numeric IPv4/IPv6 peer string into family + raw bytes. */ +static bool parse_peer_ip(const char* peer_ip, int* family, unsigned char* bytes) { + if (!peer_ip || *peer_ip == '\0') + return false; + struct in_addr v4; + if (inet_pton(AF_INET, peer_ip, &v4) == 1) { + memcpy(bytes, &v4, sizeof(v4)); + *family = AF_INET; + return true; + } + struct in6_addr v6; + if (inet_pton(AF_INET6, peer_ip, &v6) == 1) { + memcpy(bytes, &v6, sizeof(v6)); + *family = AF_INET6; + return true; + } + return false; +} + +uint64_t daemon_limits_host_hash(const char* peer_ip, bool* ok) { + if (ok) + *ok = false; + unsigned char bytes[16]; + int family = AF_UNSPEC; + if (!parse_peer_ip(peer_ip, &family, bytes)) + return 0; + uint64_t hash = 14695981039346656037ULL ^ (uint64_t)(uint32_t)family; + size_t length = family == AF_INET ? 4 : 16; + for (size_t i = 0; i < length; i++) { + hash ^= bytes[i]; + hash *= 1099511628211ULL; + } + if (hash == 0) + hash = 0x9e3779b97f4a7c15ULL; + if (ok) + *ok = true; + return hash; +} + +/* Find the bucket holding `peer_ip`, or -1 when it has no entry. */ +static int host_lookup(DaemonLimitRegistry* registry, const char* peer_ip) { + bool ok = false; + uint64_t key = daemon_limits_host_hash(peer_ip, &ok); + if (!ok) + return -1; + size_t mask = (size_t)registry->host_slots - 1; + size_t start = (size_t)(key & mask); + for (size_t i = 0; i < (size_t)registry->host_slots; i++) { + size_t idx = (start + i) & mask; + uint64_t current = atomic_load_explicit(®istry->host_key[idx], memory_order_acquire); + if (current == key) + return (int)idx; + if (current == 0) + return -1; /* no tombstones: an empty bucket ends the probe chain */ + } + return -1; +} + +/* Find or insert the bucket for `peer_ip`. Insertion is a lock-free CAS so two + * forked children racing on the same source converge on one bucket. Returns -1 + * when the table is full or the address is unparseable (callers fail open: the + * global/module caps and ACLs still apply). */ +static int host_intern(DaemonLimitRegistry* registry, const char* peer_ip) { + bool ok = false; + uint64_t key = daemon_limits_host_hash(peer_ip, &ok); + if (!ok) + return -1; + size_t mask = (size_t)registry->host_slots - 1; + size_t start = (size_t)(key & mask); + for (size_t i = 0; i < (size_t)registry->host_slots; i++) { + size_t idx = (start + i) & mask; + uint64_t current = atomic_load_explicit(®istry->host_key[idx], memory_order_acquire); + if (current == key) + return (int)idx; + if (current == 0) { + uint64_t expected = 0; + if (atomic_compare_exchange_strong_explicit(®istry->host_key[idx], &expected, key, + memory_order_acq_rel, memory_order_acquire)) + return (int)idx; + if (atomic_load_explicit(®istry->host_key[idx], memory_order_acquire) == key) + return (int)idx; + } + } + return -1; +} + +DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int per_host_cap, + int lockout_threshold, int lockout_duration_sec) { + if (max_slots < DAEMON_LIMITS_MIN_SLOTS) + max_slots = DAEMON_LIMITS_MIN_SLOTS; + if (max_slots > DAEMON_LIMITS_MAX_SLOTS) + max_slots = DAEMON_LIMITS_MAX_SLOTS; + if (module_count < 1) + module_count = 1; + if (per_host_cap < 0) + per_host_cap = 0; + if (lockout_threshold < 0) + lockout_threshold = 0; + if (lockout_duration_sec < 0) + lockout_duration_sec = 0; + + bool need_hosts = per_host_cap > 0 || (lockout_threshold > 0 && lockout_duration_sec > 0); + int host_slots = 1; + if (need_hosts) { + size_t want = (size_t)max_slots * 4; + if (want < 64) + want = 64; + if (want > DAEMON_LIMITS_MAX_HOST_SLOTS) + want = DAEMON_LIMITS_MAX_HOST_SLOTS; + host_slots = (int)next_pow2(want); + } + + size_t header = round_up(sizeof(DaemonLimitRegistry), 16); + size_t slot_bytes = + round_up((size_t)max_slots * sizeof(_Atomic int), 16) * 4; /* state,pid,module,host */ + size_t module_bytes = round_up((size_t)module_count * sizeof(_Atomic int), 16); + size_t host_key_bytes = round_up((size_t)host_slots * sizeof(_Atomic uint64_t), 16); + size_t host_int_bytes = round_up((size_t)host_slots * sizeof(_Atomic int), 16) * 2; + size_t host_until_bytes = round_up((size_t)host_slots * sizeof(_Atomic long long), 16); + size_t total = + header + slot_bytes + module_bytes + host_key_bytes + host_int_bytes + host_until_bytes + 16; + + void* map = mmap(NULL, total, PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, -1, 0); + if (map == MAP_FAILED) + return NULL; + memset(map, 0, total); + + DaemonLimitRegistry* registry = (DaemonLimitRegistry*)map; + registry->max_slots = max_slots; + registry->module_count = module_count; + registry->host_slots = host_slots; + registry->per_host_cap = per_host_cap; + registry->lockout_threshold = lockout_threshold; + registry->lockout_duration_sec = lockout_duration_sec; + registry->map_size = total; + + unsigned char* cursor = (unsigned char*)map + header; + registry->slot_state = (atomic_int*)cursor; + cursor += (size_t)max_slots * sizeof(_Atomic int); + registry->slot_pid = (atomic_int*)cursor; + cursor += (size_t)max_slots * sizeof(_Atomic int); + registry->slot_module = (atomic_int*)cursor; + cursor += (size_t)max_slots * sizeof(_Atomic int); + registry->slot_host = (atomic_int*)cursor; + cursor += (size_t)max_slots * sizeof(_Atomic int); + registry->module_active = (atomic_int*)cursor; + cursor += (size_t)module_count * sizeof(_Atomic int); + cursor = (unsigned char*)round_up((size_t)(uintptr_t)cursor, 16); + registry->host_key = (_Atomic uint64_t*)cursor; + cursor += (size_t)host_slots * sizeof(_Atomic uint64_t); + registry->host_active = (atomic_int*)cursor; + cursor += (size_t)host_slots * sizeof(_Atomic int); + registry->host_fail = (atomic_int*)cursor; + cursor += (size_t)host_slots * sizeof(_Atomic int); + cursor = (unsigned char*)round_up((size_t)(uintptr_t)cursor, 16); + registry->host_until = (atomic_llong*)cursor; + + for (int i = 0; i < max_slots; i++) { + atomic_store(®istry->slot_module[i], -1); + atomic_store(®istry->slot_host[i], -1); + } + return registry; +} + +void daemon_limits_destroy(DaemonLimitRegistry* registry) { + if (!registry) + return; + munmap(registry, registry->map_size); +} + +int daemon_limits_claim_slot(DaemonLimitRegistry* registry) { + if (!registry) + return DAEMON_LIMITS_NO_SLOT; + for (int i = 0; i < registry->max_slots; i++) { + int expected = SLOT_FREE; + if (atomic_compare_exchange_strong(®istry->slot_state[i], &expected, SLOT_CLAIMED)) { + atomic_store(®istry->slot_pid[i], 0); + atomic_store(®istry->slot_module[i], -1); + atomic_store(®istry->slot_host[i], -1); + return i; + } + } + return DAEMON_LIMITS_NO_SLOT; +} + +void daemon_limits_set_slot_pid(DaemonLimitRegistry* registry, int slot, long pid) { + if (!registry || slot < 0 || slot >= registry->max_slots) + return; + atomic_store(®istry->slot_pid[slot], (int)pid); +} + +void daemon_limits_reclaim_slot(DaemonLimitRegistry* registry, int slot) { + if (!registry || slot < 0 || slot >= registry->max_slots) + return; + int previous = + atomic_exchange_explicit(®istry->slot_state[slot], SLOT_FREE, memory_order_acq_rel); + if (previous == SLOT_REGISTERED) { + int module = atomic_load(®istry->slot_module[slot]); + int host = atomic_load(®istry->slot_host[slot]); + if (module >= 0 && module < registry->module_count) { + int current = atomic_load(®istry->module_active[module]); + while (current > 0 && + !atomic_compare_exchange_weak(®istry->module_active[module], ¤t, current - 1)) + ; + } + if (host >= 0 && host < registry->host_slots) { + int current = atomic_load(®istry->host_active[host]); + while (current > 0 && + !atomic_compare_exchange_weak(®istry->host_active[host], ¤t, current - 1)) + ; + } + } + atomic_store(®istry->slot_pid[slot], 0); +} + +void daemon_limits_reclaim_pid(DaemonLimitRegistry* registry, long pid) { + if (!registry || pid <= 0) + return; + for (int i = 0; i < registry->max_slots; i++) { + if (atomic_load(®istry->slot_state[i]) == SLOT_FREE) + continue; + if (atomic_load(®istry->slot_pid[i]) == (int)pid) { + daemon_limits_reclaim_slot(registry, i); + return; + } + } +} + +DaemonLimitResult daemon_limits_register(DaemonLimitRegistry* registry, int slot, int module_index, + const char* peer_ip, int module_cap) { + if (!registry || slot < 0 || slot >= registry->max_slots) + return DAEMON_LIMIT_UNAVAILABLE; + if (module_index < 0 || module_index >= registry->module_count) + return DAEMON_LIMIT_UNAVAILABLE; + if (atomic_load_explicit(®istry->slot_state[slot], memory_order_acquire) != SLOT_CLAIMED) + return DAEMON_LIMIT_UNAVAILABLE; + + int host = -1; + if (registry->per_host_cap > 0 || registry->lockout_threshold > 0) + host = host_intern(registry, peer_ip); + + int module_count = atomic_fetch_add(®istry->module_active[module_index], 1) + 1; + if (module_cap > 0 && module_count > module_cap) { + atomic_fetch_sub(®istry->module_active[module_index], 1); + return DAEMON_LIMIT_MODULE_FULL; + } + if (host >= 0) { + int host_count = atomic_fetch_add(®istry->host_active[host], 1) + 1; + if (registry->per_host_cap > 0 && host_count > registry->per_host_cap) { + atomic_fetch_sub(®istry->host_active[host], 1); + atomic_fetch_sub(®istry->module_active[module_index], 1); + return DAEMON_LIMIT_HOST_FULL; + } + } + atomic_store(®istry->slot_module[slot], module_index); + atomic_store(®istry->slot_host[slot], host); + atomic_store_explicit(®istry->slot_state[slot], SLOT_REGISTERED, memory_order_release); + return DAEMON_LIMIT_OK; +} + +bool daemon_limits_auth_locked(DaemonLimitRegistry* registry, const char* peer_ip, + int* seconds_remaining) { + if (!registry || registry->lockout_threshold <= 0 || registry->lockout_duration_sec <= 0) + return false; + int bucket = host_lookup(registry, peer_ip); + if (bucket < 0) + return false; + long long until = atomic_load(®istry->host_until[bucket]); + long long now = (long long)time(NULL); + if (until > now) { + if (seconds_remaining) + *seconds_remaining = (int)(until - now); + return true; + } + if (until != 0) { + /* The previous lockout has expired: clear the stale counter so the source + * gets a fresh allowance. */ + atomic_store(®istry->host_fail[bucket], 0); + atomic_store(®istry->host_until[bucket], 0); + } + return false; +} + +void daemon_limits_auth_record_failure(DaemonLimitRegistry* registry, const char* peer_ip) { + if (!registry || registry->lockout_threshold <= 0 || registry->lockout_duration_sec <= 0) + return; + int bucket = host_intern(registry, peer_ip); + if (bucket < 0) + return; + int failures = atomic_fetch_add(®istry->host_fail[bucket], 1) + 1; + if (failures >= registry->lockout_threshold) { + long long now = (long long)time(NULL); + atomic_store(®istry->host_until[bucket], now + (long long)registry->lockout_duration_sec); + } +} + +void daemon_limits_auth_record_success(DaemonLimitRegistry* registry, const char* peer_ip) { + if (!registry) + return; + int bucket = host_lookup(registry, peer_ip); + if (bucket < 0) + return; + atomic_store(®istry->host_fail[bucket], 0); + atomic_store(®istry->host_until[bucket], 0); +} diff --git a/src/shared/daemon_limits.h b/src/shared/daemon_limits.h new file mode 100644 index 0000000..47bfb0b --- /dev/null +++ b/src/shared/daemon_limits.h @@ -0,0 +1,102 @@ +#ifndef DAEMON_LIMITS_H +#define DAEMON_LIMITS_H + +#include +#include +#include + +/* Cross-process daemon connection registry. + * + * The daemon listener forks ONE child per accepted connection, so any + * per-module / per-source accounting must live in state shared across the + * forked children. This module owns a fixed-size registry carved out of an + * anonymous shared mapping (mmap(MAP_SHARED | MAP_ANONYMOUS)) created by the + * accept-loop PARENT before it forks; every child inherits the mapping (and the + * pointer to it) across fork(). + * + * Rules: + * - ONLY C11 atomics (atomic_*); never mtx_t/pthread locks, which can deadlock + * in a forked child if another thread held them at fork time. + * - No heap allocation after fork: the mapping is fixed-size and all access is + * atomic load/store/CAS over preallocated arrays. + * + * Slot lifecycle (the parent reclaims even when a child is SIGKILLed): + * FREE --(parent claim_slot)--> CLAIMED + * CLAIMED --(child register)--> REGISTERED + * any --(parent reclaim)--> FREE + * The child records its module index and per-source bucket into the slot before + * publishing REGISTERED; the parent's SIGCHLD handler matches the reaped pid to + * the slot and, when REGISTERED, decrements the module/per-source counters. + * A child killed before registering holds no counts, so reclaiming a CLAIMED + * slot only frees the slot. + * + * Per-source identity is the normalized numeric peer IP (IPv4-mapped IPv6 is + * already collapsed to IPv4 by utils_fd_peer_ip); it is interned into an + * open-addressed, linear-probing table keyed by a 64-bit hash. The same table + * also carries the cross-process auth-failure counter and lockout deadline. + */ + +typedef struct DaemonLimitRegistry DaemonLimitRegistry; + +/* Result of a per-connection admission check. */ +typedef enum { + DAEMON_LIMIT_OK = 0, /* admitted; slot is now REGISTERED */ + DAEMON_LIMIT_MODULE_FULL, /* module's `max connections` cap reached */ + DAEMON_LIMIT_HOST_FULL, /* global `max connections per host` cap reached */ + DAEMON_LIMIT_UNAVAILABLE, /* registry/slot unusable (caller fails open) */ +} DaemonLimitResult; + +/* Bounds for registry sizing. A slot is one concurrently live child. */ +#define DAEMON_LIMITS_MIN_SLOTS 16 +#define DAEMON_LIMITS_MAX_SLOTS 65536 +#define DAEMON_LIMITS_MAX_HOST_SLOTS 65536 +#define DAEMON_LIMITS_NO_SLOT (-1) + +/* Create the shared registry in the calling (parent) process. `max_slots` is + * the number of concurrently live children to track (clamped to + * [DAEMON_LIMITS_MIN_SLOTS, DAEMON_LIMITS_MAX_SLOTS]); `module_count` is the + * number of daemon modules (clamped to >= 1); `per_host_cap` and the lockout + * pair come from the daemon config (0 disables). Returns NULL on failure (e.g. + * mmap allocation); callers must degrade gracefully (global cap + ACLs still + * apply). */ +DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int per_host_cap, + int lockout_threshold, int lockout_duration_sec); + +/* Unmap the registry. Only the creating process may call this. */ +void daemon_limits_destroy(DaemonLimitRegistry* registry); + +/* Parent side: reserve a slot for the next fork. Returns the slot index or + * DAEMON_LIMITS_NO_SLOT when every slot is in use. */ +int daemon_limits_claim_slot(DaemonLimitRegistry* registry); +/* Parent side: record the forked child's pid in a claimed slot. */ +void daemon_limits_set_slot_pid(DaemonLimitRegistry* registry, int slot, long pid); +/* Parent side: release a slot, decrementing the module/per-source counters when + * the slot was actually REGISTERED. Idempotent. */ +void daemon_limits_reclaim_slot(DaemonLimitRegistry* registry, int slot); +/* Parent SIGCHLD side: reclaim the slot owned by `pid` (no-op when not found). */ +void daemon_limits_reclaim_pid(DaemonLimitRegistry* registry, long pid); + +/* Child side: admit the connection for `module_index` from `peer_ip`. Always + * tracks the module/per-source occupancy (so the parent's reclaim is + * symmetric); when `module_cap` > 0 it additionally enforces the per-module + * cap. Returns DAEMON_LIMIT_OK and publishes the slot, or a refusal reason. */ +DaemonLimitResult daemon_limits_register(DaemonLimitRegistry* registry, int slot, int module_index, + const char* peer_ip, int module_cap); + +/* Child side: true when `peer_ip` is currently locked out after too many failed + * authentications. `seconds_remaining` may be NULL. */ +bool daemon_limits_auth_locked(DaemonLimitRegistry* registry, const char* peer_ip, + int* seconds_remaining); +/* Child side: count one failed authentication for `peer_ip`; once the threshold + * is reached the source is locked out for the configured duration. */ +void daemon_limits_auth_record_failure(DaemonLimitRegistry* registry, const char* peer_ip); +/* Child side: clear the failure counter/lockout for a source that authenticated + * successfully (no-op when the source has no table entry). */ +void daemon_limits_auth_record_success(DaemonLimitRegistry* registry, const char* peer_ip); + +/* Pure helper: 64-bit FNV-1a hash of a numeric peer IP plus its family, used to + * index the per-source table. *ok is set false (and 0 returned) for a NULL or + * non-numeric address. Exposed for unit testing. */ +uint64_t daemon_limits_host_hash(const char* peer_ip, bool* ok); + +#endif diff --git a/tests/runner.c b/tests/runner.c index 9e12323..9549220 100644 --- a/tests/runner.c +++ b/tests/runner.c @@ -9,6 +9,7 @@ #include "test_credentials.h" #include "test_data.h" #include "test_daemon_conf.h" +#include "test_daemon_limits.h" #include "test_delay_updates.h" #include "test_delta.h" #include "test_file.h" @@ -85,6 +86,7 @@ int main() { RUN_TEST(test_client_cli); RUN_TEST(test_server); RUN_TEST(test_daemon_conf); + RUN_TEST(test_daemon_limits); RUN_TEST(test_motd); RUN_TEST(test_server_cli); RUN_TEST(test_fuzz_smoke); diff --git a/tests/test_daemon_limits.c b/tests/test_daemon_limits.c new file mode 100644 index 0000000..2815e0a --- /dev/null +++ b/tests/test_daemon_limits.c @@ -0,0 +1,194 @@ +#include "test_daemon_limits.h" +#include "daemon_limits.h" +#include "test_utils.h" +#include +#include +#include + +/* The per-source hash is a pure helper: numeric addresses hash to a nonzero, + * stable value and unparseable input reports failure. */ +static void test_daemon_limits_host_hash() { + bool ok = false; + uint64_t v4 = daemon_limits_host_hash("127.0.0.1", &ok); + EXPECT_TRUE(ok); + EXPECT_TRUE(v4 != 0); + EXPECT_EQ_INT((int)(daemon_limits_host_hash("127.0.0.1", NULL) == v4), 1); + + bool ok6 = false; + uint64_t v6 = daemon_limits_host_hash("2001:db8::1", &ok6); + EXPECT_TRUE(ok6); + EXPECT_TRUE(v6 != 0); + /* Distinct textual forms of different addresses must differ. */ + EXPECT_TRUE(v4 != v6); + + bool bad = true; + EXPECT_TRUE(daemon_limits_host_hash("not-an-ip", &bad) == 0); + EXPECT_FALSE(bad); + bad = true; + EXPECT_TRUE(daemon_limits_host_hash(NULL, &bad) == 0); + EXPECT_FALSE(bad); + bad = true; + EXPECT_TRUE(daemon_limits_host_hash("", &bad) == 0); + EXPECT_FALSE(bad); +} + +/* Slot reservation is a plain parent-side resource: claim until exhausted, + * reclaim, then claim again. */ +static void test_daemon_limits_slots() { + DaemonLimitRegistry* registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 2, 0, 0, 0); + EXPECT_NOT_NULL(registry); + int slots[DAEMON_LIMITS_MIN_SLOTS]; + for (int i = 0; i < DAEMON_LIMITS_MIN_SLOTS; i++) { + slots[i] = daemon_limits_claim_slot(registry); + EXPECT_EQ_INT(slots[i], i); + } + EXPECT_EQ_INT(daemon_limits_claim_slot(registry), DAEMON_LIMITS_NO_SLOT); + daemon_limits_reclaim_slot(registry, slots[3]); + int reclaimed = daemon_limits_claim_slot(registry); + EXPECT_EQ_INT(reclaimed, slots[3]); + daemon_limits_destroy(registry); +} + +/* Per-module accounting: the cap is enforced across slots and a reclaimed slot + * frees a module count. */ +static void test_daemon_limits_module_cap() { + DaemonLimitRegistry* registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 2, 0, 0, 0); + EXPECT_NOT_NULL(registry); + + int slot0 = daemon_limits_claim_slot(registry); + int slot1 = daemon_limits_claim_slot(registry); + int slot2 = daemon_limits_claim_slot(registry); + int slot3 = daemon_limits_claim_slot(registry); + EXPECT_TRUE(slot0 >= 0 && slot1 >= 0 && slot2 >= 0 && slot3 >= 0); + + EXPECT_EQ_INT(daemon_limits_register(registry, slot0, 0, "10.0.0.1", 2), DAEMON_LIMIT_OK); + EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.2", 2), DAEMON_LIMIT_OK); + EXPECT_EQ_INT(daemon_limits_register(registry, slot2, 0, "10.0.0.3", 2), + DAEMON_LIMIT_MODULE_FULL); + /* A different module has its own counter. */ + EXPECT_EQ_INT(daemon_limits_register(registry, slot2, 1, "10.0.0.3", 2), DAEMON_LIMIT_OK); + /* A module cap of 0 is unlimited. */ + EXPECT_EQ_INT(daemon_limits_register(registry, slot3, 0, "10.0.0.3", 0), DAEMON_LIMIT_OK); + + daemon_limits_reclaim_slot(registry, slot0); + daemon_limits_reclaim_slot(registry, slot1); + int slot4 = daemon_limits_claim_slot(registry); + EXPECT_TRUE(slot4 >= 0); + EXPECT_EQ_INT(daemon_limits_register(registry, slot4, 0, "10.0.0.4", 2), DAEMON_LIMIT_OK); + + daemon_limits_destroy(registry); +} + +/* Per-source accounting: the same peer hits the cap, a different peer does not. */ +static void test_daemon_limits_host_cap() { + DaemonLimitRegistry* registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 1, 1, 0, 0); + EXPECT_NOT_NULL(registry); + + int slot0 = daemon_limits_claim_slot(registry); + int slot1 = daemon_limits_claim_slot(registry); + int slot2 = daemon_limits_claim_slot(registry); + EXPECT_TRUE(slot0 >= 0 && slot1 >= 0 && slot2 >= 0); + + EXPECT_EQ_INT(daemon_limits_register(registry, slot0, 0, "10.0.0.1", 0), DAEMON_LIMIT_OK); + EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.1", 0), DAEMON_LIMIT_HOST_FULL); + EXPECT_EQ_INT(daemon_limits_register(registry, slot2, 0, "10.0.0.2", 0), DAEMON_LIMIT_OK); + /* Reclaiming the first source frees its per-host allowance. */ + daemon_limits_reclaim_slot(registry, slot0); + EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.1", 0), DAEMON_LIMIT_OK); + + daemon_limits_destroy(registry); +} + +/* The pid-indexed reclaim is what the parent's SIGCHLD handler uses: a dead + * child's module/source counts must be released. */ +static void test_daemon_limits_reclaim_pid() { + DaemonLimitRegistry* registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 1, 1, 0, 0); + EXPECT_NOT_NULL(registry); + + int slot0 = daemon_limits_claim_slot(registry); + int slot1 = daemon_limits_claim_slot(registry); + EXPECT_TRUE(slot0 >= 0 && slot1 >= 0); + daemon_limits_set_slot_pid(registry, slot0, 4242); + EXPECT_EQ_INT(daemon_limits_register(registry, slot0, 0, "10.0.0.1", 1), DAEMON_LIMIT_OK); + /* Cap (module 1) and per-host (1) are both saturated. */ + EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.1", 1), + DAEMON_LIMIT_MODULE_FULL); + + daemon_limits_reclaim_pid(registry, 4242); + EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.1", 1), DAEMON_LIMIT_OK); + /* Reclaiming an unknown pid is a no-op. */ + daemon_limits_reclaim_pid(registry, 999999); + + daemon_limits_destroy(registry); +} + +/* Cross-process lockout: failures counted in the shared mapping lock the source + * out after the threshold; a success clears it; threshold 0 disables it. */ +static void test_daemon_limits_auth_lockout() { + DaemonLimitRegistry* registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 1, 0, 2, 300); + EXPECT_NOT_NULL(registry); + + int remaining = 0; + EXPECT_FALSE(daemon_limits_auth_locked(registry, "10.0.0.1", &remaining)); + daemon_limits_auth_record_failure(registry, "10.0.0.1"); + EXPECT_FALSE(daemon_limits_auth_locked(registry, "10.0.0.1", &remaining)); + daemon_limits_auth_record_failure(registry, "10.0.0.1"); + EXPECT_TRUE(daemon_limits_auth_locked(registry, "10.0.0.1", &remaining)); + EXPECT_TRUE(remaining > 0 && remaining <= 300); + /* Another source is unaffected. */ + EXPECT_FALSE(daemon_limits_auth_locked(registry, "10.0.0.2", &remaining)); + /* A successful authentication clears the lockout. */ + daemon_limits_auth_record_success(registry, "10.0.0.1"); + EXPECT_FALSE(daemon_limits_auth_locked(registry, "10.0.0.1", &remaining)); + daemon_limits_destroy(registry); + + /* threshold 0 disables the lockout entirely. */ + registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 1, 0, 0, 300); + EXPECT_NOT_NULL(registry); + for (int i = 0; i < 50; i++) + daemon_limits_auth_record_failure(registry, "10.0.0.1"); + EXPECT_FALSE(daemon_limits_auth_locked(registry, "10.0.0.1", &remaining)); + daemon_limits_destroy(registry); +} + +/* The registry must be visible across fork(): a child's registration is seen by + * the parent, and the parent's pid reclaim releases it. */ +static void test_daemon_limits_fork_shared() { + if (is_running_under_valgrind()) + return; /* fork + shared mapping is slow/noisy under valgrind */ + DaemonLimitRegistry* registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 1, 0, 0, 0); + EXPECT_NOT_NULL(registry); + + int slot0 = daemon_limits_claim_slot(registry); + EXPECT_TRUE(slot0 >= 0); + pid_t pid = fork(); + if (pid == 0) { + if (daemon_limits_register(registry, slot0, 0, "10.0.0.1", 1) != DAEMON_LIMIT_OK) + _exit(1); + _exit(0); + } + EXPECT_TRUE(pid > 0); + daemon_limits_set_slot_pid(registry, slot0, (long)pid); + int status = 0; + EXPECT_TRUE(waitpid(pid, &status, 0) == pid); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + /* The child's module count is still held in the shared mapping. */ + int slot1 = daemon_limits_claim_slot(registry); + EXPECT_TRUE(slot1 >= 0); + EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.2", 1), + DAEMON_LIMIT_MODULE_FULL); + /* The parent reclaims the dead child's slot by pid. */ + daemon_limits_reclaim_pid(registry, (long)pid); + EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.2", 1), DAEMON_LIMIT_OK); + daemon_limits_destroy(registry); +} + +void test_daemon_limits() { + test_daemon_limits_host_hash(); + test_daemon_limits_slots(); + test_daemon_limits_module_cap(); + test_daemon_limits_host_cap(); + test_daemon_limits_reclaim_pid(); + test_daemon_limits_auth_lockout(); + test_daemon_limits_fork_shared(); +} diff --git a/tests/test_daemon_limits.h b/tests/test_daemon_limits.h new file mode 100644 index 0000000..67e905a --- /dev/null +++ b/tests/test_daemon_limits.h @@ -0,0 +1,6 @@ +#ifndef TEST_DAEMON_LIMITS_H +#define TEST_DAEMON_LIMITS_H + +void test_daemon_limits(); + +#endif From 0abaa62193d5f918d8b7c499c76b87bd5a6755dd Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:24:01 +0200 Subject: [PATCH 039/155] feat(daemon): parse per-host cap and auth lockout config keys Add global keys `max connections per host` (default 0 = unlimited), `auth lockout threshold` (default 10, 0 disables) and `auth lockout duration` (default 300 s, 0 disables). Module `max connections` now accepts 0 as unlimited. Bound the number of [module] sections (DAEMON_CONF_MAX_MODULES) so the shared registry's per-module counter array stays fixed-size; absent keys keep their defaults so old configs still load. --- src/shared/daemon_conf.c | 43 ++++++++++++++++++++++++++- src/shared/daemon_conf.h | 56 +++++++++++++++++++++++++---------- tests/test_daemon_conf.c | 63 ++++++++++++++++++++++++++++++++++++---- 3 files changed, 140 insertions(+), 22 deletions(-) diff --git a/src/shared/daemon_conf.c b/src/shared/daemon_conf.c index 01a759a..300e196 100644 --- a/src/shared/daemon_conf.c +++ b/src/shared/daemon_conf.c @@ -190,6 +190,26 @@ static bool store_max_connections(int* slot, const char* value, const char* modu return true; } +/* Parse a non-negative concurrency cap where 0 means unlimited/disabled + * (per-module `max connections`, `max connections per host`, + * `auth lockout threshold`). Negative/garbage/oversized values are rejected. */ +static bool store_optional_cap(int* slot, const char* value, int max_value, const char* key, + const char* module_name, char* err, size_t err_size) { + char* end = NULL; + errno = 0; + long n = strtol(value, &end, 10); + if (*value == '\0' || errno != 0 || *end != '\0' || n < 0 || n > max_value) { + if (module_name) + set_error(err, err_size, "module '%s': invalid '%s' '%s' (must be 0-%d)", module_name, key, + value, max_value); + else + set_error(err, err_size, "invalid '%s' '%s' (must be 0-%d)", key, value, max_value); + return false; + } + *slot = (int)n; + return true; +} + /* Parse an `auth failure delay` value: 0 (disabled) through the configured cap. */ static bool store_auth_failure_delay(int* slot, const char* value, char* err, size_t err_size) { char* end = NULL; @@ -227,6 +247,9 @@ DaemonConf* daemon_conf_create(void) { conf->global.port = DAEMON_CONF_DEFAULT_PORT; conf->global.max_connections = DAEMON_CONF_DEFAULT_MAX_CONNECTIONS; conf->global.auth_failure_delay_ms = DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS; + conf->global.max_connections_per_host = DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST; + conf->global.auth_lockout_threshold = DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD; + conf->global.auth_lockout_duration_sec = DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC; return conf; } @@ -312,8 +335,20 @@ static bool apply_global_key(DaemonConf* conf, char* key, const char* value, boo } if (key_equals(key, "max connections")) return store_max_connections(&conf->global.max_connections, value, NULL, err, err_size); + if (key_equals(key, "max connections per host")) + return store_optional_cap(&conf->global.max_connections_per_host, value, + DAEMON_CONF_MAX_CONCURRENCY_LIMIT, "max connections per host", NULL, + err, err_size); if (key_equals(key, "auth failure delay")) return store_auth_failure_delay(&conf->global.auth_failure_delay_ms, value, err, err_size); + if (key_equals(key, "auth lockout threshold")) + return store_optional_cap(&conf->global.auth_lockout_threshold, value, + DAEMON_CONF_MAX_CONCURRENCY_LIMIT, "auth lockout threshold", NULL, + err, err_size); + if (key_equals(key, "auth lockout duration")) + return store_optional_cap(&conf->global.auth_lockout_duration_sec, value, + DAEMON_CONF_MAX_AUTH_LOCKOUT_DURATION_SEC, "auth lockout duration", + NULL, err, err_size); if (key_equals(key, "hosts allow")) return store_host_list(&conf->global.hosts_allow, &conf->global.hosts_allow_count, value, "hosts allow", NULL, replace_hosts, err, err_size); @@ -400,7 +435,8 @@ static bool apply_module_key(DaemonModule* module, char* key, char* value, char* return true; } if (key_equals(key, "max connections")) - return store_max_connections(&module->max_connections, value, module->name, err, err_size); + return store_optional_cap(&module->max_connections, value, DAEMON_CONF_MAX_CONCURRENCY_LIMIT, + "max connections", module->name, err, err_size); if (key_equals(key, "hosts allow")) return store_host_list(&module->hosts_allow, &module->hosts_allow_count, value, "hosts allow", false, module->name, err, err_size); @@ -444,6 +480,11 @@ static int open_module(DaemonConf* conf, int* current_module, const char* name, set_error(err, err_size, "duplicate module '%s'", name); return -1; } + if (conf->module_count >= DAEMON_CONF_MAX_MODULES) { + set_error(err, err_size, "too many modules (limit %d); module '%s' rejected", + DAEMON_CONF_MAX_MODULES, name); + return -1; + } DaemonModule* grown = realloc(conf->modules, (size_t)(conf->module_count + 1) * sizeof(DaemonModule)); if (!grown) { diff --git a/src/shared/daemon_conf.h b/src/shared/daemon_conf.h index 3b790a7..d04399e 100644 --- a/src/shared/daemon_conf.h +++ b/src/shared/daemon_conf.h @@ -52,11 +52,10 @@ typedef struct DaemonModule { activities. Without it the daemon refuses all of them. */ char** auth_users; /* `auth users = a,b`; Wave B credential list */ int auth_user_count; - /* `max connections = N` (optional per-module cap). 0 means "not set" - * (inherit the global cap). Parsed, stored, and validated, but NOT enforced - * per-module: connections are counted in the accept-loop parent before the - * client's module is known, so only the global cap is enforced (see - * transport_tcp.c and the Daemon Mode notes in RSYNC_COMPAT.md). */ + /* `max connections = N` (optional per-module cap). 0 means unlimited. The + * per-connection child records the selected module in the shared registry + * (daemon_limits.c) once the config frame names it, so the cap is enforced + * across all forked children; the parent reclaims the slot on SIGCHLD. */ int max_connections; char** hosts_allow; /* `hosts allow = a,b`; host access allow patterns */ int hosts_allow_count; @@ -67,14 +66,25 @@ typedef struct DaemonModule { /* Global (pre-module) scalar keys. `motd file` is parsed and stored but has * no wire effect yet (MOTD display is Wave C). */ typedef struct DaemonConfGlobals { - int port; /* `port`, default DAEMON_CONF_DEFAULT_PORT (873) */ - char* motd_file; /* `motd file`, may be NULL */ - char* address; /* `address` (optional bind address), may be NULL */ - int max_connections; /* `max connections`, default - DAEMON_CONF_DEFAULT_MAX_CONNECTIONS (100) */ - int auth_failure_delay_ms; /* `auth failure delay`, milliseconds; default - DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS */ - char** hosts_allow; /* `hosts allow`; global host access allow patterns */ + int port; /* `port`, default DAEMON_CONF_DEFAULT_PORT (873) */ + char* motd_file; /* `motd file`, may be NULL */ + char* address; /* `address` (optional bind address), may be NULL */ + int max_connections; /* `max connections`, default + DAEMON_CONF_DEFAULT_MAX_CONNECTIONS (100) */ + int auth_failure_delay_ms; /* `auth failure delay`, milliseconds; default + DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS */ + int max_connections_per_host; /* `max connections per host`, concurrent cap per + source IP; default + DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST (0 = + unlimited) */ + int auth_lockout_threshold; /* `auth lockout threshold`, failed attempts from + one source before lockout; default + DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD (0 + disables) */ + int auth_lockout_duration_sec; /* `auth lockout duration`, seconds; default + DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC + (0 disables) */ + char** hosts_allow; /* `hosts allow`; global host access allow patterns */ int hosts_allow_count; char** hosts_deny; /* `hosts deny`; global host access deny patterns */ int hosts_deny_count; @@ -92,11 +102,26 @@ typedef struct DaemonConf { #define DAEMON_CONF_DEFAULT_MAX_CONNECTIONS 100 /* Default `auth failure delay` in milliseconds (0 disables the throttle). */ #define DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS 500 +/* Default `max connections per host` (0 = unlimited). */ +#define DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST 0 +/* Default cross-process auth lockout: 10 failed attempts from one source lock + * it out for 300 s (0 disables either knob). */ +#define DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD 10 +#define DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC 300 +/* Upper bound on a `max connections per host` or `auth lockout threshold` + * value, so a typo cannot size the shared registry absurdly. */ +#define DAEMON_CONF_MAX_CONCURRENCY_LIMIT 1000000 +/* Upper bound on `auth lockout duration` (7 days). */ +#define DAEMON_CONF_MAX_AUTH_LOCKOUT_DURATION_SEC 604800 /* Largest accepted `auth failure delay`, so a typo cannot pin a connection * child in nanosleep for an absurd time. */ /* Bounded well below the socket I/O timeout so a failed-auth child cannot hold * a connection slot for long enough to amplify connection-cap exhaustion. */ #define DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS 5000 +/* Upper bound on the number of [module] sections, so the shared registry's + * per-module counter array stays fixed-size. The parser rejects the next + * section past this bound. */ +#define DAEMON_CONF_MAX_MODULES 256 /* Longest accepted config line (excluding the trailing newline). Longer lines * are rejected rather than buffered unboundedly. */ #define DAEMON_CONF_MAX_LINE 4096 @@ -129,8 +154,9 @@ bool daemon_module_name_valid(const char* name); /* Parse one --dparam=KEY=VALUE (or "--dparam KEY=VALUE") override string and * apply it to the global keys only. Keys are case-insensitive and limited to * the global keys defined by the grammar (port, motd file, address, - * max connections, auth failure delay, hosts allow, hosts deny). Returns 0 on - * success, -1 on error (err filled). */ + * max connections, max connections per host, auth failure delay, + * auth lockout threshold, auth lockout duration, hosts allow, hosts deny). + * Returns 0 on success, -1 on error (err filled). */ int daemon_conf_apply_dparam(DaemonConf* conf, const char* assignment, char* err, size_t err_size); /* Host access-control matching (pure; no I/O). `daemon_host_pattern_match` diff --git a/tests/test_daemon_conf.c b/tests/test_daemon_conf.c index e0020c9..a9113a5 100644 --- a/tests/test_daemon_conf.c +++ b/tests/test_daemon_conf.c @@ -33,6 +33,11 @@ static void test_daemon_conf_create_defaults() { EXPECT_NULL(conf->global.address); EXPECT_EQ_INT(conf->global.max_connections, DAEMON_CONF_DEFAULT_MAX_CONNECTIONS); EXPECT_EQ_INT(conf->global.auth_failure_delay_ms, DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS); + EXPECT_EQ_INT(conf->global.max_connections_per_host, + DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST); + EXPECT_EQ_INT(conf->global.auth_lockout_threshold, DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD); + EXPECT_EQ_INT(conf->global.auth_lockout_duration_sec, + DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC); EXPECT_EQ_INT(conf->global.hosts_allow_count, 0); EXPECT_EQ_INT(conf->global.hosts_deny_count, 0); EXPECT_EQ_INT(conf->module_count, 0); @@ -316,6 +321,12 @@ static void test_daemon_conf_dparam_override() { EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "max connections=7", err, sizeof(err)), 0); EXPECT_EQ_INT(conf->global.max_connections, 7); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "max connections per host=3", err, sizeof(err)), 0); + EXPECT_EQ_INT(conf->global.max_connections_per_host, 3); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "auth lockout threshold=5", err, sizeof(err)), 0); + EXPECT_EQ_INT(conf->global.auth_lockout_threshold, 5); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "auth lockout duration=120", err, sizeof(err)), 0); + EXPECT_EQ_INT(conf->global.auth_lockout_duration_sec, 120); EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "AUTH FAILURE DELAY=1500", err, sizeof(err)), 0); EXPECT_EQ_INT(conf->global.auth_failure_delay_ms, 1500); EXPECT_EQ_INT( @@ -390,6 +401,9 @@ static void test_daemon_conf_limits_and_hosts_parse() { char err[256]; EXPECT_EQ_INT(write_conf("max connections = 25\n" "auth failure delay = 0\n" + "max connections per host = 4\n" + "auth lockout threshold = 3\n" + "auth lockout duration = 60\n" "hosts allow = 10.0.0.0/8, 192.168.1.0/24\n" "hosts deny = 192.168.0.1 2001:db8::/32\n" "\n" @@ -405,6 +419,9 @@ static void test_daemon_conf_limits_and_hosts_parse() { EXPECT_NOT_NULL(conf); EXPECT_EQ_INT(conf->global.max_connections, 25); EXPECT_EQ_INT(conf->global.auth_failure_delay_ms, 0); + EXPECT_EQ_INT(conf->global.max_connections_per_host, 4); + EXPECT_EQ_INT(conf->global.auth_lockout_threshold, 3); + EXPECT_EQ_INT(conf->global.auth_lockout_duration_sec, 60); EXPECT_EQ_INT(conf->global.hosts_allow_count, 2); EXPECT_EQ_STR(conf->global.hosts_allow[0], "10.0.0.0/8"); EXPECT_EQ_STR(conf->global.hosts_allow[1], "192.168.1.0/24"); @@ -419,11 +436,14 @@ static void test_daemon_conf_limits_and_hosts_parse() { daemon_conf_free(conf); const char* bad_values[] = { - "max connections = 0\n", "max connections = -1\n", - "max connections = abc\n", "auth failure delay = -1\n", - "auth failure delay = 70000\n", "auth failure delay = soon\n", - "hosts allow = 10.0.0.0/99\n", "hosts deny = 2001:db8::/129\n", - "hosts allow = *.example.com\n", "hosts deny = not-an-ip\n", + "max connections = 0\n", "max connections = -1\n", + "max connections = abc\n", "auth failure delay = -1\n", + "auth failure delay = 70000\n", "auth failure delay = soon\n", + "max connections per host = -1\n", "max connections per host = lots\n", + "auth lockout threshold = -2\n", "auth lockout threshold = many\n", + "auth lockout duration = -1\n", "auth lockout duration = forever\n", + "hosts allow = 10.0.0.0/99\n", "hosts deny = 2001:db8::/129\n", + "hosts allow = *.example.com\n", "hosts deny = not-an-ip\n", }; for (size_t i = 0; i < sizeof(bad_values) / sizeof(bad_values[0]); i++) { EXPECT_EQ_INT(write_conf(bad_values[i], &path), 0); @@ -434,7 +454,8 @@ static void test_daemon_conf_limits_and_hosts_parse() { /* The same strictness applies inside a module section. */ const char* bad_module[] = { - "[m]\npath = /x\nmax connections = 0\n", + "[m]\npath = /x\nmax connections = -1\n", + "[m]\npath = /x\nmax connections = abc\n", "[m]\npath = /x\nhosts allow = 10.0.0.0/40\n", "[m]\npath = /x\nhosts deny = 999.1.1.1/8\n", }; @@ -446,6 +467,14 @@ static void test_daemon_conf_limits_and_hosts_parse() { EXPECT_TRUE(strstr(err, "invalid") != NULL); } + /* Module `max connections = 0` is now valid and means unlimited. */ + EXPECT_EQ_INT(write_conf("[m]\npath = /x\nmax connections = 0\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_EQ_INT(conf->modules[0].max_connections, 0); + daemon_conf_free(conf); + /* An empty hosts list is not an error (no patterns are added). */ EXPECT_EQ_INT(write_conf("hosts allow = \n[m]\npath = /x\n", &path), 0); conf = daemon_conf_load(path, err, sizeof(err)); @@ -509,6 +538,27 @@ static void test_daemon_module_name_valid() { } } +static void test_daemon_conf_module_count_capped() { + size_t cap = DAEMON_CONF_MAX_MODULES; + size_t len = (cap + 8) * 32; + char* body = malloc(len); + EXPECT_NOT_NULL(body); + body[0] = '\0'; + for (size_t i = 0; i < cap + 1; i++) { + char line[48]; + snprintf(line, sizeof(line), "[m%zu]\npath = /x\n", i); + strcat(body, line); + } + char* path; + EXPECT_EQ_INT(write_conf(body, &path), 0); + free(body); + char err[256]; + const DaemonConf* conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "too many modules") != NULL); +} + void test_daemon_conf() { test_daemon_conf_create_defaults(); test_daemon_conf_full_parse(); @@ -525,6 +575,7 @@ void test_daemon_conf() { test_daemon_conf_dparam_override(); test_daemon_conf_auth_users_validated(); test_daemon_conf_limits_and_hosts_parse(); + test_daemon_conf_module_count_capped(); test_daemon_hosts_allowed(); test_daemon_module_name_valid(); } \ No newline at end of file From 4c17122b008aa7dae550c4235d6787cf12427820 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:24:05 +0200 Subject: [PATCH 040/155] feat(daemon): enforce per-module/per-host caps and shared auth lockout Wire the shared registry into the accept loop (parent claims a slot before fork, blocks SIGCHLD across fork+pid publication, and reclaims the dead child's slot from the SIGCHLD handler so per-module/per-source counts are released even on SIGKILL). The connection child records the selected module and normalized peer IP once the config frame names them: an over-cap module or source is refused at the config gate with an audit log, and a source that exceeded the auth-failure threshold is refused before a SCRAM challenge (the counter is shared across children and cleared on success). The existing global cap and host ACLs are untouched. --- src/server/server.c | 114 +++++++++++++++++++++++++++++-- src/shared/transport_tcp.c | 52 +++++++++++++- src/shared/transport_tcp.h | 13 ++++ tests/integration/test_daemon.py | 79 ++++++++++++++++++++- 4 files changed, 248 insertions(+), 10 deletions(-) diff --git a/src/server/server.c b/src/server/server.c index 494a840..6f3862c 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -2,6 +2,7 @@ #include "charset.h" #include "credentials.h" #include "daemon_conf.h" +#include "daemon_limits.h" #include "delay_updates.h" #include "file.h" #include "identity.h" @@ -56,6 +57,12 @@ static DaemonConf* g_daemon_conf = NULL; * such a module exists. */ static CredentialStore* g_credentials = NULL; +/* Cross-process connection registry (per-module and per-source caps plus the + * shared auth lockout), created once in main BEFORE the accept loop forks and + * shared read-only-by-pointer with every connection child. NULL outside daemon + * mode or when the mapping could not be allocated (global cap + ACLs remain). */ +static DaemonLimitRegistry* g_daemon_limits = NULL; + /* Opaque context threaded through to the config-frame gate: the connection's * SSL object (NULL over plaintext) so the gate can warn when a credential * exchange is not encrypted, plus the super-mode override the gate decides on. @@ -292,6 +299,56 @@ static const DaemonModule* module_gate_lookup_module(const Config* config, const return module; } +/* Index of `module` within the loaded config's module array (the registry's + * per-module counter key). Returns -1 when it cannot be resolved. */ +static int daemon_module_index(const DaemonModule* module) { + if (!g_daemon_conf || !module || module < g_daemon_conf->modules || + module >= g_daemon_conf->modules + g_daemon_conf->module_count) + return -1; + return (int)(module - g_daemon_conf->modules); +} + +/* Shared-registry admission: reserve this connection's slot for the selected + * module and the peer source IP. Enforces the per-module `max connections` and + * the global `max connections per host` across every forked child. Runs before + * auth/ownership so a client that is over a cap is refused before any work. + * The per-source cap is skipped when the peer cannot be classified (host ACLs + * fail closed separately); the module cap still applies. A missing registry + * (allocation failure / non-fork path) fails open -- the global cap and ACLs + * still bound the listener. */ +static const char* module_gate_check_limits(const Config* config, const DaemonModule* module, + ModuleGateContext* gate_ctx) { + if (!g_daemon_limits) + return NULL; + int slot = transport_tcp_current_slot(); + if (slot < 0) + return NULL; /* not on the forked accept-loop path (e.g. --stdio) */ + int module_index = daemon_module_index(module); + if (module_index < 0) + return NULL; + const char* peer = (gate_ctx && gate_ctx->has_peer_ip) ? gate_ctx->peer_ip : ""; + DaemonLimitResult result = + daemon_limits_register(g_daemon_limits, slot, module_index, peer, module->max_connections); + switch (result) { + case DAEMON_LIMIT_OK: + return NULL; + case DAEMON_LIMIT_MODULE_FULL: + log_message(LOG_LEVEL_ERROR, + "daemon module '%s': 'max connections' cap (%d) reached; refusing %s", + config->module, module->max_connections, peer[0] ? peer : "peer"); + return "requested daemon module is at its connection limit"; + case DAEMON_LIMIT_HOST_FULL: + log_message(LOG_LEVEL_ERROR, + "daemon: 'max connections per host' cap (%d) reached for %s; refusing module '%s'", + g_daemon_conf->global.max_connections_per_host, peer[0] ? peer : "peer", + config->module); + return "too many concurrent connections from this host"; + case DAEMON_LIMIT_UNAVAILABLE: + default: + return NULL; + } +} + /* Per-module client-chosen ownership / super-user policy (P7 Wave E hardening): * a daemon module refuses EVERY ownership-affecting request (--numeric-ids, * --chown, --usermap/--groupmap, --fake-super, --copy-as, explicit --super) @@ -396,6 +453,20 @@ static ModuleAuthResult module_gate_authenticate(const Config* config, const Dae ModuleGateContext* gate_ctx, const char** error) { if (module->auth_user_count == 0) return MODULE_AUTH_ACCEPTED; + /* Cross-process lockout: a source that failed too many authentications is + * refused before the challenge is sent (the counter lives in the shared + * registry, so it spans every forked child and survives a child exit). */ + if (g_daemon_limits && gate_ctx && gate_ctx->has_peer_ip) { + int remaining = 0; + if (daemon_limits_auth_locked(g_daemon_limits, gate_ctx->peer_ip, &remaining)) { + log_message(LOG_LEVEL_ERROR, + "daemon module '%s': source %s is locked out after repeated authentication " + "failures (%d s remaining); refusing", + config->module, gate_ctx->peer_ip, remaining); + *error = "too many failed authentication attempts from this host; try again later"; + return MODULE_AUTH_REFUSED; + } + } /* Fail closed: no store -> refuse (server misconfiguration, STATUS_ERROR). */ if (g_credentials == NULL) { log_message(LOG_LEVEL_ERROR, @@ -450,10 +521,16 @@ static ModuleAuthResult module_gate_authenticate(const Config* config, const Dae "daemon module '%s': authentication failed for user '%s' from %s; refusing", config->module, escaped_user ? escaped_user : "(none)", peer); free(escaped_user); - /* Rate-limit online guessing per connection (no delay on success). */ + /* Count the failure in the shared registry (locks the source out once the + * configured threshold is reached) and rate-limit online guessing per + * connection (no delay on success). */ + if (g_daemon_limits && gate_ctx->has_peer_ip) + daemon_limits_auth_record_failure(g_daemon_limits, gate_ctx->peer_ip); daemon_auth_failure_delay(); return MODULE_AUTH_TERMINATED; } + if (g_daemon_limits && gate_ctx->has_peer_ip) + daemon_limits_auth_record_success(g_daemon_limits, gate_ctx->peer_ip); char* escaped_user = output_escape(config->auth_user, config->eight_bit_output); log_message(LOG_LEVEL_INFO, "daemon module '%s': user '%s' from %s authenticated", config->module, escaped_user ? escaped_user : "", @@ -561,6 +638,9 @@ static const char* server_module_gate(const Config* config, void* context) { log_message(LOG_LEVEL_DEBUG, "daemon module '%s': peer address unavailable", config->module); } error = module_gate_check_hosts(config, module, gate_ctx); + if (error) + return error; + error = module_gate_check_limits(config, module, gate_ctx); if (error) return error; error = module_gate_check_ownership(config, module, gate_ctx); @@ -880,7 +960,9 @@ static void print_server_usage(void) { printf(" fastsyncd.conf, else /etc/fastsyncd.conf)\n"); printf(" --dparam=KEY=VALUE Override one global config key on the command line\n"); printf(" (port, motd file, address, max connections,\n"); - printf(" auth failure delay, hosts allow, hosts deny)\n"); + printf(" max connections per host, auth failure delay,\n"); + printf(" auth lockout threshold, auth lockout duration,\n"); + printf(" hosts allow, hosts deny)\n"); printf(" --no-detach Stay in the foreground (default detaches to\n"); printf(" background when running --daemon)\n"); printf(" --password-file=FILE Credential store for modules that declare\n"); @@ -1096,11 +1178,10 @@ int main(int argc, char* argv[]) { "unless the module is intentionally open to the network", g_daemon_conf->modules[i].name); if (g_daemon_conf->modules[i].max_connections > 0) - log_message(LOG_LEVEL_WARNING, - "daemon module '%s': per-module 'max connections' is stored but not enforced " - "per module; the global 'max connections' cap (%d) applies to the whole " - "listener", - g_daemon_conf->modules[i].name, g_daemon_conf->global.max_connections); + log_message(LOG_LEVEL_INFO, + "daemon module '%s': per-module 'max connections' cap = %d (enforced " + "across all connection children)", + g_daemon_conf->modules[i].name, g_daemon_conf->modules[i].max_connections); } /* Daemon credential store (Wave B). --password-file and --early-input * feed the same store, loaded BEFORE the listener forks so every @@ -1144,6 +1225,21 @@ int main(int argc, char* argv[]) { module->name, module->auth_users[j]); } } + /* Shared cross-process registry for the per-module / per-source caps and + * the auth lockout. Created HERE in the parent before any accept-loop + * fork; every connection child inherits the mapping. A failure degrades to + * "registry disabled" (the global cap and host ACLs still apply) rather + * than refusing to start. */ + g_daemon_limits = daemon_limits_create((int)g_daemon_conf->global.max_connections, + g_daemon_conf->module_count, + g_daemon_conf->global.max_connections_per_host, + g_daemon_conf->global.auth_lockout_threshold, + g_daemon_conf->global.auth_lockout_duration_sec); + if (!g_daemon_limits) + log_message(LOG_LEVEL_WARNING, + "daemon: could not allocate the shared connection registry; per-module / " + "per-host caps and the cross-process auth lockout are disabled (the global " + "'max connections' cap and host ACLs still apply)"); } else { if (!configure_authorization(opts.destination_root)) { char* escaped = output_escape(opts.destination_root, false); @@ -1167,6 +1263,8 @@ int main(int argc, char* argv[]) { } if (g_daemon_conf) server_set_max_connections(g_server, (unsigned int)g_daemon_conf->global.max_connections); + if (g_daemon_limits) + server_set_limit_registry(g_server, g_daemon_limits); if (opts.use_tls) { if (!opts.tls_cert || !opts.tls_key || !opts.tls_ca || !opts.client_cn) { fprintf(stderr, "Error: --tls requires --cert, --key, --ca, and --client-cn\n"); @@ -1206,6 +1304,8 @@ int main(int argc, char* argv[]) { release_authorization(); out: + daemon_limits_destroy(g_daemon_limits); + g_daemon_limits = NULL; daemon_conf_free(g_daemon_conf); g_daemon_conf = NULL; credentials_free(g_credentials); diff --git a/src/shared/transport_tcp.c b/src/shared/transport_tcp.c index 2758cbe..e73dbce 100644 --- a/src/shared/transport_tcp.c +++ b/src/shared/transport_tcp.c @@ -1,4 +1,5 @@ #include "transport_tcp.h" +#include "daemon_limits.h" #include "log.h" #include "protocol.h" #include "utils.h" @@ -18,15 +19,25 @@ static volatile sig_atomic_t g_active_connections = 0; +/* Shared registry installed on the active server; the SIGCHLD handler needs a + * file-scope pointer so it can reclaim the dead child's slot. Set once by + * accept_loop before the fork loop (single-threaded parent). */ +static DaemonLimitRegistry* g_limit_registry = NULL; +/* Slot reserved by the parent for the connection child currently being forked. + * Written before fork(), read by the child (which inherits the value). */ +static int g_current_slot = DAEMON_LIMITS_NO_SLOT; + static void tcp_apply_socket_timeout(int fd); static void tcp_enable_nodelay_default(int fd, int family); static void sigchld_handler(int sig) { (void)sig; int saved_errno = errno; - while (waitpid(-1, NULL, WNOHANG) > 0) { + pid_t pid; + while ((pid = waitpid(-1, NULL, WNOHANG)) > 0) { if (g_active_connections > 0) g_active_connections--; + daemon_limits_reclaim_pid(g_limit_registry, (long)pid); } errno = saved_errno; } @@ -108,6 +119,7 @@ Server* server_create_ex(int port, const ServerBindOptions* bind_opts) { server->ssl_ctx = NULL; server->max_connections = 100; server->active_connections = 0; + server->limit_registry = NULL; return server; } @@ -121,6 +133,15 @@ void server_set_max_connections(Server* server, unsigned int max_connections) { server->max_connections = max_connections; } +void server_set_limit_registry(Server* server, struct DaemonLimitRegistry* registry) { + if (server) + server->limit_registry = registry; +} + +int transport_tcp_current_slot(void) { + return g_current_slot; +} + void server_delete(Server** server) { if (server == NULL || *server == NULL) return; @@ -140,6 +161,7 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil return; } signal(SIGCHLD, sigchld_handler); + g_limit_registry = server->limit_registry; while (1) { struct sockaddr_storage client_addr; socklen_t client_len = sizeof(client_addr); @@ -159,9 +181,31 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil close(fd); continue; } + int slot = DAEMON_LIMITS_NO_SLOT; + if (server->limit_registry) { + slot = daemon_limits_claim_slot(server->limit_registry); + if (slot == DAEMON_LIMITS_NO_SLOT) { + /* The global cap bounds live children, so this only happens when the + * fixed registry is smaller than the configured cap; fail closed. */ + log_message(LOG_LEVEL_WARNING, "Connection registry slots exhausted (max %u), rejecting %s", + server->max_connections, peer); + close(fd); + continue; + } + } log_message(LOG_LEVEL_INFO, "%s from %s", log_fmt, peer); + g_current_slot = slot; + /* Block SIGCHLD across fork() and the parent's pid publication: a child + * that exits immediately must not be reaped before its slot records its + * pid, which would leak the slot and its module/source counts. */ + sigset_t blocked; + sigset_t previous; + sigemptyset(&blocked); + sigaddset(&blocked, SIGCHLD); + sigprocmask(SIG_BLOCK, &blocked, &previous); pid_t pid = fork(); if (pid == 0) { + sigprocmask(SIG_SETMASK, &previous, NULL); /* Connection children must not run the parent's global cleanup(): it * frees state (credentials / daemon conf) that the child's worker * threads may still be reading and closes fd numbers the child could @@ -177,7 +221,13 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil _exit(0); } else if (pid > 0) { g_active_connections++; + if (server->limit_registry) + daemon_limits_set_slot_pid(server->limit_registry, slot, (long)pid); + } else if (server->limit_registry) { + /* fork() failed: release the reservation so the slot is not leaked. */ + daemon_limits_reclaim_slot(server->limit_registry, slot); } + sigprocmask(SIG_SETMASK, &previous, NULL); close(fd); } } diff --git a/src/shared/transport_tcp.h b/src/shared/transport_tcp.h index e37b879..c3862a6 100644 --- a/src/shared/transport_tcp.h +++ b/src/shared/transport_tcp.h @@ -7,6 +7,10 @@ #include #include +/* Cross-process daemon registry (daemon_limits.c). Only an opaque pointer is + * stored here so the transport layer does not depend on daemon config. */ +struct DaemonLimitRegistry; + typedef struct Server { struct sockaddr_storage address; unsigned int address_length; @@ -14,6 +18,7 @@ typedef struct Server { void* ssl_ctx; unsigned int max_connections; volatile unsigned int active_connections; + struct DaemonLimitRegistry* limit_registry; } Server; typedef struct Client { @@ -48,6 +53,14 @@ Server* server_create(int port); /* Override the listener's connection cap (the global daemon `max connections` * value). A non-positive value is ignored so the default cap stands. */ void server_set_max_connections(Server* server, unsigned int max_connections); +/* Install the shared per-module / per-source registry used by the accept loop + * to reserve a slot for each forked child. NULL disables the accounting (the + * global cap and ACLs still apply). */ +void server_set_limit_registry(Server* server, struct DaemonLimitRegistry* registry); +/* Slot reserved for the connection child currently running (set by the parent + * before fork, inherited by the child). Returns DAEMON_LIMITS_NO_SLOT (-1) + * outside the accept-loop child path. */ +int transport_tcp_current_slot(void); bool server_listen(Server* server, void (*handler)(int file_descriptor)); void server_accept_loop(Server* server, void (*child_fn)(int, void*), void* child_ctx, const char* log_fmt); diff --git a/tests/integration/test_daemon.py b/tests/integration/test_daemon.py index 98c963d..1ff1d9d 100644 --- a/tests/integration/test_daemon.py +++ b/tests/integration/test_daemon.py @@ -135,7 +135,7 @@ class DaemonManager: self._proc = None self._port = None - def start(self, config_path, port_override=None, extra_args=None): + def start(self, config_path, port_override=None, extra_args=None, log_path=None): self.stop() # When no override is given the daemon binds the config file's `port` # (the plain config-port path); with an override the --dparam path. @@ -146,7 +146,8 @@ class DaemonManager: cmd += ["--dparam", f"port={port_override}"] if extra_args: cmd += extra_args - log_path = os.path.join(TEST_DATA_DIR, "fastsyncd.log") + if log_path is None: + log_path = os.path.join(TEST_DATA_DIR, "fastsyncd.log") log = open(log_path, "w") self._proc = subprocess.Popen( cmd, stdout=log, stderr=log, stdin=subprocess.DEVNULL, start_new_session=True) @@ -1208,3 +1209,77 @@ class TestDaemonTLSAuth: d.stop() os.unlink(client_creds) shutil.rmtree(cert_dir, ignore_errors=True) + + +class TestDaemonConnectionLimits: + """Wave 8: cross-process per-module / per-source connection caps and the + shared auth lockout. Each test boots its own daemon with a unique port so + the shared (per-daemon) registry state is isolated from the module-scoped + `daemon` fixture.""" + + LOCKOUT_CONF = os.path.join(TEST_DATA_DIR, "fastsyncd_lockout.conf") + CAPS_CONF = os.path.join(TEST_DATA_DIR, "fastsyncd_caps.conf") + + @pytest.mark.ci + def test_auth_lockout_is_shared_across_children(self): + """`auth lockout threshold = 1`: the first failed authentication locks the + source out for the cooldown in the SHARED registry, so a subsequent + correct-password attempt (a different forked child) is refused before a + SCRAM challenge is even sent.""" + port = _find_free_port() + with open(self.LOCKOUT_CONF, "w") as f: + f.write("port = %d\n" + "auth lockout threshold = 1\n" + "auth lockout duration = 300\n" + "\n" + "[locked]\n" + "path = %s\n" + "auth users = alice\n" + % (port, AUTH_MODULE)) + d = DaemonManager() + log_path = os.path.join(TEST_DATA_DIR, f"fastsyncd_lockout_{os.getpid()}.log") + try: + d.start(self.LOCKOUT_CONF, port_override=port, extra_args=["--password-file", CRED_FILE], + log_path=log_path) + before = _tree_file_count(AUTH_MODULE) + log_before = os.path.getsize(log_path) if os.path.exists(log_path) else 0 + # First attempt: wrong password -> records failure #1 -> locks. + wrong = _push_with_creds("127.0.0.1::locked", port, "alice", WRONG_PASS) + assert wrong.returncode != 0 + # Second attempt: CORRECT password from the same source must still be + # refused by the shared lockout. + right = _push_with_creds("127.0.0.1::locked", port, "alice", ALICE_PASS) + assert right.returncode != 0, "the shared auth lockout must refuse after threshold" + assert _tree_file_count(AUTH_MODULE) == before, "a locked-out source wrote data" + time.sleep(0.3) + with open(log_path, "rb") as f: + f.seek(log_before) + tail = f.read().decode("utf-8", "replace") + assert "locked out" in tail, tail[-400:] + finally: + d.stop() + + def test_caps_keys_accepted_and_transfer_still_works(self): + """A daemon configured with the new keys (per-host cap, lockout threshold + and duration, per-module cap) starts and serves a normal transfer.""" + port = _find_free_port() + with open(self.CAPS_CONF, "w") as f: + f.write("port = %d\n" + "max connections per host = 5\n" + "auth lockout threshold = 3\n" + "auth lockout duration = 60\n" + "\n" + "[files]\n" + "path = %s\n" + "max connections = 2\n" + % (port, FILES_MODULE)) + d = DaemonManager() + try: + d.start(self.CAPS_CONF, port_override=port) + result = _push("127.0.0.1::files", port) + assert result.returncode == 0, result.stderr or result.stdout + received = get_dest_received_dir(FILES_MODULE, SOURCE_DIR) + _, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"missing: {missing[:5]}" + finally: + d.stop() From e1f8f75e7c794dc5ea5249755fcc519f4b7993e5 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:24:09 +0200 Subject: [PATCH 041/155] docs: document daemon per-module/per-host caps and shared auth lockout --- CHANGELOG.md | 12 ++++++++++++ README.md | 17 +++++++++++++---- RSYNC_COMPAT.md | 6 +++--- 3 files changed, 28 insertions(+), 7 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 66132a2..a641f4b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,18 @@ All notable changes to FastSync are documented here. Versions match `PROTOCOL_VERSION` (printed by `fastsync --version`); the client and server must run the same version because the handshake is strict. +## [Unreleased] + +### Security + +- Enforce the daemon's per-module `max connections` cap and add a global + `max connections per host` cap plus a cross-process `auth lockout` + (`auth lockout threshold` / `auth lockout duration`). Because the listener + forks one child per connection, the counters live in an anonymous shared + mapping created before the accept loop and reclaimed by the parent's + `SIGCHLD` handler, so the per-module, per-source and auth-failure state is + shared across every child (including after `SIGKILL`). + ## [2.20.0] - 2026-09-13 ### Security diff --git a/README.md b/README.md index 9820885..58d0d10 100644 --- a/README.md +++ b/README.md @@ -506,18 +506,27 @@ defaults to the current directory. | implicit global section, then `[module]` sections). Besides `port`, `motd file`, and `address`, the global section accepts: -- `max connections = N` — cap on concurrent connections, default 100. The +- `max connections = N` — global cap on concurrent connections, default 100. The listener enforces it; `0`, negative, and non-numeric values are parse errors. +- `max connections per host = N` — cap on concurrent connections from a single + source IP, default 0 (unlimited). Enforced across all forked connection + children through a shared registry. - `auth failure delay = MS` — milliseconds to sleep after a failed authentication, default 500. `0` disables it and the value is capped at 60000, so online password guessing is rate-limited per connection. Successful auths are never delayed. +- `auth lockout threshold = N` — number of failed authentications from one source + IP before that source is locked out, default 10; `0` disables the lockout. The + failure counter is shared across every connection child, so the lockout holds + even when the next attempt is handled by a different forked child. +- `auth lockout duration = SECONDS` — how long a locked-out source is refused + (default 300). A locked-out client is refused before any SCRAM challenge is + sent; a successful authentication clears the counter. - `hosts allow` / `hosts deny` — comma- and/or whitespace-separated host access patterns. -A `[module]` may also set `max connections` (parsed and validated but not -enforced per module — the global cap applies to the whole listener) and its own -`hosts allow`/`hosts deny`. +A `[module]` may also set `max connections` (0 = unlimited; enforced per module +across all connection children) and its own `hosts allow`/`hosts deny`. Host patterns are `*` (match all), IPv4/IPv6 literals, or IPv4/IPv6 CIDR (`10.0.0.0/8`, `2001:db8::/32`). Hostnames are not resolved, so hostname globs diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index ea171bc..f0d4ba2 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -627,7 +627,7 @@ now transmits targets (the prior behavior was broken/partial); its status moved |------|-------------------|-----------------|-------| | `--daemon` | Run as rsync daemon | ✅ Implemented | Wave A: a real persistent listener. `fastsync-server --daemon --config FILE` (plus `--no-detach` to stay foreground; without it the listener detaches to the background after binding) reads a FastSync-native module config file and serves each connection confined to the requested module's `path` root (never a client-chosen root; every client-chosen-ownership/super-user request (`--numeric-ids`/`--chown`/`--usermap`/`--groupmap`/`--fake-super`/`--copy-as`/explicit `--super`) is refused unless the module opts in with `client owner = yes`, and the operator `--no-super` veto is honored). TCP/TLS via the existing `--tls` stack; plaintext still requires `--allow-unauthenticated` (same secure default as the standalone server). Client destinations use rsync's `host::module/path` form. Wire/protocol: the config frame gained a trailing daemon-module string and `PROTOCOL_VERSION` was bumped **2.14.0 → 2.15.0** (see the Daemon Mode notes below). Daemon mode is built in FastSync's own protocol/config grammar, not rsync's SMB/daemon option encoding | | `--config=FILE` | Alternate rsyncd.conf file | ✅ Implemented | Wave A: selects the daemon config file. Default when omitted (in `--daemon` mode): `~/.config/fastsync/fastsyncd.conf` if it exists, else `/etc/fastsyncd.conf`. The grammar is FastSync-native (documented in the Daemon Mode notes below) and strictly rejects unknown keys so a typo can never silently change what a module serves; requires `--daemon` | -| `--dparam=OVERRIDE` | Override global daemon config | ✅ Implemented | Wave A: overrides one global scalar from the command line (`--dparam port=8734` and `--dparam=KEY=VALUE` both work). Limited to the global keys the grammar defines (`port`, `motd file`, `address`, `max connections`, `auth failure delay`, `hosts allow`, `hosts deny`); keys are case-insensitive and unknown keys/invalid values are rejected. Requires `--daemon` | +| `--dparam=OVERRIDE` | Override global daemon config | ✅ Implemented | Wave A: overrides one global scalar from the command line (`--dparam port=8734` and `--dparam=KEY=VALUE` both work). Limited to the global keys the grammar defines (`port`, `motd file`, `address`, `max connections`, `max connections per host`, `auth failure delay`, `auth lockout threshold`, `auth lockout duration`, `hosts allow`, `hosts deny`); keys are case-insensitive and unknown keys/invalid values are rejected. Requires `--daemon` | | `--no-detach` | Don't detach from parent | ✅ Implemented | Wave A: with `--daemon`, keeps the listener in the foreground (what integration tests use). Without it the daemonizes (fork/setsid, stdio redirected to /dev/null) after the listening socket is bound. Requires `--daemon` | | `--password-file=FILE` | Read daemon password from file | ✅ Implemented | A7 daemon auth. Client: `--password-file` supplies `user:password` for a `host::module/path` destination (the username is taken from this file, so `user@host::module` stays rejected); the literal password is held client-side only for the SCRAM handshake and wiped at teardown. Server (`fastsync-server --daemon --password-file FILE`): the salted-PBKDF2 verifier store that modules with `auth users` are verified against. **Neither the password nor any replayable bearer value crosses the wire or is stored server-side** — the store holds a per-user salt plus derived keys, and the daemon proves the secret with a per-connection nonce challenge. The file must be private to its owner: both the client and server verify the exact inode they read (open-then-`fstat`, so the check cannot be raced) and refuse a `--password-file`/`--early-input` that is not owned by the current user or grants any group/other permission bit (mode 0600), mirroring the TLS private-key check. A process-substitution pipe (`--early-input <(vault ...)`) is still accepted when it satisfies those checks. See the Daemon Mode notes below for the file formats and the plaintext/TLS caveat | | `--early-input=FILE` | Use FILE for daemon early exec | ✅ Implemented | Server-only (requires `--daemon`): a second credential-store file, same new-format grammar as `--password-file`, read before the listener accepts connections (a secrets-manager / process-substitution source). Its entries layer over `--password-file`: byte-identical verifiers dedupe, a conflicting verifier for the same user is a startup error. A daemon whose modules declare `auth users` must be given at least one of the two, or it refuses to start (fail closed) | @@ -635,9 +635,9 @@ now transmits targets (the prior behavior was broken/partial); its status moved **Daemon Mode notes (Wave A protocol 2.15.0; A7 auth protocol 2.19.0; MOTD no bump):** FastSync daemon mode is supported in FastSync's own protocol/config grammar, not rsync's SMB/daemon option encoding. -- **Config grammar** (`fastsyncd.conf`): line-based; an implicit global section first, then `[module]` sections. Keys are case-insensitive, values are trimmed and may be wrapped in one layer of double quotes (`path = "/srv/my dir"`). `#` and `;` at the start of a line (after leading whitespace) are full-line comments; inline comments and `\` continuations are not supported. Lines are bounded (4096 chars). Global keys: `port` (default 873), `motd file` (the daemon sends its bounded, escaped content to a client after the module gate/auth accepts, unless the client passes `--no-motd`), `address` (optional bind address), `max connections` (positive integer cap on concurrent connections, default 100; 0/negative/garbage is a parse error), `auth failure delay` (milliseconds to sleep after a failed authentication, default 500; 0 disables, capped at 5000), `hosts allow` and `hosts deny` (comma- and/or whitespace-separated host access patterns — see the host access control note below). Module keys: `path` (required; the daemon-side authorized root for that module), `read only` (yes/no/true/false/1/0, default no), `client owner` (yes/no/true/false/1/0, default no; opts the module into client-chosen ownership — see below), `auth users` (comma list), `max connections` (optional per-module cap; parsed and stored but **not enforced** — the global cap applies to the whole listener), `hosts allow`/`hosts deny` (per-module host access lists). **Unknown keys and malformed lines are parse-and-reject errors** (never silently ignored), so a typo cannot change what a module serves. +- **Config grammar** (`fastsyncd.conf`): line-based; an implicit global section first, then `[module]` sections. Keys are case-insensitive, values are trimmed and may be wrapped in one layer of double quotes (`path = "/srv/my dir"`). `#` and `;` at the start of a line (after leading whitespace) are full-line comments; inline comments and `\` continuations are not supported. Lines are bounded (4096 chars), and at most 256 `[module]` sections are accepted. Global keys: `port` (default 873), `motd file` (the daemon sends its bounded, escaped content to a client after the module gate/auth accepts, unless the client passes `--no-motd`), `address` (optional bind address), `max connections` (positive integer cap on concurrent connections, default 100; 0/negative/garbage is a parse error), `max connections per host` (concurrent-connection cap per source IP, default 0 = unlimited), `auth failure delay` (milliseconds to sleep after a failed authentication, default 500; 0 disables, capped at 5000), `auth lockout threshold` (failed authentications from one source before lockout, default 10; 0 disables), `auth lockout duration` (seconds a locked-out source is refused, default 300), `hosts allow` and `hosts deny` (comma- and/or whitespace-separated host access patterns — see the host access control note below). Module keys: `path` (required; the daemon-side authorized root for that module), `read only` (yes/no/true/false/1/0, default no), `client owner` (yes/no/true/false/1/0, default no; opts the module into client-chosen ownership — see below), `auth users` (comma list), `max connections` (optional per-module cap, 0 = unlimited; enforced across all connection children), `hosts allow`/`hosts deny` (per-module host access lists). **Unknown keys and malformed lines are parse-and-reject errors** (never silently ignored), so a typo cannot change what a module serves. - **Host access control (`hosts allow`/`hosts deny`):** both keys accept a comma- and/or whitespace-separated list of patterns and may appear globally and/or per module (multiple config-file lines append; a `--dparam` override replaces). Supported patterns are `*` (match all), an IPv4 or IPv6 literal (`10.0.0.1`, `2001:db8::1`), and an IPv4/IPv6 CIDR (`10.0.0.0/8`, `2001:db8::/32`). Hostname patterns are **not** supported: because the peer is always a numeric address and no reverse DNS is performed, a hostname/glob pattern would silently never match, so it is rejected at load time (fail-closed) instead of being accepted as a dead rule. An IPv4 peer on a dual-stack IPv6 listener is normalized from its `::ffff:a.b.c.d` form so IPv4 patterns match it. rsync-like semantics: a matching `hosts deny` rejects; if any `hosts allow` entries exist, a peer matching none of them is rejected; deny takes precedence over allow. The daemon enforces the global list first, then the selected module's list, **before authentication** in `server_module_gate`, with an audit log line naming the peer, the module and the outcome. The numeric peer address is obtained with `getpeername`+`inet_ntop` (`utils_fd_peer_ip`, handling both address families); when it cannot be obtained a module with any ACL fails closed (refused), while an ACL-free module continues and logs at debug. A malformed pattern (e.g. an out-of-range CIDR prefix) is a parse error at load time. -- **Connection cap and auth throttle:** the global `max connections` key (default 100) is plumbed into the listener (`transport_tcp.c`), which rejects a connection once the accept-loop parent's active-child count reaches it; the IPv4/IPv6 peer is logged for every accepted connection. The optional per-module `max connections` key is parsed and validated but **not enforced** (connections are counted in the parent before the client's module is known); the daemon logs a startup warning for any module that sets it. On a failed authentication the per-connection child sleeps the global `auth failure delay` (default 500 ms, 0 disables, capped at 5000) via `nanosleep` before the connection closes, rate-limiting online guessing without delaying a success. +- **Connection caps, shared registry and auth lockout:** the global `max connections` key (default 100) is plumbed into the listener (`transport_tcp.c`), which rejects a connection once the accept-loop parent's active-child count reaches it; the IPv4/IPv6 peer is logged for every accepted connection. Because the listener forks one child per connection, the per-module `max connections` cap, the global `max connections per host` cap, and the auth-failure counter live in a fixed-size registry carved from an anonymous shared mapping (`daemon_limits.c`, `mmap(MAP_SHARED|MAP_ANONYMOUS)`) created by the parent before the accept loop, so every forked child shares the same counters (C11 atomics only — never a pthread lock, which can deadlock in a forked child). The parent reserves a registry slot per accepted connection and the child records the selected module and source IP once known; the parent's `SIGCHLD` handler reclaims the slot when the child dies (including `SIGKILL`), decrementing the per-module and per-source counts. The per-module cap (0 = unlimited) is enforced after the module lookup and before auth; per-source identity reuses the normalized numeric peer address (`utils_fd_peer_ip`, IPv4-mapped IPv6 collapsed to IPv4). A failed authentication increments the shared per-source failure count and, once `auth lockout threshold` (default 10; 0 disables) is reached, the source is refused for `auth lockout duration` seconds (default 300) before any challenge is sent, even when the next attempt is handled by a different forked child; a successful authentication clears the counter. On a failed authentication the per-connection child still sleeps the global `auth failure delay` (default 500 ms, 0 disables, capped at 5000) via `nanosleep`, rate-limiting online guessing without delaying a success. A missing registry (allocation failure) degrades to the global cap and host ACLs rather than refusing to start. - **Module selection & confinement:** the client requests a module with an rsync-style `host::module[/path]` destination. The module name crosses the wire as a trailing string on the config frame (bumping `PROTOCOL_VERSION` 2.14.0 → 2.15.0; the bump is required because the config-frame layout changed and the strict same-version handshake is what prevents a peer from desynchronizing on the new trailing field). The daemon looks the module up in ITS OWN config and uses the module's `path` as the authorized root through the exact same `configure_authorization` confinement the standalone server applies to `--destination-root` (`file_open_secure_parent`, `has_path_traversal`, `path_is_within`); the client never supplies the root, every client-chosen-ownership/super-user request is refused unless the module declares `client owner = yes` (the daemon's per-module opt-in, see below), and the operator `--no-super` veto forces super-user activities off for every daemon connection. The client's `/path` part is relative inside the module and is rejected if absolute or if it contains `..`. Unknown modules are refused before any data moves (the run fails cleanly at the config handshake). An absolute destination and a module request against a non-daemon server are also refused. - **`client owner` (client-chosen-ownership opt-in):** by default a daemon module refuses every request that would let the client pick an owner or ask for super-user activities — `--numeric-ids`, `--chown`, `--usermap`/`--groupmap`, `--fake-super`, `--copy-as`, and an explicit `--super` — at the config handshake (before `STATUS_OK`), because a daemon has no per-module opt-in for client-chosen ownership and any anonymous client could otherwise force arbitrary owner ids inside the module root. `client owner = yes` opts a single module in, allowing those requests within that module's root (the standalone listener and the SSH `--stdio` server always honor them for their single operator-authorized root). Without the opt-in the daemon also forces super-user **device** activity off for that connection — char/block device-node creation (`--devices`) and `--write-devices` — even under the default `AUTO` mode, so a non-opted module can never be made to `mknod` or write a raw device; those entries are skipped (not refused) so an ordinary `-a` push still succeeds without device nodes. The opt-in does **not** lift the privilege requirement: `--copy-as` still needs a root receiver, and the operator `--no-super` veto still forces super-user activities off for every connection. The daemon logs a prominent startup warning for each `client owner = yes` module so the operator's deliberate choice is visible. - **`read only` safe default:** every network transfer FastSync currently supports is a push that writes under the module root, so a `read only` module refuses the connection (clear server log "module is read only"; the client exits non-zero, nothing is transferred). A future pull/list operation can be opened up when it exists; the knob is already stored. From 87f6cb02431d6682bc9592a90d458eee58625594 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:28:26 +0200 Subject: [PATCH 042/155] refactor(config): single X-macro table for serialized fields Every Config field that crosses the wire was declared in up to six places (struct member, config_set_defaults, send_*, receive_*, and the two CLI option tables) and could drift silently. Add CONFIG_WIRE_FIELDS in config.h: one ordered per-segment table where each serialized field is declared once with its C type, default and wire codec (KIND). config.h now expands the table to declare the struct members; config_set_defaults() expands it to assign the defaults; and config_send_wire_block()/config_receive() expand the per-segment lists to emit/consume the frame. The per-segment function names, call order and segment boundaries are preserved exactly. Fields with genuinely custom logic keep dedicated helpers but are still declared once in the table: the protocol-version handshake (HEADER), daemon SCRAM auth (STR_REDACTED_AUTH), the daemon module name (STR_MODULE), the repeated count+array blocks (BLOCK_SKIP_SUFFIXES/BLOCK_BASIS/BLOCK_IDMAP), --copy-as presence/ids (COPY_AS_*), and the derived --delta / use_xattrs bits (DERIVED_DELTA, BOOL_XATTR_DERIVE). The version field remains a special header (validated before any other field is parsed) and is sent by config_send_wire_block() explicitly. No public field is renamed and PROTOCOL_VERSION stays "2.20.0". Because the struct declaration order is no longer the wire order, the wire order is now enforced solely by the table and by a byte-exact golden test (follow-up commit). Add config_send_wire_block() so that test can hash the frame body without the STATUS_OK handshake. --- src/shared/config.c | 966 +++++++++++++++++--------------------------- src/shared/config.h | 635 +++++++++++++++++------------ 2 files changed, 735 insertions(+), 866 deletions(-) diff --git a/src/shared/config.c b/src/shared/config.c index 64adf10..24ccc1f 100644 --- a/src/shared/config.c +++ b/src/shared/config.c @@ -18,33 +18,16 @@ #include static void config_set_defaults(Config* config) { - config->version = str_dup(PROTOCOL_VERSION); - config->send_directory = NULL; - config->receive_root_directory = NULL; - config->save_to_disk = false; - config->use_multithreading = false; config->scanner_threads = 0; - config->use_chunk_serialization = false; - config->use_compression = false; - config->use_metadata = false; - config->use_executability = false; config->metadata_explicitly_disabled = false; config->show_progress = false; config->dry_run = false; - config->remove_source_files = false; - config->use_delete = false; - config->compression_level = 5; config->compression_threads = 0; - config->use_sendfile = false; - config->chunk_size = DEFAULT_CHUNK_SIZE; config->ssh_port = 22; config->transport = TRANSPORT_TCP; config->ssh_destination = NULL; - config->module = NULL; - config->auth_user = NULL; config->auth_password = NULL; config->password_file = NULL; - config->iconv_spec = NULL; config->fastsync_server_path = NULL; config->exclude_patterns = NULL; config->exclude_count = 0; @@ -52,16 +35,7 @@ static void config_set_defaults(Config* config) { config->include_count = 0; config->max_size = 0; config->min_size = 0; - config->max_alloc = DEFAULT_MAX_ALLOC; - config->use_incremental = false; - config->ignore_times = false; - config->size_only = false; - config->use_delta = false; config->whole_file = false; - config->fuzzy = false; - config->modify_window = 0; - config->delta_block_size = DELTA_BLOCK_SIZE_DEFAULT; - config->delta_max_file_size = DELTA_MAX_FILE_SIZE; config->use_tls = false; config->tls_cert = NULL; config->tls_key = NULL; @@ -75,27 +49,10 @@ static void config_set_defaults(Config* config) { config->timeout = 0; config->contimeout = 10; config->quiet = false; - config->backup = false; - config->backup_dir = NULL; config->stats = false; config->max_depth = 0; config->log_file = NULL; - config->follow_symlinks = false; - config->partial = false; - config->copy_links = false; - config->safe_links = false; - config->copy_unsafe_links = false; config->copy_dirlinks = false; - config->munge_links = false; - config->keep_dirlinks = false; - config->preserve_hard_links = false; - config->preserve_acls = false; - config->preserve_xattrs = false; - config->preserve_devices = false; - config->preserve_sparse = false; - config->preserve_specials = false; - config->copy_devices = false; - config->write_devices = false; config->itemize_changes = false; config->out_format = NULL; config->log_file_format = NULL; @@ -103,49 +60,23 @@ static void config_set_defaults(Config* config) { config->debug_level = 0; config->list_only = false; config->human_readable = false; - config->eight_bit_output = false; - config->existing = false; - config->ignore_existing = false; - config->update = false; - config->inplace = false; - config->delay_updates = false; - config->use_fsync = false; - config->append = false; - config->append_verify = false; - config->preallocate = false; - config->delete_excluded = false; - config->delete_after = false; - config->max_delete = -1; config->ignore_errors = false; - config->force_delete = false; config->ignore_missing_args = false; - config->delete_missing_args = false; config->filters = NULL; config->files_from = NULL; config->files_from_set = NULL; config->from0 = false; config->cvs_exclude = false; config->per_dir_filter = false; - config->prune_empty_dirs = false; config->one_file_system = false; - config->relative = false; config->no_implied_dirs = false; config->dirs = false; - config->mkpath = false; config->rsh_command = NULL; config->blocking_io = false; config->outbuf = OUTBUF_BLOCK; config->old_args = false; - config->temp_dir = NULL; config->remote_options = NULL; config->remote_option_count = 0; - config->basis_dirs = NULL; - config->basis_count = 0; - config->partial_dir = NULL; - config->suffix = NULL; - config->delete_before = false; - config->delete_during = false; - config->delete_delay = false; config->address = NULL; config->ipv6 = false; config->ipv4 = false; @@ -153,35 +84,9 @@ static void config_set_defaults(Config* config) { config->sockopt_count = 0; config->daemon = false; config->no_motd = false; - config->checksum = false; - config->checksum_algo = CHECKSUM_ALGO_XXH64; - config->checksum_seed = 0; - config->compress_choice = NULL; - config->chmod_spec = NULL; - config->skip_compress_suffixes = NULL; - config->skip_compress_count = 0; - config->skip_compress_set = false; - config->numeric_ids = false; - config->chown_uid_set = false; - config->chown_uid = 0; - config->chown_gid_set = false; - config->chown_gid = 0; - config->usermap = NULL; - config->usermap_count = 0; - config->groupmap = NULL; - config->groupmap_count = 0; - config->super_mode = SUPER_MODE_AUTO; config->delay_context = NULL; - config->preserve_atimes = false; - config->preserve_crtimes = false; - config->omit_dir_times = false; - config->omit_link_times = false; config->open_noatime = false; config->use_xattrs = false; - config->fake_super = false; - config->copy_as_set = false; - config->copy_as_uid = 0; - config->copy_as_gid = 0; config->trust_sender = false; config->stop_after_mins = 0; config->stop_at = 0; @@ -189,6 +94,12 @@ static void config_set_defaults(Config* config) { config->write_batch = NULL; config->only_write_batch = NULL; config->read_batch = NULL; + + /* Serialized fields: defaults come from the CONFIG_WIRE_FIELDS table so the + * member declaration, default and wire codec can never drift apart. */ +#define CONFIG_DEFAULT_FIELD(name, ctype, def, kind) config->name = def; + CONFIG_WIRE_FIELDS(CONFIG_DEFAULT_FIELD) +#undef CONFIG_DEFAULT_FIELD } static bool valid_wire_bool(int value) { @@ -812,408 +723,52 @@ void config_delete(Config* config) { free(config); } -/* Each helper is deliberately ordered to match the wire format. Keep the - * helper call order in config_send and config_receive unchanged when adding - * fields. */ -static bool send_core_fields(int fd, const Config* c) { - if (!send_str(fd, c->version) || !send_int(fd, c->eight_bit_output)) +/* --------------------------------------------------------------------------- + * Wire codec helpers. + * + * The CONFIG_WIRE_*_FIELDS tables in config.h drive the send/receive + * sequences below. Each field's KIND names a CONFIG_SEND_ / + * CONFIG_RECV_ macro (defined after the helpers) that expands to the + * exact primitive call the previous hand-written code used, so the byte + * stream is unchanged. Fields whose per-field logic is not a plain scalar + * (bounded enums, redacted auth, repeated count+array blocks) delegate to a + * dedicated helper here. + * ------------------------------------------------------------------------- */ + +/* --max-alloc: raw 64-bit value, clamped server-side and installed as the + * session allocation ceiling. A zero value is rejected. */ +static bool config_receive_max_alloc(int fd, unsigned long long* value) { + if (!receive_n_data(fd, value, sizeof(*value)) || *value == 0) return false; - protocol_set_8_bit_output(c->eight_bit_output); - if (!send_n_data(fd, &c->max_alloc, sizeof(c->max_alloc))) - return false; - return send_str(fd, c->send_directory) && send_str(fd, c->receive_root_directory) && - send_int(fd, c->save_to_disk) && send_int(fd, c->use_multithreading) && - send_int(fd, c->use_chunk_serialization) && send_int(fd, c->use_compression) && - send_int(fd, c->use_metadata) && send_int(fd, c->use_executability) && - send_int(fd, c->compression_level) && - send_n_data(fd, &c->chunk_size, sizeof(c->chunk_size)) && send_int(fd, c->use_sendfile); -} - -static bool send_delta_fields(int fd, const Config* c) { - return send_int(fd, c->use_delete) && send_int(fd, c->use_incremental) && - send_int(fd, c->size_only) && send_int(fd, c->ignore_times) && - send_int(fd, c->use_delta && !c->whole_file) && - send_n_data(fd, &c->delta_block_size, sizeof(c->delta_block_size)) && - send_n_data(fd, &c->delta_max_file_size, sizeof(unsigned long long)); -} - -static bool send_file_options(int fd, const Config* c) { - /* Device/special preservation flags cross the wire so the receiver knows a - * special/device entry must be recreated. Trailing fields; protocol 2.13.0. */ - return send_int(fd, c->backup) && send_str(fd, c->backup_dir ? c->backup_dir : "") && - send_int(fd, c->remove_source_files) && send_int(fd, c->follow_symlinks) && - send_int(fd, c->copy_links) && send_int(fd, c->safe_links) && - send_int(fd, c->copy_unsafe_links) && send_int(fd, c->preserve_hard_links) && - send_int(fd, c->preserve_acls) && send_int(fd, c->preserve_xattrs) && - send_int(fd, c->preserve_devices) && send_int(fd, c->preserve_sparse) && - send_int(fd, c->preserve_specials) && send_int(fd, c->copy_devices) && - send_int(fd, c->write_devices); -} - -static bool send_selection_options(int fd, const Config* c) { - return send_int(fd, c->ignore_existing) && send_int(fd, c->existing) && send_int(fd, c->update) && - send_int(fd, c->inplace) && send_int(fd, c->delay_updates) && send_int(fd, c->append) && - send_int(fd, c->use_fsync) && send_int(fd, c->append_verify) && - send_int(fd, c->delete_excluded) && send_int(fd, c->force_delete) && - send_int(fd, c->delete_missing_args) && send_int(fd, c->delete_after) && - send_int(fd, c->preallocate) && send_n_data(fd, &c->max_delete, sizeof(c->max_delete)) && - send_int(fd, c->relative) && send_int(fd, c->prune_empty_dirs) && - send_int(fd, c->mkpath) && send_int(fd, c->delete_during) && send_int(fd, c->delete_delay); -} - -static bool send_skip_compress_options(int fd, const Config* c) { - if (!send_int(fd, c->skip_compress_set) || !send_int(fd, c->skip_compress_count)) - return false; - for (int i = 0; i < c->skip_compress_count; i++) { - if (!send_str(fd, c->skip_compress_suffixes[i])) - return false; - } + if (*value > MAX_SERVER_ALLOC) + *value = MAX_SERVER_ALLOC; + protocol_session_set_max_alloc(NULL, *value); return true; } -static bool send_resume_options(int fd, const Config* c) { - return send_str(fd, c->temp_dir ? c->temp_dir : "") && send_int(fd, c->partial) && - send_str(fd, c->partial_dir ? c->partial_dir : "") && - send_str(fd, c->suffix ? c->suffix : "") && send_int(fd, c->delete_before) && - send_int(fd, c->checksum) && send_int(fd, c->modify_window) && - send_str(fd, c->compress_choice ? c->compress_choice : "") && - send_str(fd, c->chmod_spec ? c->chmod_spec : "") && send_skip_compress_options(fd, c); -} - -static bool send_basis_options(int fd, const Config* c) { - if (!send_int(fd, c->basis_count)) +/* Optional string: the sender serializes an unset (NULL) string as "", so the + * receiver canonicalizes the empty wire value back to NULL to preserve + * NULL-vs-empty semantics. */ +static bool config_receive_optional_str(int fd, ConfigStringBudget* budget, char** out) { + char* value = config_receive_str(fd, budget); + if (!value) return false; - for (int i = 0; i < c->basis_count; i++) { - if (!send_int(fd, (int)c->basis_dirs[i].type) || - !send_str(fd, c->basis_dirs[i].path ? c->basis_dirs[i].path : "")) - return false; + if (*value == '\0') { + free(value); + *out = NULL; + return true; } + *out = value; return true; } -/* -y/--fuzzy (receiver-side similar-file basis selection). Trailing field on - * the config frame; protocol 2.9.0. */ -static bool send_fuzzy_option(int fd, const Config* c) { - return send_int(fd, c->fuzzy); -} - -/* --checksum-choice/--cc + --checksum-seed. The algorithm id and seed travel - * with the config so the receiver hashes the on-disk old file with the same - * parameters the sender used for its digest (see checksum.h). Trailing fields - * on the config frame; protocol 2.10.0. */ -static bool send_checksum_options(int fd, const Config* c) { - return send_int(fd, c->checksum_algo) && - send_n_data(fd, &c->checksum_seed, sizeof(c->checksum_seed)); -} - -static bool receive_core_fields(int fd, Config* c, ConfigStringBudget* budget) { - int value; - if (!receive_wire_bool(fd, &c->eight_bit_output)) - return false; - protocol_set_8_bit_output(c->eight_bit_output); - if (!receive_n_data(fd, &c->max_alloc, sizeof(c->max_alloc)) || c->max_alloc == 0) - return false; - if (c->max_alloc > MAX_SERVER_ALLOC) - c->max_alloc = MAX_SERVER_ALLOC; - protocol_session_set_max_alloc(NULL, c->max_alloc); - c->send_directory = config_receive_str(fd, budget); - c->receive_root_directory = config_receive_str(fd, budget); - if (!c->send_directory || !c->receive_root_directory) - return false; - if (!receive_wire_bool(fd, &c->save_to_disk) || !receive_wire_bool(fd, &c->use_multithreading) || - !receive_wire_bool(fd, &c->use_chunk_serialization) || - !receive_wire_bool(fd, &c->use_compression) || !receive_wire_bool(fd, &c->use_metadata) || - !receive_wire_bool(fd, &c->use_executability)) - return false; - if (!receive_int(fd, &value)) - return false; - c->compression_level = value; - if (!receive_n_data(fd, &c->chunk_size, sizeof(c->chunk_size))) - return false; - if (!receive_wire_bool(fd, &c->use_sendfile)) - return false; - return true; -} - -static bool receive_delta_fields(int fd, Config* c) { - if (!receive_wire_bool(fd, &c->use_delete)) - return false; - if (!receive_wire_bool(fd, &c->use_incremental)) - return false; - if (!receive_wire_bool(fd, &c->size_only)) - return false; - if (!receive_wire_bool(fd, &c->ignore_times)) - return false; - if (!receive_wire_bool(fd, &c->use_delta)) - return false; - return receive_n_data(fd, &c->delta_block_size, sizeof(c->delta_block_size)) && - receive_n_data(fd, &c->delta_max_file_size, sizeof(unsigned long long)); -} - -static bool receive_file_options(int fd, Config* c, ConfigStringBudget* budget) { - if (!receive_wire_bool(fd, &c->backup)) - return false; - char* backup_dir = config_receive_str(fd, budget); - if (!backup_dir) - return false; - if (*backup_dir != '\0') { - c->backup_dir = backup_dir; - } else { - /* The sender serializes an unset (NULL) string as "", so canonicalize the - empty wire value back to NULL to preserve NULL-vs-empty semantics. */ - free(backup_dir); - } - if (!receive_wire_bool(fd, &c->remove_source_files)) - return false; - bool* flags[] = {&c->follow_symlinks, &c->copy_links, &c->safe_links, - &c->copy_unsafe_links, &c->preserve_hard_links, &c->preserve_acls, - &c->preserve_xattrs, &c->preserve_devices, &c->preserve_sparse, - &c->preserve_specials, &c->copy_devices, &c->write_devices}; - for (size_t i = 0; i < sizeof(flags) / sizeof(flags[0]); i++) { - if (!receive_wire_bool(fd, flags[i])) - return false; - } - return true; -} - -static bool receive_selection_options(int fd, Config* c) { - bool* flags[] = {&c->ignore_existing, - &c->existing, - &c->update, - &c->inplace, - &c->delay_updates, - &c->append, - &c->use_fsync, - &c->append_verify, - &c->delete_excluded, - &c->force_delete, - &c->delete_missing_args, - &c->delete_after, - &c->preallocate}; - for (size_t i = 0; i < sizeof(flags) / sizeof(flags[0]); i++) { - if (!receive_wire_bool(fd, flags[i])) - return false; - } - if (!receive_n_data(fd, &c->max_delete, sizeof(c->max_delete))) - return false; - if (!receive_wire_bool(fd, &c->relative)) - return false; - if (!receive_wire_bool(fd, &c->prune_empty_dirs)) - return false; - if (!receive_wire_bool(fd, &c->mkpath)) - return false; - if (!receive_wire_bool(fd, &c->delete_during)) - return false; - return receive_wire_bool(fd, &c->delete_delay); -} - -static bool receive_resume_options(int fd, Config* c, ConfigStringBudget* budget) { - char* temp_dir = config_receive_str(fd, budget); - if (!temp_dir) - return false; - if (*temp_dir != '\0') { - c->temp_dir = temp_dir; - } else { - free(temp_dir); - } - if (!receive_wire_bool(fd, &c->partial)) - return false; - /* These options have NULL client defaults, so the sender transmits an empty - string for "unset". Canonicalize the empty wire value back to NULL so - receivers observe exactly what the client configured (plain --backup, for - example, must not look like --backup-dir ""). */ - char* partial_dir = config_receive_str(fd, budget); - if (!partial_dir) - return false; - if (*partial_dir != '\0') { - c->partial_dir = partial_dir; - } else { - free(partial_dir); - } - char* suffix = config_receive_str(fd, budget); - if (!suffix) - return false; - if (*suffix != '\0') { - c->suffix = suffix; - } else { - free(suffix); - } - if (!receive_wire_bool(fd, &c->delete_before)) - return false; - if (!receive_wire_bool(fd, &c->checksum)) - return false; - if (!receive_n_data(fd, &c->modify_window, sizeof(c->modify_window))) - return false; - c->compress_choice = config_receive_str(fd, budget); - if (!c->compress_choice) - return false; - c->chmod_spec = config_receive_str(fd, budget); - if (!c->chmod_spec || !receive_wire_bool(fd, &c->skip_compress_set) || - !receive_int(fd, &c->skip_compress_count) || c->skip_compress_count < 0 || - c->skip_compress_count > MAX_SKIP_COMPRESS_SUFFIXES) - return false; - if (c->skip_compress_count > 0) { - c->skip_compress_suffixes = calloc((size_t)c->skip_compress_count, sizeof(char*)); - if (!c->skip_compress_suffixes) - return false; - for (int i = 0; i < c->skip_compress_count; i++) { - c->skip_compress_suffixes[i] = config_receive_str(fd, budget); - if (!c->skip_compress_suffixes[i]) - return false; - } - } - return true; -} - -static bool receive_basis_options(int fd, Config* c, ConfigStringBudget* budget) { - int count; - if (!receive_int(fd, &count)) - return false; - if (count < 0 || count > MAX_BASIS_DIRS) - return false; - for (int i = 0; i < count; i++) { - int type; - if (!receive_int(fd, &type) || type <= BASIS_DEST_NONE || type > BASIS_DEST_LINK) - return false; - char* path = config_receive_str(fd, budget); - if (!path) - return false; - /* config_basis_append validates and canonicalizes the path; a rejected - path (absolute / traversal / empty) drops the whole connection. */ - bool ok = config_basis_append(c, (BasisDestType)type, path) == 0; - free(path); - if (!ok) - return false; - } - return true; -} - -static bool receive_fuzzy_option(int fd, Config* c) { - return receive_wire_bool(fd, &c->fuzzy); -} - -static bool receive_checksum_options(int fd, Config* c) { - int algo; - if (!receive_int(fd, &algo) || !checksum_algo_valid(algo)) - return false; - c->checksum_algo = algo; - return receive_n_data(fd, &c->checksum_seed, sizeof(c->checksum_seed)); -} - -/* --numeric-ids / --usermap / --groupmap / --chown (identity mapping). The - * receiver needs these to apply the ownership the client requested, so they - * cross the config frame. Trailing fields; protocol 2.11.0. */ -static bool send_identity_map(int fd, const IdentityMap* map, int count) { - if (!send_int(fd, count)) - return false; - for (int i = 0; i < count; i++) { - if (!send_int(fd, map[i].from) || !send_int(fd, map[i].to)) - return false; - } - return true; -} - -static bool send_identity_options(int fd, const Config* c) { - return send_int(fd, c->numeric_ids) && send_int(fd, c->chown_uid_set) && - send_int(fd, c->chown_uid) && send_int(fd, c->chown_gid_set) && - send_int(fd, c->chown_gid) && send_identity_map(fd, c->usermap, c->usermap_count) && - send_identity_map(fd, c->groupmap, c->groupmap_count); -} - -static bool receive_identity_map(int fd, int* pcount, IdentityMap** pmap) { - int count; - if (!receive_int(fd, &count) || count < 0 || count > MAX_IDENTITY_MAP) - return false; - if (count > 0) { - IdentityMap* map = calloc((size_t)count, sizeof(IdentityMap)); - if (!map) - return false; - for (int i = 0; i < count; i++) { - if (!receive_int(fd, &map[i].from) || !receive_int(fd, &map[i].to)) { - free(map); - return false; - } - } - *pmap = map; - } - *pcount = count; - return true; -} - -static bool receive_identity_options(int fd, Config* c) { - int numeric_ids; - if (!receive_int(fd, &numeric_ids) || !valid_wire_bool(numeric_ids)) - return false; - c->numeric_ids = numeric_ids != 0; - if (!receive_wire_bool(fd, &c->chown_uid_set) || !receive_int(fd, &c->chown_uid) || - !receive_wire_bool(fd, &c->chown_gid_set) || !receive_int(fd, &c->chown_gid)) - return false; - if (c->chown_uid < IDENTITY_MATCH_ANY || c->chown_gid < IDENTITY_MATCH_ANY) - return false; - return receive_identity_map(fd, &c->usermap_count, &c->usermap) && - receive_identity_map(fd, &c->groupmap_count, &c->groupmap); -} - -/* -U/--atimes, -N/--crtimes (affect both sender capture and receiver apply) - * and -O/--omit-dir-times, -J/--omit-link-times (receiver-side prefs) all cross - * the wire so the receiver knows what to apply / suppress. --open-noatime is - * client-only (it only governs the sender's source reads) and is never - * serialized. Trailing fields; protocol 2.12.0. */ -static bool send_metadata_times_options(int fd, const Config* c) { - return send_int(fd, c->preserve_atimes) && send_int(fd, c->preserve_crtimes) && - send_int(fd, c->omit_dir_times) && send_int(fd, c->omit_link_times); -} - -static bool receive_metadata_times_options(int fd, Config* c) { - return receive_wire_bool(fd, &c->preserve_atimes) && - receive_wire_bool(fd, &c->preserve_crtimes) && receive_wire_bool(fd, &c->omit_dir_times) && - receive_wire_bool(fd, &c->omit_link_times); -} - -/* Phase 4 symlink-trust: --munge-links and -K/--keep-dirlinks. Both CROSS the - * wire (the receiver unmunges symlink targets and, with -K, follows an in-root - * destination symlink-to-directory). -k/--copy-dirlinks is sender-only and is - * never serialized. Trailing fields; protocol 2.13.0. */ -static bool send_symlink_trust_options(int fd, const Config* c) { - return send_int(fd, c->munge_links) && send_int(fd, c->keep_dirlinks); -} - -static bool receive_symlink_trust_options(int fd, Config* c) { - return receive_wire_bool(fd, &c->munge_links) && receive_wire_bool(fd, &c->keep_dirlinks); -} - -/* -X/--xattrs, -A/--acls, --fake-super (Phase-4). The receiver learns - * preserve_xattrs/preserve_acls from the earlier file-options block and - * recomputes the derived use_xattrs there; only --fake-super (receiver-side - * behavior) needs an extra wire bit. Trailing field; protocol 2.13.0. */ -static bool send_phase4_xattr_options(int fd, const Config* c) { - return send_int(fd, c->fake_super); -} - -static bool receive_phase4_xattr_options(int fd, Config* c) { - if (!receive_wire_bool(fd, &c->fake_super)) - return false; - c->use_xattrs = c->preserve_acls || c->preserve_xattrs; - return true; -} - -/* Daemon module selection (Wave A, protocol 2.15.0). Trailing string on the - * config frame, sent after the Phase-4 xattr block and before the ack. The - * client composes it from a host::module/path destination; an unset module is - * serialized as "" and canonicalized back to NULL on receive so the two never - * look different to a peer. */ -static bool send_daemon_module(int fd, const Config* c) { - return send_str(fd, c->module ? c->module : ""); -} - -static bool receive_daemon_module(int fd, Config* c, ConfigStringBudget* budget) { +/* Daemon module name (Wave A, protocol 2.15.0): an unset module is "" (-> NULL + * on receive). A hostile over-long/invalid name is rejected with an explicit + * STATUS_ERROR rather than logged and accepted. */ +static bool config_receive_module(int fd, Config* c, ConfigStringBudget* budget) { char* module = config_receive_str(fd, budget); if (!module) return false; - /* Guard against a hostile client flooding the log with an over-long module - * name: only an empty string (module-less) or a valid module name - * (bounded by DAEMON_MAX_MODULE_NAME) is accepted. This is an input - * guard, not a wire-format change. */ if (*module != '\0' && !daemon_module_name_valid(module)) { log_message(LOG_LEVEL_WARNING, "Daemon client sent an invalid or over-long module name"); free(module); @@ -1228,27 +783,15 @@ static bool receive_daemon_module(int fd, Config* c, ConfigStringBudget* budget) return true; } -/* Daemon password credentials (A7 remediation, protocol 2.19.0). A single - * presence int is followed, when set, by ONLY the username; the password is - * never serialized. The daemon answers an auth-required module with the SCRAM - * challenge (see the auth exchange below). */ -static bool send_daemon_auth(int fd, const Config* c) { - bool present = c->auth_user != NULL && c->auth_user[0] != '\0'; - if (!send_int(fd, present ? 1 : 0)) - return false; - if (!present) - return true; - /* Redacted send: the username must never reach a --verbose debug log. */ - return send_str_redacted(fd, c->auth_user); -} - -static bool receive_daemon_auth(int fd, Config* c, ConfigStringBudget* budget) { +/* Daemon auth username (A7 remediation, protocol 2.19.0): a presence int is + * followed, when set, by ONLY the redacted username; the password is never + * serialized. */ +static bool config_receive_auth_user(int fd, Config* c, ConfigStringBudget* budget) { int present; if (!receive_int(fd, &present) || !valid_wire_bool(present)) return false; if (!present) return true; - /* Redacted receive: never log the incoming username body. */ char* user = config_receive_str_redacted(fd, budget); if (!user) return false; @@ -1261,6 +804,296 @@ static bool receive_daemon_auth(int fd, Config* c, ConfigStringBudget* budget) { return true; } +static bool config_send_auth_user(int fd, const Config* c) { + bool present = c->auth_user != NULL && c->auth_user[0] != '\0'; + if (!send_int(fd, present ? 1 : 0)) + return false; + if (!present) + return true; + /* Redacted send: the username must never reach a --verbose debug log. */ + return send_str_redacted(fd, c->auth_user); +} + +static bool config_receive_checksum_algo(int fd, int* value) { + int algo; + if (!receive_int(fd, &algo) || !checksum_algo_valid(algo)) + return false; + *value = algo; + return true; +} + +static bool config_receive_super_mode(int fd, SuperMode* value) { + int mode; + if (!receive_int(fd, &mode) || mode < SUPER_MODE_AUTO || mode > SUPER_MODE_OFF) + return false; + *value = (SuperMode)mode; + return true; +} + +/* chown override ids: IDENTITY_MATCH_ANY (-1) is the lowest legal value. */ +static bool config_receive_identity_id(int fd, int32_t* value) { + int v; + if (!receive_int(fd, &v) || v < IDENTITY_MATCH_ANY) + return false; + *value = v; + return true; +} + +static bool config_receive_skip_count(int fd, int* value) { + if (!receive_int(fd, value) || *value < 0 || *value > MAX_SKIP_COMPRESS_SUFFIXES) + return false; + return true; +} + +static bool config_receive_basis_count(int fd, int* value) { + if (!receive_int(fd, value) || *value < 0 || *value > MAX_BASIS_DIRS) + return false; + return true; +} + +static bool config_receive_idmap_count(int fd, int* value) { + if (!receive_int(fd, value) || *value < 0 || *value > MAX_IDENTITY_MAP) + return false; + return true; +} + +static bool config_receive_copy_as_presence(int fd, bool* value) { + int present; + if (!receive_int(fd, &present) || !valid_wire_bool(present)) + return false; + *value = present != 0; + return true; +} + +/* --copy-as ids are forced onto the ownership path, so a hostile peer must not + * smuggle a negative sentinel. */ +static bool config_receive_copy_as_id(int fd, int32_t* value) { + int v; + if (!receive_int(fd, &v) || v < 0) + return false; + *value = v; + return true; +} + +static bool send_skip_compress_suffixes(int fd, const Config* c) { + for (int i = 0; i < c->skip_compress_count; i++) { + if (!send_str(fd, c->skip_compress_suffixes[i])) + return false; + } + return true; +} + +static bool receive_skip_compress_suffixes(int fd, Config* c, ConfigStringBudget* budget) { + if (c->skip_compress_count <= 0) + return true; + c->skip_compress_suffixes = calloc((size_t)c->skip_compress_count, sizeof(char*)); + if (!c->skip_compress_suffixes) + return false; + for (int i = 0; i < c->skip_compress_count; i++) { + c->skip_compress_suffixes[i] = config_receive_str(fd, budget); + if (!c->skip_compress_suffixes[i]) + return false; + } + return true; +} + +static bool send_basis_entries(int fd, const Config* c) { + for (int i = 0; i < c->basis_count; i++) { + if (!send_int(fd, (int)c->basis_dirs[i].type) || + !send_str(fd, c->basis_dirs[i].path ? c->basis_dirs[i].path : "")) + return false; + } + return true; +} + +static bool receive_basis_entries(int fd, Config* c, ConfigStringBudget* budget) { + /* The count was read by the preceding INT_BASISCOUNT entry; config_basis_append + * rebuilds basis_count as it validates and canonicalizes each path. */ + int count = c->basis_count; + c->basis_count = 0; + for (int i = 0; i < count; i++) { + int type; + if (!receive_int(fd, &type) || type <= BASIS_DEST_NONE || type > BASIS_DEST_LINK) + return false; + char* path = config_receive_str(fd, budget); + if (!path) + return false; + bool ok = config_basis_append(c, (BasisDestType)type, path) == 0; + free(path); + if (!ok) + return false; + } + return true; +} + +static bool send_identity_entries(int fd, const IdentityMap* map, int count) { + for (int i = 0; i < count; i++) { + if (!send_int(fd, map[i].from) || !send_int(fd, map[i].to)) + return false; + } + return true; +} + +static bool receive_identity_entries(int fd, ConfigStringBudget* budget, int count, + IdentityMap** out) { + (void)budget; + if (count <= 0) + return true; + IdentityMap* map = calloc((size_t)count, sizeof(IdentityMap)); + if (!map) + return false; + for (int i = 0; i < count; i++) { + if (!receive_int(fd, &map[i].from) || !receive_int(fd, &map[i].to)) { + free(map); + return false; + } + } + *out = map; + return true; +} + +/* --------------------------------------------------------------------------- + * KIND dispatch. A table entry X(member, ctype, def, KIND) expands to + * CONFIG_SEND_(member) in a sender and CONFIG_RECV_(member) in a + * receiver. Send macros are bool expressions; receive macros are bool + * expressions too (strings allocate through `budget`). + * ------------------------------------------------------------------------- */ +#define CONFIG_SEND_BOOL(name) send_int(fd, c->name) +#define CONFIG_RECV_BOOL(name) receive_wire_bool(fd, &c->name) + +#define CONFIG_SEND_INT(name) send_int(fd, c->name) +#define CONFIG_RECV_INT(name) receive_int(fd, &c->name) + +#define CONFIG_SEND_RAW(name) send_n_data(fd, &c->name, sizeof(c->name)) +#define CONFIG_RECV_RAW(name) receive_n_data(fd, &c->name, sizeof(c->name)) + +#define CONFIG_SEND_BOOL_8BIT(name) \ + (send_int(fd, c->name) && (protocol_set_8_bit_output(c->name), true)) +#define CONFIG_RECV_BOOL_8BIT(name) \ + (receive_wire_bool(fd, &c->name) && (protocol_set_8_bit_output(c->name), true)) + +#define CONFIG_SEND_RAW_MAXALLOC(name) send_n_data(fd, &c->name, sizeof(c->name)) +#define CONFIG_RECV_RAW_MAXALLOC(name) config_receive_max_alloc(fd, &c->name) + +/* --delta is sent as (use_delta && !whole_file); whole_file never crosses the + * wire, so the receiver observes the effective bit. */ +#define CONFIG_SEND_DERIVED_DELTA(name) send_int(fd, c->name && !c->whole_file) +#define CONFIG_RECV_DERIVED_DELTA(name) receive_wire_bool(fd, &c->name) + +#define CONFIG_SEND_STR(name) send_str(fd, c->name) +#define CONFIG_RECV_STR(name) ((c->name = config_receive_str(fd, budget)) != NULL) + +#define CONFIG_SEND_STR_OPT(name) send_str(fd, c->name ? c->name : "") +#define CONFIG_RECV_STR_OPT(name) config_receive_optional_str(fd, budget, &c->name) + +#define CONFIG_SEND_STR_KEEP(name) send_str(fd, c->name ? c->name : "") +#define CONFIG_RECV_STR_KEEP(name) ((c->name = config_receive_str(fd, budget)) != NULL) + +#define CONFIG_SEND_STR_MODULE(name) send_str(fd, c->name ? c->name : "") +#define CONFIG_RECV_STR_MODULE(name) config_receive_module(fd, c, budget) + +#define CONFIG_SEND_STR_REDACTED_AUTH(name) config_send_auth_user(fd, c) +#define CONFIG_RECV_STR_REDACTED_AUTH(name) config_receive_auth_user(fd, c, budget) + +#define CONFIG_SEND_INT_CHECKSUM_ALGO(name) send_int(fd, c->name) +#define CONFIG_RECV_INT_CHECKSUM_ALGO(name) config_receive_checksum_algo(fd, &c->name) + +#define CONFIG_SEND_SUPERMODE(name) send_int(fd, (int)c->name) +#define CONFIG_RECV_SUPERMODE(name) config_receive_super_mode(fd, &c->name) + +#define CONFIG_SEND_INT_IDENTITY(name) send_int(fd, c->name) +#define CONFIG_RECV_INT_IDENTITY(name) config_receive_identity_id(fd, &c->name) + +#define CONFIG_SEND_INT_SKIPCOUNT(name) send_int(fd, c->name) +#define CONFIG_RECV_INT_SKIPCOUNT(name) config_receive_skip_count(fd, &c->name) + +#define CONFIG_SEND_INT_BASISCOUNT(name) send_int(fd, c->name) +#define CONFIG_RECV_INT_BASISCOUNT(name) config_receive_basis_count(fd, &c->name) + +#define CONFIG_SEND_INT_IDMAPCOUNT(name) send_int(fd, c->name) +#define CONFIG_RECV_INT_IDMAPCOUNT(name) config_receive_idmap_count(fd, &c->name) + +/* use_xattrs is derived receiver-side from the xattr/acl preservation flags + * that crossed the wire in the file-options block. */ +#define CONFIG_SEND_BOOL_XATTR_DERIVE(name) send_int(fd, c->name) +#define CONFIG_RECV_BOOL_XATTR_DERIVE(name) \ + (receive_wire_bool(fd, &c->name) && \ + (c->use_xattrs = (c->preserve_acls || c->preserve_xattrs), true)) + +#define CONFIG_SEND_COPY_AS_PRESENCE(name) send_int(fd, c->name ? 1 : 0) +#define CONFIG_RECV_COPY_AS_PRESENCE(name) config_receive_copy_as_presence(fd, &c->name) + +/* The uid/gid follow the presence int only when --copy-as is set. */ +#define CONFIG_SEND_COPY_AS_ID(name) (!c->copy_as_set || send_int(fd, c->name)) +#define CONFIG_RECV_COPY_AS_ID(name) (!c->copy_as_set || config_receive_copy_as_id(fd, &c->name)) + +#define CONFIG_SEND_BLOCK_SKIP_SUFFIXES(name) send_skip_compress_suffixes(fd, c) +#define CONFIG_RECV_BLOCK_SKIP_SUFFIXES(name) receive_skip_compress_suffixes(fd, c, budget) + +#define CONFIG_SEND_BLOCK_BASIS(name) send_basis_entries(fd, c) +#define CONFIG_RECV_BLOCK_BASIS(name) receive_basis_entries(fd, c, budget) + +#define CONFIG_SEND_BLOCK_IDMAP(name) send_identity_entries(fd, c->name, c->name##_count) +#define CONFIG_RECV_BLOCK_IDMAP(name) \ + receive_identity_entries(fd, budget, c->name##_count, &c->name) + +/* One table entry, applied in sequence. XSEND/XRECV are statement macros so + * consecutive entries read as a plain sequence of assignments. */ +#define XSEND(name, ctype, def, kind) ok = ok && (CONFIG_SEND_##kind(name)); +#define XRECV(name, ctype, def, kind) ok = ok && (CONFIG_RECV_##kind(name)); + +#define CONFIG_DEFINE_SEND(fn, fields) \ + static bool fn(int fd, const Config* c) { \ + bool ok = true; \ + fields(XSEND) return ok; \ + } + +#define CONFIG_DEFINE_RECV(fn, fields) \ + static bool fn(int fd, Config* c, ConfigStringBudget* budget) { \ + (void)budget; \ + bool ok = true; \ + fields(XRECV) return ok; \ + } + +CONFIG_DEFINE_SEND(send_core_fields, CONFIG_WIRE_CORE_FIELDS) +CONFIG_DEFINE_SEND(send_delta_fields, CONFIG_WIRE_DELTA_FIELDS) +CONFIG_DEFINE_SEND(send_file_options, CONFIG_WIRE_FILE_OPTIONS_FIELDS) +CONFIG_DEFINE_SEND(send_selection_options, CONFIG_WIRE_SELECTION_FIELDS) +CONFIG_DEFINE_SEND(send_resume_options, CONFIG_WIRE_RESUME_FIELDS) +CONFIG_DEFINE_SEND(send_basis_options, CONFIG_WIRE_BASIS_FIELDS) +CONFIG_DEFINE_SEND(send_fuzzy_option, CONFIG_WIRE_FUZZY_FIELDS) +CONFIG_DEFINE_SEND(send_checksum_options, CONFIG_WIRE_CHECKSUM_FIELDS) +CONFIG_DEFINE_SEND(send_identity_options, CONFIG_WIRE_IDENTITY_FIELDS) +CONFIG_DEFINE_SEND(send_metadata_times_options, CONFIG_WIRE_METADATA_TIMES_FIELDS) +CONFIG_DEFINE_SEND(send_symlink_trust_options, CONFIG_WIRE_SYMLINK_TRUST_FIELDS) +CONFIG_DEFINE_SEND(send_phase4_xattr_options, CONFIG_WIRE_XATTR_FIELDS) +CONFIG_DEFINE_SEND(send_daemon_module, CONFIG_WIRE_MODULE_FIELDS) +CONFIG_DEFINE_SEND(send_daemon_auth, CONFIG_WIRE_DAEMON_AUTH_FIELDS) +CONFIG_DEFINE_SEND(send_iconv_spec, CONFIG_WIRE_ICONV_FIELDS) +CONFIG_DEFINE_SEND(send_privilege_options, CONFIG_WIRE_PRIVILEGE_FIELDS) +CONFIG_DEFINE_SEND(send_copy_as_options, CONFIG_WIRE_COPY_AS_FIELDS) + +CONFIG_DEFINE_RECV(receive_core_fields, CONFIG_WIRE_CORE_FIELDS) +CONFIG_DEFINE_RECV(receive_delta_fields, CONFIG_WIRE_DELTA_FIELDS) +CONFIG_DEFINE_RECV(receive_file_options, CONFIG_WIRE_FILE_OPTIONS_FIELDS) +CONFIG_DEFINE_RECV(receive_selection_options, CONFIG_WIRE_SELECTION_FIELDS) +CONFIG_DEFINE_RECV(receive_resume_options, CONFIG_WIRE_RESUME_FIELDS) +CONFIG_DEFINE_RECV(receive_basis_options, CONFIG_WIRE_BASIS_FIELDS) +CONFIG_DEFINE_RECV(receive_fuzzy_option, CONFIG_WIRE_FUZZY_FIELDS) +CONFIG_DEFINE_RECV(receive_checksum_options, CONFIG_WIRE_CHECKSUM_FIELDS) +CONFIG_DEFINE_RECV(receive_identity_options, CONFIG_WIRE_IDENTITY_FIELDS) +CONFIG_DEFINE_RECV(receive_metadata_times_options, CONFIG_WIRE_METADATA_TIMES_FIELDS) +CONFIG_DEFINE_RECV(receive_symlink_trust_options, CONFIG_WIRE_SYMLINK_TRUST_FIELDS) +CONFIG_DEFINE_RECV(receive_phase4_xattr_options, CONFIG_WIRE_XATTR_FIELDS) +CONFIG_DEFINE_RECV(receive_daemon_module, CONFIG_WIRE_MODULE_FIELDS) +CONFIG_DEFINE_RECV(receive_daemon_auth, CONFIG_WIRE_DAEMON_AUTH_FIELDS) +CONFIG_DEFINE_RECV(receive_iconv_spec, CONFIG_WIRE_ICONV_FIELDS) +CONFIG_DEFINE_RECV(receive_privilege_options, CONFIG_WIRE_PRIVILEGE_FIELDS) +CONFIG_DEFINE_RECV(receive_copy_as_options, CONFIG_WIRE_COPY_AS_FIELDS) + +#undef XSEND +#undef XRECV + /* Client half of the SCRAM challenge/response (A7 remediation). Called by * config_send after the config frame is written and the server answered * STATUS_AUTH_CHALLENGE. The plaintext password lives only in @@ -1349,96 +1182,35 @@ static bool client_auth_exchange(int fd, const Config* c) { return ok; } -/* --iconv CONVERT_SPEC (protocol 2.16.0). Trailing string on the config frame, - * sent after the Wave A/B daemon-auth block and before the ack, so the - * receiver knows the wire charset before the first file name arrives. The full - * spec travels (LOCAL,REMOTE) and each end derives its own LOCAL and the wire - * (REMOTE) charset symmetrically; an unset spec is serialized as "" and - * canonicalized back to NULL on receive. */ -static bool send_iconv_spec(int fd, const Config* c) { - return send_str(fd, c->iconv_spec ? c->iconv_spec : ""); -} +/* The --iconv, --super/--no-super and --copy-as segment functions are + * generated above from CONFIG_WIRE_ICONV_FIELDS, CONFIG_WIRE_PRIVILEGE_FIELDS + * and CONFIG_WIRE_COPY_AS_FIELDS. */ -static bool receive_iconv_spec(int fd, Config* c, ConfigStringBudget* budget) { - char* spec = config_receive_str(fd, budget); - if (!spec) - return false; - if (*spec == '\0') { - free(spec); - c->iconv_spec = NULL; - return true; - } - c->iconv_spec = spec; - return true; -} - -/* --super / --no-super privilege policy (P7 Wave E, protocol 2.18.0). One - * trailing int on the config frame, sent after the --iconv spec and before the - * STATUS_OK ack, so the receiver knows whether it may attempt super-user - * activities (ownership application, char/block device-node creation) that are - * already confined below the authorized receive root. The received value is - * validated to the SUPER_MODE_AUTO..SUPER_MODE_OFF range (also re-checked by - * validate_received_config). */ -static bool send_privilege_options(int fd, const Config* c) { - return send_int(fd, (int)c->super_mode); -} - -static bool receive_privilege_options(int fd, Config* c) { - int mode; - if (!receive_int(fd, &mode) || mode < SUPER_MODE_AUTO || mode > SUPER_MODE_OFF) - return false; - c->super_mode = (SuperMode)mode; - return true; -} - -/* --copy-as=USER[:GROUP] (P7 Wave E, protocol 2.18.0). Trailing block on the - * config frame, sent after the --super int and before the ack: a presence int, - * then (when set) the target uid and gid as int32. The receiver forces the - * ownership of every entry it writes to these ids through the confined - * fd-relative identity path and requires privilege; both ids are validated - * `>= 0` on receive so a hostile peer cannot smuggle a negative (sentinel) - * value into the ownership path. */ -static bool send_copy_as_options(int fd, const Config* c) { - if (!send_int(fd, c->copy_as_set ? 1 : 0)) - return false; - if (!c->copy_as_set) - return true; - return send_int(fd, c->copy_as_uid) && send_int(fd, c->copy_as_gid); -} - -static bool receive_copy_as_options(int fd, Config* c) { - int present; - if (!receive_int(fd, &present) || !valid_wire_bool(present)) - return false; - if (!present) { - c->copy_as_set = false; - return true; - } - int uid, gid; - if (!receive_int(fd, &uid) || !receive_int(fd, &gid) || uid < 0 || gid < 0) - return false; - c->copy_as_set = true; - c->copy_as_uid = uid; - c->copy_as_gid = gid; - return true; +bool config_send_wire_block(int file_descriptor, const Config* config) { + protocol_session_set_max_alloc(NULL, config->max_alloc); + /* The version is the frame header: the receiver validates it before parsing + * any other field (see config_receive_with_validate), so it is not part of + * the generated segment sequence. It is still declared once, in + * CONFIG_WIRE_HEADER_FIELDS. */ + return send_str(file_descriptor, config->version) && send_core_fields(file_descriptor, config) && + send_delta_fields(file_descriptor, config) && send_file_options(file_descriptor, config) && + send_selection_options(file_descriptor, config) && + send_resume_options(file_descriptor, config) && + send_basis_options(file_descriptor, config) && + send_fuzzy_option(file_descriptor, config) && + send_checksum_options(file_descriptor, config) && + send_identity_options(file_descriptor, config) && + send_metadata_times_options(file_descriptor, config) && + send_symlink_trust_options(file_descriptor, config) && + send_phase4_xattr_options(file_descriptor, config) && + send_daemon_module(file_descriptor, config) && send_daemon_auth(file_descriptor, config) && + send_iconv_spec(file_descriptor, config) && + send_privilege_options(file_descriptor, config) && + send_copy_as_options(file_descriptor, config); } bool config_send(int file_descriptor, const Config* config) { - protocol_session_set_max_alloc(NULL, config->max_alloc); - if (!send_core_fields(file_descriptor, config) || !send_delta_fields(file_descriptor, config) || - !send_file_options(file_descriptor, config) || - !send_selection_options(file_descriptor, config) || - !send_resume_options(file_descriptor, config) || - !send_basis_options(file_descriptor, config) || !send_fuzzy_option(file_descriptor, config) || - !send_checksum_options(file_descriptor, config) || - !send_identity_options(file_descriptor, config) || - !send_metadata_times_options(file_descriptor, config) || - !send_symlink_trust_options(file_descriptor, config) || - !send_phase4_xattr_options(file_descriptor, config) || - !send_daemon_module(file_descriptor, config) || !send_daemon_auth(file_descriptor, config) || - !send_iconv_spec(file_descriptor, config) || - !send_privilege_options(file_descriptor, config) || - !send_copy_as_options(file_descriptor, config)) + if (!config_send_wire_block(file_descriptor, config)) return false; Status status; if (!receive_status(file_descriptor, &status)) @@ -1477,22 +1249,22 @@ Config* config_receive_with_validate(int file_descriptor, ConfigValidateFunc val goto error; } if (!receive_core_fields(file_descriptor, config, &budget) || - !receive_delta_fields(file_descriptor, config) || + !receive_delta_fields(file_descriptor, config, &budget) || !receive_file_options(file_descriptor, config, &budget) || - !receive_selection_options(file_descriptor, config) || + !receive_selection_options(file_descriptor, config, &budget) || !receive_resume_options(file_descriptor, config, &budget) || !receive_basis_options(file_descriptor, config, &budget) || - !receive_fuzzy_option(file_descriptor, config) || - !receive_checksum_options(file_descriptor, config) || - !receive_identity_options(file_descriptor, config) || - !receive_metadata_times_options(file_descriptor, config) || - !receive_symlink_trust_options(file_descriptor, config) || - !receive_phase4_xattr_options(file_descriptor, config) || + !receive_fuzzy_option(file_descriptor, config, &budget) || + !receive_checksum_options(file_descriptor, config, &budget) || + !receive_identity_options(file_descriptor, config, &budget) || + !receive_metadata_times_options(file_descriptor, config, &budget) || + !receive_symlink_trust_options(file_descriptor, config, &budget) || + !receive_phase4_xattr_options(file_descriptor, config, &budget) || !receive_daemon_module(file_descriptor, config, &budget) || !receive_daemon_auth(file_descriptor, config, &budget) || !receive_iconv_spec(file_descriptor, config, &budget) || - !receive_privilege_options(file_descriptor, config) || - !receive_copy_as_options(file_descriptor, config)) + !receive_privilege_options(file_descriptor, config, &budget) || + !receive_copy_as_options(file_descriptor, config, &budget)) goto error; if (config->compress_choice[0] != '\0' && strcmp(config->compress_choice, "zstd") != 0 && strcmp(config->compress_choice, "none") != 0) { diff --git a/src/shared/config.h b/src/shared/config.h index 0aba9d6..f7c5998 100644 --- a/src/shared/config.h +++ b/src/shared/config.h @@ -75,87 +75,203 @@ typedef struct { * privilege_super_mode_permitted() in identity.h. */ typedef enum SuperMode { SUPER_MODE_AUTO = 0, SUPER_MODE_ON = 1, SUPER_MODE_OFF = 2 } SuperMode; +/* =========================================================================== + * Config wire-field table (single source of truth for protocol 2.20.0). + * + * Every field below crosses the wire. The table is the ONLY place a + * serialized field is named: config.h expands CONFIG_WIRE_FIELDS() to declare + * the struct member, config_set_defaults() expands it to assign the default, + * and config_send_wire_block()/config_receive_with_validate() expand the + * per-segment lists to emit/consume the frame in exactly this order. Do NOT + * reorder entries and do NOT change a field's segment/KIND without a + * PROTOCOL_VERSION bump: the resulting byte stream is pinned by + * test_config_wire_golden(). + * + * Entry layout: X(MEMBER, CTYPE, DEFAULT, KIND) + * MEMBER struct member name (public; never rename) + * CTYPE C type of the member + * DEFAULT default-value expression used by config_set_defaults() + * KIND wire codec, dispatched to CONFIG_SEND_/CONFIG_RECV_ + * in config.c (strings receive through a ConfigStringBudget). + * + * Fields with genuinely custom logic keep dedicated helpers but are still + * declared here exactly once: the protocol-version handshake (HEADER), the + * daemon SCRAM auth username (STR_REDACTED_AUTH), the daemon module name + * (STR_MODULE), repeated count+array blocks (BLOCK_*), --copy-as presence + * (COPY_AS_*), and the derived --delta / use_xattrs bits (DERIVED_DELTA, + * BOOL_XATTR_DERIVE). + * =========================================================================== */ +#define CONFIG_WIRE_HEADER_FIELDS(X) X(version, char*, str_dup(PROTOCOL_VERSION), STR) + +#define CONFIG_WIRE_CORE_FIELDS(X) \ + X(eight_bit_output, bool, false, BOOL_8BIT) \ + X(max_alloc, unsigned long long, DEFAULT_MAX_ALLOC, RAW_MAXALLOC) \ + X(send_directory, char*, NULL, STR) \ + X(receive_root_directory, char*, NULL, STR) \ + X(save_to_disk, bool, false, BOOL) \ + X(use_multithreading, bool, false, BOOL) \ + X(use_chunk_serialization, bool, false, BOOL) \ + X(use_compression, bool, false, BOOL) \ + X(use_metadata, bool, false, BOOL) \ + X(use_executability, bool, false, BOOL) \ + X(compression_level, int, 5, INT) \ + X(chunk_size, unsigned long long, DEFAULT_CHUNK_SIZE, RAW) \ + X(use_sendfile, bool, false, BOOL) + +#define CONFIG_WIRE_DELTA_FIELDS(X) \ + X(use_delete, bool, false, BOOL) \ + X(use_incremental, bool, false, BOOL) \ + X(size_only, bool, false, BOOL) \ + X(ignore_times, bool, false, BOOL) \ + X(use_delta, bool, false, DERIVED_DELTA) \ + X(delta_block_size, uint32_t, DELTA_BLOCK_SIZE_DEFAULT, RAW) \ + X(delta_max_file_size, unsigned long long, DELTA_MAX_FILE_SIZE, RAW) + +#define CONFIG_WIRE_FILE_OPTIONS_FIELDS(X) \ + X(backup, bool, false, BOOL) \ + X(backup_dir, char*, NULL, STR_OPT) \ + X(remove_source_files, bool, false, BOOL) \ + X(follow_symlinks, bool, false, BOOL) \ + X(copy_links, bool, false, BOOL) \ + X(safe_links, bool, false, BOOL) \ + X(copy_unsafe_links, bool, false, BOOL) \ + X(preserve_hard_links, bool, false, BOOL) \ + X(preserve_acls, bool, false, BOOL) \ + X(preserve_xattrs, bool, false, BOOL) \ + X(preserve_devices, bool, false, BOOL) \ + X(preserve_sparse, bool, false, BOOL) \ + X(preserve_specials, bool, false, BOOL) \ + X(copy_devices, bool, false, BOOL) \ + X(write_devices, bool, false, BOOL) + +#define CONFIG_WIRE_SELECTION_FIELDS(X) \ + X(ignore_existing, bool, false, BOOL) \ + X(existing, bool, false, BOOL) \ + X(update, bool, false, BOOL) \ + X(inplace, bool, false, BOOL) \ + X(delay_updates, bool, false, BOOL) \ + X(append, bool, false, BOOL) \ + X(use_fsync, bool, false, BOOL) \ + X(append_verify, bool, false, BOOL) \ + X(delete_excluded, bool, false, BOOL) \ + X(force_delete, bool, false, BOOL) \ + X(delete_missing_args, bool, false, BOOL) \ + X(delete_after, bool, false, BOOL) \ + X(preallocate, bool, false, BOOL) \ + X(max_delete, int, -1, RAW) \ + X(relative, bool, false, BOOL) \ + X(prune_empty_dirs, bool, false, BOOL) \ + X(mkpath, bool, false, BOOL) \ + X(delete_during, bool, false, BOOL) \ + X(delete_delay, bool, false, BOOL) + +#define CONFIG_WIRE_RESUME_FIELDS(X) \ + X(temp_dir, char*, NULL, STR_OPT) \ + X(partial, bool, false, BOOL) \ + X(partial_dir, char*, NULL, STR_OPT) \ + X(suffix, char*, NULL, STR_OPT) \ + X(delete_before, bool, false, BOOL) \ + X(checksum, bool, false, BOOL) \ + X(modify_window, int, 0, RAW) \ + X(compress_choice, char*, NULL, STR_KEEP) \ + X(chmod_spec, char*, NULL, STR_KEEP) \ + X(skip_compress_set, bool, false, BOOL) \ + X(skip_compress_count, int, 0, INT_SKIPCOUNT) \ + X(skip_compress_suffixes, char**, NULL, BLOCK_SKIP_SUFFIXES) + +#define CONFIG_WIRE_BASIS_FIELDS(X) \ + X(basis_count, int, 0, INT_BASISCOUNT) \ + X(basis_dirs, BasisDest*, NULL, BLOCK_BASIS) + +#define CONFIG_WIRE_FUZZY_FIELDS(X) X(fuzzy, bool, false, BOOL) + +#define CONFIG_WIRE_CHECKSUM_FIELDS(X) \ + X(checksum_algo, int, CHECKSUM_ALGO_XXH64, INT_CHECKSUM_ALGO) \ + X(checksum_seed, uint64_t, 0, RAW) + +#define CONFIG_WIRE_IDENTITY_FIELDS(X) \ + X(numeric_ids, bool, false, BOOL) \ + X(chown_uid_set, bool, false, BOOL) \ + X(chown_uid, int32_t, 0, INT_IDENTITY) \ + X(chown_gid_set, bool, false, BOOL) \ + X(chown_gid, int32_t, 0, INT_IDENTITY) \ + X(usermap_count, int, 0, INT_IDMAPCOUNT) \ + X(usermap, IdentityMap*, NULL, BLOCK_IDMAP) \ + X(groupmap_count, int, 0, INT_IDMAPCOUNT) \ + X(groupmap, IdentityMap*, NULL, BLOCK_IDMAP) + +#define CONFIG_WIRE_METADATA_TIMES_FIELDS(X) \ + X(preserve_atimes, bool, false, BOOL) \ + X(preserve_crtimes, bool, false, BOOL) \ + X(omit_dir_times, bool, false, BOOL) \ + X(omit_link_times, bool, false, BOOL) + +#define CONFIG_WIRE_SYMLINK_TRUST_FIELDS(X) \ + X(munge_links, bool, false, BOOL) \ + X(keep_dirlinks, bool, false, BOOL) + +#define CONFIG_WIRE_XATTR_FIELDS(X) X(fake_super, bool, false, BOOL_XATTR_DERIVE) + +#define CONFIG_WIRE_MODULE_FIELDS(X) X(module, char*, NULL, STR_MODULE) + +#define CONFIG_WIRE_DAEMON_AUTH_FIELDS(X) X(auth_user, char*, NULL, STR_REDACTED_AUTH) + +#define CONFIG_WIRE_ICONV_FIELDS(X) X(iconv_spec, char*, NULL, STR_OPT) + +#define CONFIG_WIRE_PRIVILEGE_FIELDS(X) X(super_mode, SuperMode, SUPER_MODE_AUTO, SUPERMODE) + +#define CONFIG_WIRE_COPY_AS_FIELDS(X) \ + X(copy_as_set, bool, false, COPY_AS_PRESENCE) \ + X(copy_as_uid, int32_t, 0, COPY_AS_ID) \ + X(copy_as_gid, int32_t, 0, COPY_AS_ID) + +/* All serialized fields, in exact wire order. Concatenating the per-segment + * lists here is what keeps the declaration order = the wire order. */ +#define CONFIG_WIRE_FIELDS(X) \ + CONFIG_WIRE_HEADER_FIELDS(X) \ + CONFIG_WIRE_CORE_FIELDS(X) \ + CONFIG_WIRE_DELTA_FIELDS(X) \ + CONFIG_WIRE_FILE_OPTIONS_FIELDS(X) \ + CONFIG_WIRE_SELECTION_FIELDS(X) \ + CONFIG_WIRE_RESUME_FIELDS(X) \ + CONFIG_WIRE_BASIS_FIELDS(X) \ + CONFIG_WIRE_FUZZY_FIELDS(X) \ + CONFIG_WIRE_CHECKSUM_FIELDS(X) \ + CONFIG_WIRE_IDENTITY_FIELDS(X) \ + CONFIG_WIRE_METADATA_TIMES_FIELDS(X) \ + CONFIG_WIRE_SYMLINK_TRUST_FIELDS(X) \ + CONFIG_WIRE_XATTR_FIELDS(X) \ + CONFIG_WIRE_MODULE_FIELDS(X) \ + CONFIG_WIRE_DAEMON_AUTH_FIELDS(X) \ + CONFIG_WIRE_ICONV_FIELDS(X) \ + CONFIG_WIRE_PRIVILEGE_FIELDS(X) \ + CONFIG_WIRE_COPY_AS_FIELDS(X) + typedef struct Config { - char* version; - char* send_directory; - char* receive_root_directory; - bool save_to_disk; - bool use_multithreading; /* -j/--threads=N: number of parallel scanner worker threads for the -m * pipeline. 0 (the default, also set by bare -j/--threads) means "use the * scanner's built-in default" (4). CLIENT-ONLY: it is a local scheduling * concern and is NEVER serialized into the wire config frame. */ int scanner_threads; - bool use_chunk_serialization; - bool use_compression; - bool use_sendfile; - bool use_metadata; - bool use_executability; bool metadata_explicitly_disabled; bool show_progress; bool dry_run; - bool remove_source_files; - bool use_delete; - int compression_level; int compression_threads; - unsigned long long chunk_size; int ssh_port; TransportType transport; char* ssh_destination; - /* Daemon module selection (Wave A, protocol 2.15.0). Client-composed from a - * host::module/path destination; NULL or "" means "no module" (the ordinary - * standalone-server path). Crosses the wire as a trailing config-frame - * string so the daemon can look the module up in its own config and confine - * the connection to the module's root (never a client-chosen root). */ - char* module; - /* Daemon password authentication (A7 remediation, protocol 2.19.0). - * Client-composed from a --password-file whose first meaningful line is - * `user:password`: the client sends ONLY the username in the config frame - * (auth_user); the literal password is kept in auth_password CLIENT-SIDE for - * the duration of the SCRAM challenge/response and is NEVER serialized. Both - * are NULL when the client has no credentials to present; a module WITHOUT - * `auth users` stays open and the server ignores any credentials that do - * arrive (the client sends them opportunistically and the server decides). */ - char* auth_user; char* auth_password; /* Client-only path of --password-file (never crosses the wire; it is read to * populate auth_user/auth_password before connecting). */ char* password_file; char* fastsync_server_path; - /* --iconv=CONVERT_SPEC (protocol 2.16.0, rsync compatibility): convert the - * charset of FILE NAMES at the wire boundary. CONVERT_SPEC is - * "LOCAL[,REMOTE]": LOCAL is the charset of our own file names, REMOTE is - * the remote side's charset and defaults to LOCAL. The sender converts - * every path LOCAL->REMOTE before transmitting it; the receiver converts - * every received path back REMOTE->LOCAL before creating/writing it. The - * FULL SPEC crosses the wire as a trailing config-frame string so each end - * derives its own LOCAL and the wire (REMOTE) charset symmetrically. NULL - * (or "") means no conversion: identity with zero overhead. See charset.c - * and the PROTOCOL_VERSION note below. */ - char* iconv_spec; char** exclude_patterns; int exclude_count; char** include_patterns; int include_count; unsigned long long max_size; unsigned long long min_size; - unsigned long long max_alloc; - bool use_incremental; - bool ignore_times; - bool size_only; - bool use_delta; bool whole_file; - /* -y/--fuzzy: when a file must be transferred and the destination holds no - * usable file at the exact path, the receiver may reuse a SIMILAR-named - * existing regular file in the same destination directory as the delta - * basis so the sender transmits only the differences. Crosses the wire - * (the receiver performs the candidate search); the CLI implies - * --incremental + --delta because the similar-basis only matters on the - * receiver-driven delta path. Off by default. */ - bool fuzzy; - int modify_window; - uint32_t delta_block_size; - unsigned long long delta_max_file_size; bool use_tls; char* server_host; int server_port; @@ -170,18 +286,10 @@ typedef struct Config { /* --contimeout: connect()/accept timeout, transport layer only. */ int contimeout; bool quiet; - bool backup; - char* backup_dir; bool stats; int max_depth; FILE* log_file; - bool follow_symlinks; - bool partial; - // Issue #120: Symlink handling - bool copy_links; - bool safe_links; - bool copy_unsafe_links; /* Phase 4 symlink-trust. -k/--copy-dirlinks and --munge-links are * CLIENT/sender-side only (they decide how the SENDER scans and rewrites * symlinks; the receiver never reads them), so they never cross the wire. @@ -189,31 +297,6 @@ typedef struct Config { * symlink-to-directory as a directory) and CROSSES the wire along with * --munge-links (so the receiver knows to unmunge). */ bool copy_dirlinks; /* client-only, sender-side (-k) */ - bool munge_links; /* crosses the wire */ - bool keep_dirlinks; /* crosses the wire (-K) */ - - // Issue #121: Extended metadata preservation - bool preserve_hard_links; - bool preserve_acls; - bool preserve_xattrs; - bool preserve_devices; - bool preserve_sparse; - /* Phase 4 special/devices: preserve special files (FIFOs, sockets) and device - * nodes on the destination by recreating them (mknod/mkfifo) instead of - * transferring content. preserve_specials mirrors rsync --specials (the - * special-file half of -D); preserve_devices mirrors --devices (the device - * half of -D); both CROSS the wire so the receiver knows a special/device - * entry must be recreated rather than written as a regular file. */ - bool preserve_specials; - /* --copy-devices: copy the CONTENT of a source device as an ordinary regular - * file on the destination (rsync's non-privileged safe mode), instead of - * recreating the device node. CROSSES the wire (receiver treats the entry as - * a regular file, which is the default, so this is belt-and-braces). */ - bool copy_devices; - /* --write-devices: write the received data directly INTO an existing device - * node on the destination instead of creating a regular file. Dangeroud; - * see RSYNC_COMPAT.md for the tight gating. CROSSES the wire. */ - bool write_devices; // Issue #122: Output/logging options bool itemize_changes; @@ -223,57 +306,17 @@ typedef struct Config { int debug_level; bool list_only; bool human_readable; - bool eight_bit_output; - // Issue #127: Transfer modes - bool existing; - bool ignore_existing; - bool update; - bool inplace; - bool delay_updates; - bool use_fsync; - bool append; - bool append_verify; - /* --preallocate: allocates the destination file's full expected space up - * front (before any data is written) so a transfer that would overflow disk - * fails fast at allocation time and the file is laid out contiguously, - * avoiding fragmentation. Receiver-side, crosses the wire. */ - bool preallocate; - - // Issue #128: Extended delete options - /* --delete-excluded: also delete destination entries that were excluded on - * the source. Default (off) matches rsync: excluded paths are protected from - * deletion. Crosses the wire (the sender encodes the choice by whether it - * transmits a protected-prefix list with the keep-set manifest). */ - bool delete_excluded; - bool delete_after; - /* --max-delete=NUM: the receiver refuses to delete more than NUM entries per - * run (all-or-nothing: when the extras would exceed NUM nothing is removed and - * the transfer fails with a distinct error). -1 == no client limit (the - * server hard bound MAX_SERVER_DELETE_COUNT still applies). */ - int max_delete; /* --ignore-errors (client-only, never serialized): a sender-side source I/O * error (an unreadable directory during the scan) normally aborts the run so * no deletion happens; with --ignore-errors the scan continues and the * (partial) keep-set is still transmitted so the deletion runs. */ bool ignore_errors; - /* --force (receiver-side): a regular file may replace a destination - * directory by removing that (possibly non-empty, symlink-safe) directory - * tree first, instead of failing the write. Crosses the wire. */ - bool force_delete; /* --ignore-missing-args (client-only, never serialized): a --files-from * entry that does not exist under the source is silently skipped instead of * failing the run. Sender-side only: nothing is sent for it and it never * enters the keep-set. Implied by --delete-missing-args. */ bool ignore_missing_args; - /* --delete-missing-args: implies --ignore-missing-args; additionally each - * missing entry's destination mirror (computed like a present entry's wire - * path) is deleted receiver-side. Crosses the wire and is gated by the - * server's --allow-delete policy like --delete. rsync-parity: independent - * of ordinary --delete processing (it does not imply --delete); a non-empty - * directory mirror is only removed with --force or --delete in effect, and - * the missing-args deletions are not counted toward --max-delete. */ - bool delete_missing_args; // Issue #129: Advanced file selection. These fields are CLIENT-ONLY: they are // never serialized to the wire (the receiver must not learn them). @@ -283,23 +326,14 @@ typedef struct Config { bool from0; /* -0/--from0: NUL-delimited *-from files */ bool cvs_exclude; /* -C/--cvs-exclude: standard CVS ignore set */ bool per_dir_filter; /* -F: apply per-directory .rsync-filter files */ - bool prune_empty_dirs; bool one_file_system; /* -x/--one-file-system: do not cross filesystem boundaries */ - /* -R/--relative: crosses the wire; with --files-from listed entries keep - * their bare relative destination path (no source-root mirror prefix). */ - bool relative; /* --no-implied-dirs: client-only. With -R + --files-from, refuse to place a * listed file whose ancestor directory is not itself explicitly listed. */ bool no_implied_dirs; /* -d/--dirs: client-only. Transfer the directory entries named by the * source argument / --files-from list without recursing into contents. */ bool dirs; - /* --mkpath: crosses the wire. Tells the server to create the destination - * root directory (and missing leading components below its authorized root) - * at connection start instead of requiring it to already exist. */ - bool mkpath; - // Issue #130: Remote shell/connection options /* -e/--rsh: the remote-shell program used to establish the SSH transport. * NULL means the default "ssh". Client-only launch concern: NEVER crosses * the wire (it is not meaningful to the daemon/server handshake). */ @@ -312,7 +346,6 @@ typedef struct Config { * concern: NEVER crosses the wire. */ int outbuf; bool old_args; - char* temp_dir; /* --remote-option=OPT (Phase 5, long form only): one or more extra command-line * options to append to the REMOTE server invocation over SSH. CLIENT-ONLY: * they are composed into the remote command line by ssh_build_remote_command() @@ -321,34 +354,6 @@ typedef struct Config { * do NOT cross the wire and are never parsed on the receiver process. */ char** remote_options; int remote_option_count; - /* Alternate basis directories, ordered by command-line appearance. Each - * entry's type selects compare/copy/link behavior on an exact match. These - * cross the wire so the receiver can consult them; they are interpreted - * relative to the destination root and confined there. */ - BasisDest* basis_dirs; - int basis_count; - - // PR #174: Partial transfer resumption - char* partial_dir; - - // PR #178: Backup versioning - char* suffix; - - // PR #179: Delete policies - bool delete_before; - - /* rsync deletion-timing family (real from Phase 3). At most one of - delete_before / delete_during / delete_delay / delete_after may be set, and - only together with use_delete (the CLI implies --delete for each of them). - delete_before and delete_during select the EARLY engine mode: the keep-set - manifest is transmitted before any file data and extras are removed then, - acknowledged, before the first data byte. delete_delay and delete_after - select the LATE commit mode: extras are removed only after the whole - transfer has succeeded (plain --delete keeps this mode). The exact - semantics and the divergences from rsync are documented in RSYNC_COMPAT.md - and in config_delete_timing_early() below. */ - bool delete_during; - bool delete_delay; // PR #181: IPv6 and bind address char* address; @@ -370,119 +375,19 @@ typedef struct Config { * MOTD is shown when a daemon offers one). */ bool no_motd; - // PR #183: Checksum comparison - bool checksum; - - // PR #184: Compression algorithm negotiation - char* compress_choice; - char* chmod_spec; - - /* --checksum-choice / --cc and --checksum-seed. checksum_algo is the id of - * the whole-file content-digest algorithm used by the per-file --incremental - * handshake (sender computes it, receiver compares it to skip unchanged - * files) and by the basis-dir content verification. checksum_seed is passed - * to xxHash64 (and to the delta block strong hash, low 32 bits); md5 has no - * seed so it is ignored there. Both cross the wire: the receiver MUST hash - * the on-disk old file with the same algorithm and seed to reach a matching - * digest. Defaults (XXH64 / seed 0) reproduce the pre-existing behavior - * byte-for-byte. */ - int checksum_algo; /* ChecksumAlgo, default CHECKSUM_ALGO_XXH64 */ - uint64_t checksum_seed; /* default 0 */ - - char** skip_compress_suffixes; - int skip_compress_count; - bool skip_compress_set; - - // Issue #131: Identity mapping. These configure whether and how the receiver - // applies ownership when it is actually preserved/applied. ALL of them cross - // the wire (protocol 2.11.0) so the receiver resolves and applies ownership - // with the exact policy the client requested. Plain -M/--preserve still does - // NOT apply ownership (FastSync's deliberate conservative default); it is - // only attempted when at least one of these is set (see identity.h). - /* --numeric-ids: no name lookup, use the transmitted numeric ids raw. */ - bool numeric_ids; - /* --chown USER (owner) override; IDENTITY_CURRENT = the receiver's euid. */ - bool chown_uid_set; - int32_t chown_uid; - /* --chown :GROUP (group) override; IDENTITY_CURRENT = the receiver's egid. */ - bool chown_gid_set; - int32_t chown_gid; - /* --usermap / --groupmap entries, in order (first match wins). */ - IdentityMap* usermap; - int usermap_count; - IdentityMap* groupmap; - int groupmap_count; - - /* --super / --no-super (P7 Wave E, protocol 2.18.0): receiver-side privilege - * policy for super-user activities confined below the authorized receive - * root. SUPER_MODE_AUTO (default) preserves the pre-existing best-effort - * behavior: the confined super-user operation is ALWAYS attempted and an - * unprivileged attempt is refused by the kernel and skipped per entry. - * SUPER_MODE_ON (--super) explicitly REQUESTS those activities (char/block - * device-node creation, --write-devices); it does NOT imply --numeric-ids and - * never enables ownership application on its own. SUPER_MODE_OFF - * (--no-super) FORBIDS them even when running as root. FastSync NEVER - * elevates privileges (no setuid/seteuid/setgid) and never bypasses the - * fd-relative confinement (file_open_secure_parent, O_NOFOLLOW, root checks); - * --super only permits an attempt that is already confined. Crosses the wire - * as a trailing int so the receiver can enforce the policy. See - * privilege_super_permitted() and identity_ownership_requested() in - * identity.h. */ - SuperMode super_mode; - // Receiver-side runtime staging registry for --delay-updates. Never sent // over the wire and never set on the sender side. DelayUpdatesContext* delay_context; - // Phase 4: metadata time preservation. -U/--atimes and -N/--crtimes capture - // and transmit the source access / birth time (both sender and receiver - // effect, so they CROSS the wire). --omit-dir-times/-O and - // --omit-link-times/-J are receiver-side prefs (CROSS the wire). Their - // exact capture/transmit/apply semantics are documented in RSYNC_COMPAT.md. - /* -U/--atimes: preserve source access times on the destination. */ - bool preserve_atimes; - /* -N/--crtimes: capture+transmit source birth time; see RSYNC_COMPAT for the - * receiver not-applied divergence. */ - bool preserve_crtimes; - /* -O/--omit-dir-times: do not apply mtimes to directories. */ - bool omit_dir_times; - /* -J/--omit-link-times: do not apply times to symlinks. */ - bool omit_link_times; /* --open-noatime: CLIENT-ONLY (never crosses the wire). The sender opens * source files with O_NOATIME so reading for transfer does not bump the * source access time. */ bool open_noatime; - // Phase 4: xattr / ACL / fake-super preservation. - /* -X/--xattrs and -A/--acls toggle the sender's capture and the receiver's - * application of per-file extended attributes (xattrs). Both cross the wire: - * the sender only transmits the bounded, whitelisted attribute set it - * captures and the receiver re-validates namespaces/sizes before applying - * fd-relative. With neither set (the default) no xattr block is sent, so the - * wire is byte-identical to prior protocol versions for unaffected runs. */ /* true when preserve_xattrs || preserve_acls; the sender/receiver gate the * xattr wire block on this single flag. */ bool use_xattrs; - /* --fake-super: receiver-only. When set, each written file additionally gets - * a reserved user.fastsync.stat xattr recording the source uid/gid/mode/mtime - * so a later privileged restore could re-apply them. Crosses the wire. */ - bool fake_super; - /* --copy-as=USER[:GROUP] (P7 Wave E, protocol 2.18.0). Safe-subset - * implementation, a documented divergence from rsync's real identity switch: - * the receiver does NOT change its process credentials (FastSync's receiver - * is multithreaded, so a setuid/seteuid drop would be unsafe). Instead the - * receiver FORCES the ownership of every entry it writes to copy_as_uid / - * copy_as_gid through the existing confined, fd-relative identity path - * (fchown/fchownat), which REQUIRES receiver privilege (root); an - * unprivileged receiver REFUSES the whole transfer up front at the config - * handshake (never a silent wrong-ownership result). All three fields CROSS - * the wire as a trailing config-frame block so the receiver learns the - * requested ids; see the PROTOCOL_VERSION note below. */ - bool copy_as_set; - int32_t copy_as_uid; - int32_t copy_as_gid; - // Phase 5: --trust-sender /* Long-form-only, receiver-local policy. rsync's --trust-sender tells the * receiving side to trust that the sender already produced a sane file list, * relaxing the receiver's own up-front re-validation of every incoming path. @@ -501,7 +406,6 @@ typedef struct Config { * default; only relaxes validation when explicitly requested. */ bool trust_sender; - // Phase 6: --stop-after / --stop-at /* Client-only sender-side transfer stop deadlines. --stop-after=MINS stops * the transfer after a number of elapsed minutes (checked against * CLOCK_MONOTONIC so clock changes do not skew it); --stop-at=TIME stops at @@ -513,7 +417,6 @@ typedef struct Config { time_t stop_at; /* --stop-at=... absolute wall-clock deadline */ bool stop_at_set; /* true when --stop-at was given */ - // Phase 6: --write-batch / --only-write-batch / --read-batch /* Client-only residual-batch paths. A residual batch is a self-contained * single-file record of the whole source tree (full file images using the * chunk codec), independent of any live server. --write-batch=FILE runs the @@ -525,6 +428,196 @@ typedef struct Config { char* write_batch; /* --write-batch=FILE path, or NULL */ char* only_write_batch; /* --only-write-batch=FILE path, or NULL */ char* read_batch; /* --read-batch=FILE path, or NULL */ + + /* =================================================================== + * Serialized wire fields. Their members, defaults and send/receive + * sequence are generated from the CONFIG_WIRE_*_FIELDS table above (the + * single source of truth); they are declared here in exact wire order. + * The per-field notes were moved here from their original positions and + * are listed in wire order. + * =================================================================== */ + /* copy_links */ + // Issue #120: Symlink handling + /* preserve_hard_links */ + // Issue #121: Extended metadata preservation + /* preserve_specials */ + /* Phase 4 special/devices: preserve special files (FIFOs, sockets) and device + * nodes on the destination by recreating them (mknod/mkfifo) instead of + * transferring content. preserve_specials mirrors rsync --specials (the + * special-file half of -D); preserve_devices mirrors --devices (the device + * half of -D); both CROSS the wire so the receiver knows a special/device + * entry must be recreated rather than written as a regular file. */ + /* copy_devices */ + /* --copy-devices: copy the CONTENT of a source device as an ordinary regular + * file on the destination (rsync's non-privileged safe mode), instead of + * recreating the device node. CROSSES the wire (receiver treats the entry as + * a regular file, which is the default, so this is belt-and-braces). */ + /* write_devices */ + /* --write-devices: write the received data directly INTO an existing device + * node on the destination instead of creating a regular file. Dangeroud; + * see RSYNC_COMPAT.md for the tight gating. CROSSES the wire. */ + /* existing */ + // Issue #127: Transfer modes + /* delete_excluded */ + /* --delete-excluded: also delete destination entries that were excluded on + * the source. Default (off) matches rsync: excluded paths are protected from + * deletion. Crosses the wire (the sender encodes the choice by whether it + * transmits a protected-prefix list with the keep-set manifest). */ + /* force_delete */ + /* --force (receiver-side): a regular file may replace a destination + * directory by removing that (possibly non-empty, symlink-safe) directory + * tree first, instead of failing the write. Crosses the wire. */ + /* delete_missing_args */ + /* --delete-missing-args: implies --ignore-missing-args; additionally each + * missing entry's destination mirror (computed like a present entry's wire + * path) is deleted receiver-side. Crosses the wire and is gated by the + * server's --allow-delete policy like --delete. rsync-parity: independent + * of ordinary --delete processing (it does not imply --delete); a non-empty + * directory mirror is only removed with --force or --delete in effect, and + * the missing-args deletions are not counted toward --max-delete. */ + /* preallocate */ + /* --preallocate: allocates the destination file's full expected space up + * front (before any data is written) so a transfer that would overflow disk + * fails fast at allocation time and the file is laid out contiguously, + * avoiding fragmentation. Receiver-side, crosses the wire. */ + /* max_delete */ + /* --max-delete=NUM: the receiver refuses to delete more than NUM entries per + * run (all-or-nothing: when the extras would exceed NUM nothing is removed and + * the transfer fails with a distinct error). -1 == no client limit (the + * server hard bound MAX_SERVER_DELETE_COUNT still applies). */ + /* relative */ + /* -R/--relative: crosses the wire; with --files-from listed entries keep + * their bare relative destination path (no source-root mirror prefix). */ + /* mkpath */ + /* --mkpath: crosses the wire. Tells the server to create the destination + * root directory (and missing leading components below its authorized root) + * at connection start instead of requiring it to already exist. */ + /* delete_during */ + /* rsync deletion-timing family (real from Phase 3). At most one of + delete_before / delete_during / delete_delay / delete_after may be set, and + only together with use_delete (the CLI implies --delete for each of them). + delete_before and delete_during select the EARLY engine mode: the keep-set + manifest is transmitted before any file data and extras are removed then, + acknowledged, before the first data byte. delete_delay and delete_after + select the LATE commit mode: extras are removed only after the whole + transfer has succeeded (plain --delete keeps this mode). The exact + semantics and the divergences from rsync are documented in RSYNC_COMPAT.md + and in config_delete_timing_early() below. */ + /* partial_dir */ + // PR #174: Partial transfer resumption + /* suffix */ + // PR #178: Backup versioning + /* delete_before */ + // PR #179: Delete policies + /* checksum */ + // PR #183: Checksum comparison + /* compress_choice */ + // PR #184: Compression algorithm negotiation + /* basis_dirs */ + /* Alternate basis directories, ordered by command-line appearance. Each + * entry's type selects compare/copy/link behavior on an exact match. These + * cross the wire so the receiver can consult them; they are interpreted + * relative to the destination root and confined there. */ + /* fuzzy */ + /* -y/--fuzzy: when a file must be transferred and the destination holds no + * usable file at the exact path, the receiver may reuse a SIMILAR-named + * existing regular file in the same destination directory as the delta + * basis so the sender transmits only the differences. Crosses the wire + * (the receiver performs the candidate search); the CLI implies + * --incremental + --delta because the similar-basis only matters on the + * receiver-driven delta path. Off by default. */ + /* checksum_algo / checksum_seed */ + /* --checksum-choice / --cc and --checksum-seed. checksum_algo is the id of + * the whole-file content-digest algorithm used by the per-file --incremental + * handshake (sender computes it, receiver compares it to skip unchanged + * files) and by the basis-dir content verification. checksum_seed is passed + * to xxHash64 (and to the delta block strong hash, low 32 bits); md5 has no + * seed so it is ignored there. Both cross the wire: the receiver MUST hash + * the on-disk old file with the same algorithm and seed to reach a matching + * digest. */ + /* munge_links / keep_dirlinks */ + /* Phase 4 symlink-trust: both cross the wire (the receiver unmunges symlink + * targets and, with -K, follows an in-root destination symlink-to-directory); + * -k/--copy-dirlinks is sender-only and is never serialized. */ + /* numeric_ids */ + /* --numeric-ids: no name lookup, use the transmitted numeric ids raw. */ + /* chown_uid_set */ + /* --chown USER (owner) override; IDENTITY_CURRENT = the receiver's euid. */ + /* chown_gid_set */ + /* --chown :GROUP (group) override; IDENTITY_CURRENT = the receiver's egid. */ + /* usermap */ + /* --usermap / --groupmap entries, in order (first match wins). */ + /* preserve_atimes */ + /* -U/--atimes: preserve source access times on the destination. */ + /* preserve_crtimes */ + /* -N/--crtimes: capture+transmit source birth time; see RSYNC_COMPAT for the + * receiver not-applied divergence. */ + /* omit_dir_times */ + /* -O/--omit-dir-times: do not apply mtimes to directories. */ + /* omit_link_times */ + /* -J/--omit-link-times: do not apply times to symlinks. */ + /* fake_super */ + /* --fake-super: receiver-only. When set, each written file additionally gets + * a reserved user.fastsync.stat xattr recording the source uid/gid/mode/mtime + * so a later privileged restore could re-apply them. Crosses the wire. */ + /* module */ + /* Daemon module selection (Wave A, protocol 2.15.0). Client-composed from a + * host::module/path destination; NULL or "" means "no module" (the ordinary + * standalone-server path). Crosses the wire as a trailing config-frame + * string so the daemon can look the module up in its own config and confine + * the connection to the module's root (never a client-chosen root). */ + /* auth_user */ + /* Daemon password authentication (A7 remediation, protocol 2.19.0). + * Client-composed from a --password-file whose first meaningful line is + * `user:password`: the client sends ONLY the username in the config frame + * (auth_user); the literal password is kept in auth_password CLIENT-SIDE for + * the duration of the SCRAM challenge/response and is NEVER serialized. Both + * are NULL when the client has no credentials to present; a module WITHOUT + * `auth users` stays open and the server ignores any credentials that do + * arrive (the client sends them opportunistically and the server decides). */ + /* iconv_spec */ + /* --iconv=CONVERT_SPEC (protocol 2.16.0, rsync compatibility): convert the + * charset of FILE NAMES at the wire boundary. CONVERT_SPEC is + * "LOCAL[,REMOTE]": LOCAL is the charset of our own file names, REMOTE is + * the remote side's charset and defaults to LOCAL. The sender converts + * every path LOCAL->REMOTE before transmitting it; the receiver converts + * every received path back REMOTE->LOCAL before creating/writing it. The + * FULL SPEC crosses the wire as a trailing config-frame string so each end + * derives its own LOCAL and the wire (REMOTE) charset symmetrically. NULL + * (or "") means no conversion: identity with zero overhead. See charset.c + * and the PROTOCOL_VERSION note below. */ + /* super_mode */ + /* --super / --no-super (P7 Wave E, protocol 2.18.0): receiver-side privilege + * policy for super-user activities confined below the authorized receive + * root. SUPER_MODE_AUTO (default) preserves the pre-existing best-effort + * behavior: the confined super-user operation is ALWAYS attempted and an + * unprivileged attempt is refused by the kernel and skipped per entry. + * SUPER_MODE_ON (--super) explicitly REQUESTS those activities (char/block + * device-node creation, --write-devices); it does NOT imply --numeric-ids and + * never enables ownership application on its own. SUPER_MODE_OFF + * (--no-super) FORBIDS them even when running as root. FastSync NEVER + * elevates privileges (no setuid/seteuid/setgid) and never bypasses the + * fd-relative confinement (file_open_secure_parent, O_NOFOLLOW, root checks); + * --super only permits an attempt that is already confined. Crosses the wire + * as a trailing int so the receiver can enforce the policy. See + * privilege_super_permitted() and identity_ownership_requested() in + * identity.h. */ + /* copy_as_set */ + /* --copy-as=USER[:GROUP] (P7 Wave E, protocol 2.18.0). Safe-subset + * implementation, a documented divergence from rsync's real identity switch: + * the receiver does NOT change its process credentials (FastSync's receiver + * is multithreaded, so a setuid/seteuid drop would be unsafe). Instead the + * receiver FORCES the ownership of every entry it writes to copy_as_uid / + * copy_as_gid through the existing confined, fd-relative identity path + * (fchown/fchownat), which REQUIRES receiver privilege (root); an + * unprivileged receiver REFUSES the whole transfer up front at the config + * handshake (never a silent wrong-ownership result). All three fields CROSS + * the wire as a trailing config-frame block so the receiver learns the + * requested ids; see the PROTOCOL_VERSION note below. */ + +#define CONFIG_STRUCT_MEMBER(name, ctype, def, kind) ctype name; + CONFIG_WIRE_FIELDS(CONFIG_STRUCT_MEMBER) +#undef CONFIG_STRUCT_MEMBER } Config; /* Phase 5 (remote-option wave): 2.13.0 -> 2.14.0. @@ -699,6 +792,10 @@ void config_delete(Config* config); void config_burn_auth(Config* config); bool config_send(int file_descriptor, const Config* config); +/* Emit the config frame BODY (every serialized field, in wire order) without + * the trailing STATUS_OK handshake. config_send() is this plus the handshake; + * the wire-compatibility golden test uses it to hash the exact byte stream. */ +bool config_send_wire_block(int file_descriptor, const Config* config); Config* config_receive(int file_descriptor); bool config_is_remote_dest(const char* s); void config_parse_ssh_dest(Config* config); From 4e918a1b69caa4fb5ee9643bef423c1bc7d03b55 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:28:31 +0200 Subject: [PATCH 043/155] test(config): pin wire bytes and round-trip every field test_config_wire_golden() serializes a fully-populated Config through config_send_wire_block() and pins the exact frame to len=633 and FNV-1a hash 6163263374908258816, captured from the pre-X-macro implementation. Any field reorder, resize or codec change fails the test. test_config_wire_roundtrip_all_fields() serializes/deserializes a defaults Config and a fully-populated Config over a socketpair and compares every serialized field. The comparison is itself generated from CONFIG_WIRE_FIELDS (one CONFIG_CMP_ per table entry), so a new table entry automatically extends coverage; it cannot fall out of sync. It normalizes the receiver's NULL/"" canonicalization, the max_alloc server clamp and the derived use_delta/use_xattrs bits. --- tests/test_config.c | 304 ++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 304 insertions(+) diff --git a/tests/test_config.c b/tests/test_config.c index 7689cf5..41d6a30 100644 --- a/tests/test_config.c +++ b/tests/test_config.c @@ -1,5 +1,6 @@ #include "test_config.h" #include "config.h" +#include "delta.h" #include "identity.h" #include "multiprocessing.h" #include "protocol.h" @@ -2194,6 +2195,307 @@ static void test_config_receive_rejects_unified_invariants() { } } +/* --------------------------------------------------------------------------- + * Wire round-trip equivalence. + * + * config_wire_equal() is generated from the SAME CONFIG_WIRE_FIELDS table as + * the serializer, so it can never miss a serialized field: adding a table + * entry automatically extends this comparison. Each KIND maps to a comparison + * macro; STR_OPT/STR_KEEP normalize the NULL-vs-"" canonicalization the + * receiver performs, RAW_MAXALLOC models the server-side clamp, and + * DERIVED_DELTA compares the effective (whole_file-suppressed) bit. + * ------------------------------------------------------------------------- */ +static void golden_config_populate(Config* c); + +static bool str_opt_equal(const char* a, const char* b) { + if (a == NULL || a[0] == '\0') + return b == NULL || b[0] == '\0'; + return b != NULL && strcmp(a, b) == 0; +} + +static bool idmap_equal(const IdentityMap* a, int ac, const IdentityMap* b, int bc) { + if (ac != bc) + return false; + for (int i = 0; i < ac; i++) { + if (a[i].from != b[i].from || a[i].to != b[i].to) + return false; + } + return true; +} + +static bool skip_suffixes_equal(const Config* a, const Config* b) { + if (a->skip_compress_count != b->skip_compress_count) + return false; + for (int i = 0; i < a->skip_compress_count; i++) { + if (!str_opt_equal(a->skip_compress_suffixes[i], b->skip_compress_suffixes[i])) + return false; + } + return true; +} + +static bool basis_equal(const Config* a, const Config* b) { + if (a->basis_count != b->basis_count) + return false; + for (int i = 0; i < a->basis_count; i++) { + if (a->basis_dirs[i].type != b->basis_dirs[i].type || + !str_opt_equal(a->basis_dirs[i].path, b->basis_dirs[i].path)) + return false; + } + return true; +} + +#define CONFIG_CMP_BOOL(a, b, name) ((a)->name == (b)->name) +#define CONFIG_CMP_INT(a, b, name) ((a)->name == (b)->name) +#define CONFIG_CMP_RAW(a, b, name) ((a)->name == (b)->name) +#define CONFIG_CMP_BOOL_8BIT(a, b, name) ((a)->name == (b)->name) +#define CONFIG_CMP_RAW_MAXALLOC(a, b, name) \ + ((b)->name == ((a)->name > MAX_SERVER_ALLOC ? MAX_SERVER_ALLOC : (a)->name)) +#define CONFIG_CMP_DERIVED_DELTA(a, b, name) ((b)->name == ((a)->name && !(a)->whole_file)) +#define CONFIG_CMP_STR(a, b, name) \ + ((a)->name != NULL && (b)->name != NULL && strcmp((a)->name, (b)->name) == 0) +#define CONFIG_CMP_STR_OPT(a, b, name) str_opt_equal((a)->name, (b)->name) +#define CONFIG_CMP_STR_KEEP(a, b, name) str_opt_equal((a)->name, (b)->name) +#define CONFIG_CMP_STR_MODULE(a, b, name) str_opt_equal((a)->name, (b)->name) +#define CONFIG_CMP_STR_REDACTED_AUTH(a, b, name) str_opt_equal((a)->name, (b)->name) +#define CONFIG_CMP_INT_CHECKSUM_ALGO(a, b, name) ((a)->name == (b)->name) +#define CONFIG_CMP_SUPERMODE(a, b, name) ((a)->name == (b)->name) +#define CONFIG_CMP_INT_IDENTITY(a, b, name) ((a)->name == (b)->name) +#define CONFIG_CMP_INT_SKIPCOUNT(a, b, name) ((a)->name == (b)->name) +#define CONFIG_CMP_INT_BASISCOUNT(a, b, name) ((a)->name == (b)->name) +#define CONFIG_CMP_INT_IDMAPCOUNT(a, b, name) ((a)->name == (b)->name) +#define CONFIG_CMP_BOOL_XATTR_DERIVE(a, b, name) ((a)->name == (b)->name) +#define CONFIG_CMP_COPY_AS_PRESENCE(a, b, name) ((a)->name == (b)->name) +#define CONFIG_CMP_COPY_AS_ID(a, b, name) (!(a)->copy_as_set || (a)->name == (b)->name) +#define CONFIG_CMP_BLOCK_SKIP_SUFFIXES(a, b, name) skip_suffixes_equal((a), (b)) +#define CONFIG_CMP_BLOCK_BASIS(a, b, name) basis_equal((a), (b)) +#define CONFIG_CMP_BLOCK_IDMAP(a, b, name) \ + idmap_equal((a)->name, (a)->name##_count, (b)->name, (b)->name##_count) + +#define WIRE_CMP(name, ctype, def, kind) \ + &&(CONFIG_CMP_##kind(a, b, name) \ + ? true \ + : (fprintf(stderr, " mismatched field: %s\n", #name), false)) + +static bool config_wire_equal(const Config* a, const Config* b) { + return true CONFIG_WIRE_FIELDS(WIRE_CMP); +} + +static bool roundtrip_and_compare(const Config* send_cfg) { + int p[2]; + if (socketpair(AF_UNIX, SOCK_STREAM, 0, p) != 0) + return false; + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + io_set_bwlimit(0); + Config* recv = config_receive(p[0]); + bool equal = recv != NULL && config_wire_equal(send_cfg, recv); + config_delete(recv); + close(p[0]); + _exit(equal ? 0 : 1); + } + close(p[0]); + io_set_fds(p[1], p[1]); + io_set_bwlimit(0); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + return sent && WIFEXITED(status) && WEXITSTATUS(status) == 0; +} + +/* Every serialized field must survive a frame round-trip, for a defaults config + * and for a fully-populated config. */ +static void test_config_wire_roundtrip_all_fields() { + if (is_running_under_valgrind()) + return; + + Config* defaults = config_create(); + EXPECT_NOT_NULL(defaults); + defaults->send_directory = str_dup("/src"); + defaults->receive_root_directory = str_dup("/dst"); + EXPECT_TRUE(roundtrip_and_compare(defaults)); + config_delete(defaults); + + Config* populated = config_create(); + EXPECT_NOT_NULL(populated); + golden_config_populate(populated); + /* Keep the populated config within the server-side validation bounds. */ + populated->delta_max_file_size = DELTA_MAX_FILE_SIZE; + populated->whole_file = false; + /* "X" is not part of FastSync's chmod grammar (see parse_clause), so use a + * spec the receiver-side validator accepts. */ + free(populated->chmod_spec); + populated->chmod_spec = str_dup("u=rw,go=r"); + EXPECT_TRUE(roundtrip_and_compare(populated)); + config_delete(populated); +} + +/* Populate every serialized field with a non-default value so the wire frame + * exercises each table entry. The values are deterministic. */ +static void golden_config_populate(Config* c) { + c->eight_bit_output = true; + c->max_alloc = 123456789ULL; + c->send_directory = str_dup("/golden/src"); + c->receive_root_directory = str_dup("/golden/dst"); + c->save_to_disk = true; + c->use_multithreading = true; + c->use_chunk_serialization = false; + c->use_compression = false; + c->use_metadata = true; + c->use_executability = true; + c->compression_level = 7; + c->chunk_size = 65536; + c->use_sendfile = false; + c->use_delete = true; + c->use_incremental = true; + c->size_only = true; + c->ignore_times = true; + c->use_delta = true; + c->whole_file = false; + c->delta_block_size = 4096; + c->delta_max_file_size = 987654321ULL; + c->backup = true; + c->backup_dir = str_dup("/golden/backup"); + c->remove_source_files = true; + c->follow_symlinks = true; + c->copy_links = true; + c->safe_links = true; + c->copy_unsafe_links = true; + c->preserve_hard_links = true; + c->preserve_acls = true; + c->preserve_xattrs = true; + c->preserve_devices = true; + c->preserve_sparse = true; + c->preserve_specials = true; + c->copy_devices = true; + c->write_devices = true; + c->ignore_existing = true; + c->existing = true; + c->update = true; + c->inplace = false; + c->delay_updates = false; + c->append = false; + c->use_fsync = true; + c->append_verify = false; + c->delete_excluded = true; + c->force_delete = true; + c->delete_missing_args = true; + c->delete_after = true; + c->preallocate = true; + c->max_delete = 42; + c->relative = true; + c->prune_empty_dirs = true; + c->mkpath = true; + c->delete_during = false; + c->delete_delay = false; + c->temp_dir = str_dup("/golden/tmp"); + c->partial = true; + c->partial_dir = str_dup("/golden/partial"); + c->suffix = str_dup(".golden"); + c->delete_before = false; + c->checksum = true; + c->modify_window = 3; + c->compress_choice = str_dup("zstd"); + c->chmod_spec = str_dup("u=rwX,go=rX"); + c->skip_compress_set = true; + c->skip_compress_count = 2; + c->skip_compress_suffixes = calloc(2, sizeof(char*)); + c->skip_compress_suffixes[0] = str_dup(".gz"); + c->skip_compress_suffixes[1] = str_dup(".xz"); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_COMPARE, "compare"), 0); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "link"), 0); + c->fuzzy = true; + c->checksum_algo = CHECKSUM_ALGO_MD5; + c->checksum_seed = 0x1122334455667788ULL; + c->numeric_ids = true; + c->chown_uid_set = true; + c->chown_uid = 1234; + c->chown_gid_set = true; + c->chown_gid = 5678; + c->usermap_count = 2; + c->usermap = calloc(2, sizeof(IdentityMap)); + c->usermap[0].from = IDENTITY_MATCH_ANY; + c->usermap[0].to = 1000; + c->usermap[1].from = 5; + c->usermap[1].to = 6; + c->groupmap_count = 1; + c->groupmap = calloc(1, sizeof(IdentityMap)); + c->groupmap[0].from = 7; + c->groupmap[0].to = 8; + c->preserve_atimes = true; + c->preserve_crtimes = true; + c->omit_dir_times = true; + c->omit_link_times = true; + c->munge_links = true; + c->keep_dirlinks = true; + c->fake_super = true; + c->module = str_dup("goldenmod"); + c->auth_user = str_dup("goldenuser"); + c->auth_password = str_dup("golden-pw"); + c->iconv_spec = str_dup("UTF-8,UTF-8"); + c->super_mode = SUPER_MODE_ON; + c->copy_as_set = true; + c->copy_as_uid = 111; + c->copy_as_gid = 222; +} + +/* FNV-1a 64 over the exact config-frame bytes emitted by + * config_send_wire_block(). This pins field order and width: any reorder or + * resize changes the hash. */ +static unsigned long long capture_wire_hash(const Config* cfg, size_t* out_len) { + int p[2]; + if (socketpair(AF_UNIX, SOCK_STREAM, 0, p) != 0) + return 0; + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + io_set_bwlimit(0); + bool ok = config_send_wire_block(p[0], cfg); + close(p[0]); + _exit(ok ? 0 : 1); + } + close(p[0]); + unsigned long long h = 1469598103934665603ULL; + unsigned char buf[4096]; + ssize_t n; + size_t total = 0; + while ((n = read(p[1], buf, sizeof(buf))) > 0) { + for (ssize_t i = 0; i < n; i++) { + h ^= (unsigned long long)buf[i]; + h *= 1099511628211ULL; + } + total += (size_t)n; + } + close(p[1]); + int status = 0; + waitpid(pid, &status, 0); + if (!WIFEXITED(status) || WEXITSTATUS(status) != 0) + return 0; + *out_len = total; + return h; +} + +/* Byte-for-byte wire compatibility guard (protocol 2.20.0). The expected hash + * was captured from the pre-X-macro implementation; the refactor MUST NOT + * change it. */ +static void test_config_wire_golden() { + if (is_running_under_valgrind()) + return; + Config* c = config_create(); + EXPECT_NOT_NULL(c); + golden_config_populate(c); + size_t len = 0; + unsigned long long h = capture_wire_hash(c, &len); + printf(" wire golden: len=%zu hash=%llu\n", len, h); + /* Captured from the pre-X-macro (protocol 2.20.0) implementation. */ + EXPECT_TRUE(len == 633); + EXPECT_TRUE(h == 6163263374908258816ULL); + config_delete(c); +} + void test_config() { test_config_lifecycle(); test_config_ssh_dest(); @@ -2249,6 +2551,8 @@ void test_config() { test_config_receive_with_validate_rejects(); test_config_invariants_error_all_combinations(); test_config_receive_rejects_unified_invariants(); + test_config_wire_golden(); + test_config_wire_roundtrip_all_fields(); } test_identity_copy_as_refused(); test_identity_ownership_requested(); From 5c8970c64fdcaeae4858456584fd58eefd34a3cf Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:34:59 +0200 Subject: [PATCH 044/155] chore(opencode): fix drifted agent/skill docs and repo hygiene The agent and skill definitions had drifted badly from the current codebase and tooling, repeating the same class of bug as the benchmark tool (references to nonexistent scripts and invented flags): - Replace the removed `python3 test.py` with the real integration command (`python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"`) across agents and skills. - Fix `feature-scout`'s fabricated CLI flag list (--host, --server-mode, --use-* etc.) using the authoritative src/client/usage.c flags. - Fix `perf-analyst` benchmark flags (-m -c -> -j -z) and point at benchmark/bench.py instead of stale numbers. - Correct `code-explainer` (no getopt_long; --sendfile not -f) and version drift in the release skill (1.1.0 -> 2.20.0). - Replace GitHub/`gh` workflows with Gitea/`tea` (PRs target dev; issues via tea; branch strategy updated in all agents). - Use the built-in `-DSANITIZER=address|thread` CMake option instead of hand-rolled -fsanitize flags. - Add `-p 8080 --allow-unauthenticated` to plain-TCP server examples. - Merge the redundant security-screener into security-auditor; drop the duplicate (16 agents remain). Repo hygiene: gitignore `root/` and `test_partial_install_tmp/`, remove the empty leftover trees, delete the tracked scratch scripts tmux.sh and to_one_file.py, and note the compile_commands.json symlink in README. --- .gitignore | 4 + .opencode/agents/architect.md | 2 +- .opencode/agents/c-reviewer.md | 2 +- .opencode/agents/cmake-expert.md | 6 +- .opencode/agents/code-explainer.md | 10 +- .opencode/agents/code-quality-guardian.md | 2 +- .opencode/agents/debugger.md | 18 +- .opencode/agents/doc-generator.md | 2 +- .opencode/agents/feature-scout.md | 77 +++-- .opencode/agents/integrator.md | 5 +- .opencode/agents/issue-creator.md | 33 +- .opencode/agents/perf-analyst.md | 12 +- .opencode/agents/protocol-designer.md | 2 +- .opencode/agents/refactorer.md | 2 +- .opencode/agents/security-auditor.md | 352 +++++++++++++++++++--- .opencode/agents/security-screener.md | 310 ------------------- .opencode/agents/test-writer.md | 2 +- .opencode/skills/debug-workflow/SKILL.md | 29 +- .opencode/skills/pr-build/SKILL.md | 14 +- .opencode/skills/pr-review/SKILL.md | 4 +- .opencode/skills/release/SKILL.md | 29 +- .opencode/skills/security-audit/SKILL.md | 4 +- AGENTS.md | 6 +- README.md | 2 + tmux.sh | 15 - to_one_file.py | 14 - 26 files changed, 459 insertions(+), 499 deletions(-) delete mode 100644 .opencode/agents/security-screener.md delete mode 100755 tmux.sh delete mode 100644 to_one_file.py diff --git a/.gitignore b/.gitignore index a6030f3..d238172 100644 --- a/.gitignore +++ b/.gitignore @@ -8,3 +8,7 @@ build-*/ build2/ build3/ build_docker2/ + +# Test/run artifacts +root/ +test_partial_install_tmp/ diff --git a/.opencode/agents/architect.md b/.opencode/agents/architect.md index cce442f..878cf5a 100644 --- a/.opencode/agents/architect.md +++ b/.opencode/agents/architect.md @@ -128,7 +128,7 @@ Do not wait for the user to tell you CI failed — check proactively. The user s ## Branch Strategy -Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b `) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging. +Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b `), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head `. Wait for CI to pass before merging. ## Dependency Installation diff --git a/.opencode/agents/c-reviewer.md b/.opencode/agents/c-reviewer.md index 9c0ba24..07e3fd3 100644 --- a/.opencode/agents/c-reviewer.md +++ b/.opencode/agents/c-reviewer.md @@ -92,7 +92,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf ## Branch Strategy -Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b `) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging. +Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b `), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head `. Wait for CI to pass before merging. ## Dependency Installation diff --git a/.opencode/agents/cmake-expert.md b/.opencode/agents/cmake-expert.md index 430bfc9..211f104 100644 --- a/.opencode/agents/cmake-expert.md +++ b/.opencode/agents/cmake-expert.md @@ -84,7 +84,7 @@ tests/integration/ — Python pytest integration tests ### Dependencies - **zstd** — found via `find_library(ZSTD_LIBRARY zstd)` - **OpenSSL** — found via `find_package(OpenSSL REQUIRED)` (TLS 1.2+ transport) -- **xxHash** — fetched via `FetchContent` from GitHub (delta transfer hashing, v0.8.3) +- **xxHash** — fetched via `FetchContent` from the upstream repository (delta transfer hashing, v0.8.3) - **pthreads** — found via `find_package(Threads REQUIRED)` - **C11 standard** — required - **CMake 3.22+** — minimum version @@ -159,7 +159,7 @@ cmake -B build -S . -DCMAKE_BUILD_TYPE=RelWithDebInfo ```bash cmake -B build -S . cmake --build build -j$(nproc) -./build/server +./build/server -p 8080 --allow-unauthenticated ./build/client ./build/tests ``` @@ -187,7 +187,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf ## Branch Strategy -Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b `) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging. +Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b `), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head `. Wait for CI to pass before merging. ## Dependency Installation diff --git a/.opencode/agents/code-explainer.md b/.opencode/agents/code-explainer.md index b36ba1b..5039f7a 100644 --- a/.opencode/agents/code-explainer.md +++ b/.opencode/agents/code-explainer.md @@ -27,7 +27,7 @@ FastSync is a file synchronization tool (like rsync, but faster). It transfers f cmake -B build -S . && cmake --build build -j$(nproc) # Server (TCP mode) -./build/server +./build/server -p 8080 --allow-unauthenticated # Client (TCP mode) ./build/client --source-dir /path/to/send --dest-dir /path/to/receive --save-to-disk @@ -37,13 +37,13 @@ cmake -B build -S . && cmake --build build -j$(nproc) # Run tests ./build/tests # unit tests -python3 test.py # integration tests +python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" # integration tests ``` ## Code Walkthrough ### Client Entry Point (`src/client/client_cli.c`) -- Parses CLI arguments using `getopt_long` +- Parses CLI arguments using a custom option-table parser (`OPTION_TABLE` in `src/client/client_cli.c`); there is no `getopt*` usage - Creates `Config` struct with all options - Detects SSH destinations (contains `:`) - Calls into `client_send.c` for the actual transfer @@ -109,7 +109,7 @@ Collection of files for batch transfer. Serialized with file count, then per-fil zstd streaming compression via `ZSTD_compressStream2`/`ZSTD_decompressStream`. Compression happens per-chunk in the sender stage. Level 1-22 (default 5). Streaming means memory usage stays bounded regardless of file size. ### "How does sendfile() work?" -On Linux, `sendfile()` copies data directly from kernel file buffer to socket, bypassing userspace. ~2x faster for large files. Enabled with `-f` flag. Only works with TCP (not SSH, not compression). +On Linux, `sendfile()` copies data directly from kernel file buffer to socket, bypassing userspace. ~2x faster for large files. Enabled with `--sendfile` (long form only). Only works with TCP (not SSH, not compression). ### "How does incremental sync work?" Client sends file metadata (path, size, mtime) to server. Server checks if destination file has same size+mtime. If match, server responds `STATUS_OK` (skip). If mismatch, server responds `STATUS_NEXT` (send). @@ -138,7 +138,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf ## Branch Strategy -Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b `) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging. +Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b `), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head `. Wait for CI to pass before merging. ## Dependency Installation diff --git a/.opencode/agents/code-quality-guardian.md b/.opencode/agents/code-quality-guardian.md index d28e2b9..e8aab2f 100644 --- a/.opencode/agents/code-quality-guardian.md +++ b/.opencode/agents/code-quality-guardian.md @@ -316,7 +316,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf ## Branch Strategy -Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b `) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging. +Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b `), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head `. Wait for CI to pass before merging. ## Dependency Installation diff --git a/.opencode/agents/debugger.md b/.opencode/agents/debugger.md index d5a9fd1..10c7822 100644 --- a/.opencode/agents/debugger.md +++ b/.opencode/agents/debugger.md @@ -14,10 +14,9 @@ Diagnose crashes, memory errors, hangs, and logic bugs. You use structured debug ### Memory Errors ```bash # AddressSanitizer (fast, recommended first) -cmake -B build -S . -DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer" \ - -DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address" -cmake --build build -j$(nproc) -./build/client # or ./build/server +cmake -B build-asan -S . -DSANITIZER=address +cmake --build build-asan -j$(nproc) +./build-asan/client # or ./build-asan/server -p 8080 --allow-unauthenticated # Valgrind (slower, more thorough) valgrind --leak-check=full --show-leak-kinds=all --track-origins=yes \ @@ -32,10 +31,9 @@ valgrind --tool=drd ./build/client ... ### Thread Sanitizer ```bash -cmake -B build -S . -DCMAKE_C_FLAGS="-fsanitize=thread" \ - -DCMAKE_EXE_LINKER_FLAGS="-fsanitize=thread" -cmake --build build -j$(nproc) -./build/tests +cmake -B build-tsan -S . -DSANITIZER=thread +cmake --build build-tsan -j$(nproc) +./build-tsan/tests ``` ### GDB @@ -143,7 +141,7 @@ gprof ./build/client gmon.out ### Step 5: Verify - Run `./build/tests` (unit tests) -- Run `python3 test.py` (integration tests) +- Run `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"` (integration tests) - Run under valgrind again to confirm clean - Test under ASan again @@ -162,7 +160,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf ## Branch Strategy -Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b `) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging. +Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b `), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head `. Wait for CI to pass before merging. ## Dependency Installation diff --git a/.opencode/agents/doc-generator.md b/.opencode/agents/doc-generator.md index 1062c6c..48d2ad0 100644 --- a/.opencode/agents/doc-generator.md +++ b/.opencode/agents/doc-generator.md @@ -96,7 +96,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf ## Branch Strategy -Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b `) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging. +Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b `), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head `. Wait for CI to pass before merging. ## Dependency Installation diff --git a/.opencode/agents/feature-scout.md b/.opencode/agents/feature-scout.md index 449e746..43521c4 100644 --- a/.opencode/agents/feature-scout.md +++ b/.opencode/agents/feature-scout.md @@ -16,12 +16,18 @@ Scan the codebase for patterns that suggest new feature opportunities. You ident ### Module Map ``` src/client/ Client-side: CLI parsing, scanning, sending - client_cli.c Entry point, argument parsing, config setup + client_cli.c Entry point, OPTION_TABLE parser, config setup + usage.c Usage/help text (authoritative CLI flag list) client_send.c Transfer orchestration, pipeline management + client_validation.c Destination/CLI validation scanner.c BFS directory traversal, chunk building + change_list.c File change-list bookkeeping src/server/ Server-side: listening, receiving, writing server.c TCP accept loop, per-connection handling + server_cli.c Server option-table CLI parsing + receiver.c Receiver-side file handling + receiver_pipeline.c Receiver worker pipeline src/shared/ Shared libraries (used by both client and server) protocol.c/h Wire protocol: status codes, send/receive primitives @@ -32,40 +38,63 @@ src/shared/ Shared libraries (used by both client and server) data.c/h Generic buffer type (Data) metadata.c/h File metadata (mode, uid, gid, mtime) file.c/h File representation + file_send.c/h Sender-side file transfer + file_receive.c/h Receiver-side file transfer + file_list.c/h File list model + file_store.c/h Destination file store array_list.c/h Dynamic array + delta.c/h Delta transfer algorithm + checksum.c/h Whole-file/block checksums (xxHash, md5) + filter.c/h rsync-style filter rules + batch.c/h Batch files (--write-batch/--read-batch) + charset.c/h Filename charset conversion (--iconv) + chmod.c/h Permission modification (--chmod) + xattr.c/h Extended attributes + hardlink.c/h Hard-link handling + identity.c/h uid/gid mapping (--usermap/--groupmap/--chown) + credentials.c/h Daemon credentials + daemon_conf.c/h Daemon module configuration + motd.c/h Daemon MOTD + delay_updates.c/h Delayed update staging + stop_condition.c/h Stop-after/stop-at handling transport_tcp.c/h TCP client/server with sendfile() zero-copy transport_ssh.c/h SSH transport with ControlMaster transport_tls.c/h TLS encryption via OpenSSL multiprocessing.c/h Fork-based concurrency log.c/h Logging utilities utils.c/h Shared utilities + file_types.h Shared file type definitions ``` -### Existing CLI Flags (from client_cli.c) +### Existing CLI Flags (authoritative source: `src/client/usage.c`) ``` ---source-dir Source directory to sync (required) ---dest-dir Destination directory on server (required) ---host Server hostname/IP (required) ---port Server TCP port ---server-mode Listen as server ---use-compression, -c Enable zstd compression ---use-multithreading, -m Enable multithreaded transfer ---use-sendfile, -s Use sendfile() zero-copy TCP ---use-ssh, -S Use SSH transport ---use-tls, -T Enable TLS encryption ---cert TLS certificate file ---key TLS key file ---ca TLS CA certificate file ---insecure Skip TLS verification ---bwlimit Bandwidth limit ---delete Delete files not in source ---include Include filter pattern ---exclude Exclude filter pattern ---dry-run Print what would be transferred ---save-to-disk Save transferred files to disk (for server tests) +--source-dir Source directory +--dest-dir Destination directory on server +--server-host Server IP address (default: 127.0.0.1) +--server-port Server port (default: 8080); --port is an alias +-c, --checksum Verify content by checksum instead of size+mtime +-z, --compress [level] Enable compression (level 1-22, default 5) +-j, --threads[=N] Enable multithreaded scanner/loader/sender pipeline +--chunk-serialization Enable chunk serialization (long form only) +--sendfile sendfile() zero-copy (TCP only; long form only) +-s, --secluded-args Protect-args compatibility option (no effect) +--tls Enable TLS encryption; --cert/--key/--ca give PEMs +--bwlimit Bandwidth limit in kilobytes per second +--delete Delete files on receiver not in source +--incremental Skip files unchanged since last transfer +--delta Delta transfer for changed files (needs --incremental) +-f, --filter=RULE rsync-style filter rule (+/- include/exclude) +--exclude Exclude files matching pattern +--include Only include files matching pattern +-m, --prune-empty-dirs Do not transfer empty directory entries +-n, --dry-run Show what would be transferred +--save-to-disk Write received files to disk --version Print version and exit ---help Print help +--help Show help ``` +> Always confirm the current flags with `./build/client --help`; the table above +> is a representative subset. `src/client/usage.c` is the authoritative list and +> `OPTION_TABLE` in `src/client/client_cli.c` is the parser (there is no `getopt*`). ## Feature Scout Checklist @@ -288,7 +317,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf ## Branch Strategy -Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b `) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging. +Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b `), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head `. Wait for CI to pass before merging. ## Dependency Installation diff --git a/.opencode/agents/integrator.md b/.opencode/agents/integrator.md index eea51ae..567729e 100644 --- a/.opencode/agents/integrator.md +++ b/.opencode/agents/integrator.md @@ -35,13 +35,14 @@ mkdir -p /tmp/fastsync_test/src echo "test content" > /tmp/fastsync_test/src/file.txt # Start server -./build/server & +./build/server -p 8080 --allow-unauthenticated & SERVER_PID=$! sleep 0.5 # Run client ./build/client --source-dir /tmp/fastsync_test/src \ --dest-dir /tmp/fastsync_test/dst \ + --server-port 8080 \ --save-to-disk # Verify @@ -156,7 +157,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf ## Branch Strategy -Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b `) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging. +Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b `), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head `. Wait for CI to pass before merging. ## Dependency Installation diff --git a/.opencode/agents/issue-creator.md b/.opencode/agents/issue-creator.md index c7296da..65ced1e 100644 --- a/.opencode/agents/issue-creator.md +++ b/.opencode/agents/issue-creator.md @@ -1,5 +1,5 @@ --- -description: Top-level orchestrator that analyzes the FastSync codebase by delegating to specialized sub-agents and creates GitHub issues from their findings. +description: Top-level orchestrator that analyzes the FastSync codebase by delegating to specialized sub-agents and creates Gitea issues from their findings. mode: subagent --- @@ -12,7 +12,7 @@ You are the primary orchestrator agent. Your job is to: 2. Decide which specialized sub-agents to dispatch for analysis 3. Delegate analysis work using the task tool 4. Receive structured findings from sub-agents -5. Create GitHub issues from those findings using `gh issue create` +5. Create Gitea issues from those findings using `tea issues create` 6. Coordinate the overall analysis workflow end-to-end > **Environment rule:** for CI, dependency installation must use the project's custom Docker image (repo-root `Dockerfile`, same as CI). For local development, use `nix-shell` (see `README.md`). See `AGENTS.md`. @@ -97,7 +97,7 @@ First, read the repository structure to understand what exists: ### Phase 2: Determine Analysis Scope Based on what the user requests or what needs attention: - **New features wanted?** → Dispatch `feature-scout` sub-agent -- **Security audit needed?** → Dispatch `security-screener` sub-agent +- **Security audit needed?** → Dispatch `security-auditor` sub-agent - **Code quality review?** → Dispatch `code-quality-guardian` sub-agent - **All of the above?** → Run all three in parallel @@ -110,7 +110,7 @@ Context: ``` ``` -Task: Ask the security-screener agent to analyze the codebase. +Task: Ask the security-auditor agent to analyze the codebase. Context: ``` @@ -138,14 +138,14 @@ Each sub-agent returns findings in this structured format: - **Labels**: comma-separated labels for the issue ``` -### Phase 5: Create GitHub Issues -For each finding, create a GitHub issue: +### Phase 5: Create Gitea Issues +For each finding, create a Gitea issue: ```bash -gh issue create \ +tea issues create --repo TapTap/FastSync \ --title "" \ - --label "" \ - --body "## Description + --labels "" \ + --description "## Description ## Location @@ -175,11 +175,13 @@ _This issue was automatically generated by the issue-creator agent._" ### Duplicate Detection Before creating an issue: -1. Check existing open issues: `gh issue list --state open --label "