From 5334397b817babe93c78c0a2a3bb011f4d4b60fb Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:23:58 +0200 Subject: [PATCH 1/8] feat(daemon): add shared cross-process connection registry The daemon forks one child per accepted connection, so per-module and per-source accounting must live in state shared across the children. Add a fixed-size registry carved from an anonymous shared mapping (mmap(MAP_SHARED|MAP_ANONYMOUS)) created before the accept loop: a slot lifecycle (FREE/CLAIMED/REGISTERED) with parent claim/reclaim and a lock-free, open-addressed per-source table for the per-host occupancy and the shared auth-failure counter. C11 atomics only; no pthread locks across fork. Unit tests cover slot exhaustion, the module/host caps, pid reclaim and fork-shared visibility. --- CMakeLists.txt | 2 + src/shared/daemon_limits.c | 356 +++++++++++++++++++++++++++++++++++++ src/shared/daemon_limits.h | 102 +++++++++++ tests/runner.c | 2 + tests/test_daemon_limits.c | 194 ++++++++++++++++++++ tests/test_daemon_limits.h | 6 + 6 files changed, 662 insertions(+) create mode 100644 src/shared/daemon_limits.c create mode 100644 src/shared/daemon_limits.h create mode 100644 tests/test_daemon_limits.c create mode 100644 tests/test_daemon_limits.h diff --git a/CMakeLists.txt b/CMakeLists.txt index e61b0fa..479cbca 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -86,6 +86,7 @@ set(SHARED_SRCS src/shared/config.c src/shared/credentials.c src/shared/daemon_conf.c + src/shared/daemon_limits.c src/shared/data.c src/shared/delay_updates.c src/shared/delta.c @@ -205,6 +206,7 @@ set(TEST_SRCS tests/test_config.c tests/test_credentials.c tests/test_daemon_conf.c + tests/test_daemon_limits.c tests/test_data.c tests/test_delay_updates.c tests/test_delta.c diff --git a/src/shared/daemon_limits.c b/src/shared/daemon_limits.c new file mode 100644 index 0000000..2ba7e4e --- /dev/null +++ b/src/shared/daemon_limits.c @@ -0,0 +1,356 @@ +#include "daemon_limits.h" +#include +#include +#include +#include +#include +#include +#include +#include + +/* Slot lifecycle states (stored in slot_state). */ +enum { + SLOT_FREE = 0, + SLOT_CLAIMED = 1, + SLOT_REGISTERED = 2, +}; + +/* The registry header lives at the base of the shared mapping; the pointer + * fields point at the arrays carved out of the same mapping. Absolute pointers + * remain valid in a forked child because fork() clones the address space and + * mapping, so parent and child observe the same virtual addresses. */ +struct DaemonLimitRegistry { + int max_slots; + int module_count; + int host_slots; /* power of two; 1 when no per-source tracking is needed */ + int per_host_cap; + int lockout_threshold; + int lockout_duration_sec; + size_t map_size; + _Atomic int* slot_state; + _Atomic int* slot_pid; + _Atomic int* slot_module; + _Atomic int* slot_host; /* per-source table bucket, or -1 */ + _Atomic int* module_active; + _Atomic uint64_t* host_key; /* 0 == empty bucket */ + _Atomic int* host_active; + _Atomic int* host_fail; + _Atomic long long* host_until; /* epoch seconds the lockout expires */ +}; + +static size_t round_up(size_t n, size_t align) { + return (n + align - 1) & ~(align - 1); +} + +static size_t next_pow2(size_t n) { + size_t p = 1; + while (p < n) + p <<= 1; + return p; +} + +/* Parse a numeric IPv4/IPv6 peer string into family + raw bytes. */ +static bool parse_peer_ip(const char* peer_ip, int* family, unsigned char* bytes) { + if (!peer_ip || *peer_ip == '\0') + return false; + struct in_addr v4; + if (inet_pton(AF_INET, peer_ip, &v4) == 1) { + memcpy(bytes, &v4, sizeof(v4)); + *family = AF_INET; + return true; + } + struct in6_addr v6; + if (inet_pton(AF_INET6, peer_ip, &v6) == 1) { + memcpy(bytes, &v6, sizeof(v6)); + *family = AF_INET6; + return true; + } + return false; +} + +uint64_t daemon_limits_host_hash(const char* peer_ip, bool* ok) { + if (ok) + *ok = false; + unsigned char bytes[16]; + int family = AF_UNSPEC; + if (!parse_peer_ip(peer_ip, &family, bytes)) + return 0; + uint64_t hash = 14695981039346656037ULL ^ (uint64_t)(uint32_t)family; + size_t length = family == AF_INET ? 4 : 16; + for (size_t i = 0; i < length; i++) { + hash ^= bytes[i]; + hash *= 1099511628211ULL; + } + if (hash == 0) + hash = 0x9e3779b97f4a7c15ULL; + if (ok) + *ok = true; + return hash; +} + +/* Find the bucket holding `peer_ip`, or -1 when it has no entry. */ +static int host_lookup(DaemonLimitRegistry* registry, const char* peer_ip) { + bool ok = false; + uint64_t key = daemon_limits_host_hash(peer_ip, &ok); + if (!ok) + return -1; + size_t mask = (size_t)registry->host_slots - 1; + size_t start = (size_t)(key & mask); + for (size_t i = 0; i < (size_t)registry->host_slots; i++) { + size_t idx = (start + i) & mask; + uint64_t current = atomic_load_explicit(®istry->host_key[idx], memory_order_acquire); + if (current == key) + return (int)idx; + if (current == 0) + return -1; /* no tombstones: an empty bucket ends the probe chain */ + } + return -1; +} + +/* Find or insert the bucket for `peer_ip`. Insertion is a lock-free CAS so two + * forked children racing on the same source converge on one bucket. Returns -1 + * when the table is full or the address is unparseable (callers fail open: the + * global/module caps and ACLs still apply). */ +static int host_intern(DaemonLimitRegistry* registry, const char* peer_ip) { + bool ok = false; + uint64_t key = daemon_limits_host_hash(peer_ip, &ok); + if (!ok) + return -1; + size_t mask = (size_t)registry->host_slots - 1; + size_t start = (size_t)(key & mask); + for (size_t i = 0; i < (size_t)registry->host_slots; i++) { + size_t idx = (start + i) & mask; + uint64_t current = atomic_load_explicit(®istry->host_key[idx], memory_order_acquire); + if (current == key) + return (int)idx; + if (current == 0) { + uint64_t expected = 0; + if (atomic_compare_exchange_strong_explicit(®istry->host_key[idx], &expected, key, + memory_order_acq_rel, memory_order_acquire)) + return (int)idx; + if (atomic_load_explicit(®istry->host_key[idx], memory_order_acquire) == key) + return (int)idx; + } + } + return -1; +} + +DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int per_host_cap, + int lockout_threshold, int lockout_duration_sec) { + if (max_slots < DAEMON_LIMITS_MIN_SLOTS) + max_slots = DAEMON_LIMITS_MIN_SLOTS; + if (max_slots > DAEMON_LIMITS_MAX_SLOTS) + max_slots = DAEMON_LIMITS_MAX_SLOTS; + if (module_count < 1) + module_count = 1; + if (per_host_cap < 0) + per_host_cap = 0; + if (lockout_threshold < 0) + lockout_threshold = 0; + if (lockout_duration_sec < 0) + lockout_duration_sec = 0; + + bool need_hosts = per_host_cap > 0 || (lockout_threshold > 0 && lockout_duration_sec > 0); + int host_slots = 1; + if (need_hosts) { + size_t want = (size_t)max_slots * 4; + if (want < 64) + want = 64; + if (want > DAEMON_LIMITS_MAX_HOST_SLOTS) + want = DAEMON_LIMITS_MAX_HOST_SLOTS; + host_slots = (int)next_pow2(want); + } + + size_t header = round_up(sizeof(DaemonLimitRegistry), 16); + size_t slot_bytes = + round_up((size_t)max_slots * sizeof(_Atomic int), 16) * 4; /* state,pid,module,host */ + size_t module_bytes = round_up((size_t)module_count * sizeof(_Atomic int), 16); + size_t host_key_bytes = round_up((size_t)host_slots * sizeof(_Atomic uint64_t), 16); + size_t host_int_bytes = round_up((size_t)host_slots * sizeof(_Atomic int), 16) * 2; + size_t host_until_bytes = round_up((size_t)host_slots * sizeof(_Atomic long long), 16); + size_t total = + header + slot_bytes + module_bytes + host_key_bytes + host_int_bytes + host_until_bytes + 16; + + void* map = mmap(NULL, total, PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, -1, 0); + if (map == MAP_FAILED) + return NULL; + memset(map, 0, total); + + DaemonLimitRegistry* registry = (DaemonLimitRegistry*)map; + registry->max_slots = max_slots; + registry->module_count = module_count; + registry->host_slots = host_slots; + registry->per_host_cap = per_host_cap; + registry->lockout_threshold = lockout_threshold; + registry->lockout_duration_sec = lockout_duration_sec; + registry->map_size = total; + + unsigned char* cursor = (unsigned char*)map + header; + registry->slot_state = (atomic_int*)cursor; + cursor += (size_t)max_slots * sizeof(_Atomic int); + registry->slot_pid = (atomic_int*)cursor; + cursor += (size_t)max_slots * sizeof(_Atomic int); + registry->slot_module = (atomic_int*)cursor; + cursor += (size_t)max_slots * sizeof(_Atomic int); + registry->slot_host = (atomic_int*)cursor; + cursor += (size_t)max_slots * sizeof(_Atomic int); + registry->module_active = (atomic_int*)cursor; + cursor += (size_t)module_count * sizeof(_Atomic int); + cursor = (unsigned char*)round_up((size_t)(uintptr_t)cursor, 16); + registry->host_key = (_Atomic uint64_t*)cursor; + cursor += (size_t)host_slots * sizeof(_Atomic uint64_t); + registry->host_active = (atomic_int*)cursor; + cursor += (size_t)host_slots * sizeof(_Atomic int); + registry->host_fail = (atomic_int*)cursor; + cursor += (size_t)host_slots * sizeof(_Atomic int); + cursor = (unsigned char*)round_up((size_t)(uintptr_t)cursor, 16); + registry->host_until = (atomic_llong*)cursor; + + for (int i = 0; i < max_slots; i++) { + atomic_store(®istry->slot_module[i], -1); + atomic_store(®istry->slot_host[i], -1); + } + return registry; +} + +void daemon_limits_destroy(DaemonLimitRegistry* registry) { + if (!registry) + return; + munmap(registry, registry->map_size); +} + +int daemon_limits_claim_slot(DaemonLimitRegistry* registry) { + if (!registry) + return DAEMON_LIMITS_NO_SLOT; + for (int i = 0; i < registry->max_slots; i++) { + int expected = SLOT_FREE; + if (atomic_compare_exchange_strong(®istry->slot_state[i], &expected, SLOT_CLAIMED)) { + atomic_store(®istry->slot_pid[i], 0); + atomic_store(®istry->slot_module[i], -1); + atomic_store(®istry->slot_host[i], -1); + return i; + } + } + return DAEMON_LIMITS_NO_SLOT; +} + +void daemon_limits_set_slot_pid(DaemonLimitRegistry* registry, int slot, long pid) { + if (!registry || slot < 0 || slot >= registry->max_slots) + return; + atomic_store(®istry->slot_pid[slot], (int)pid); +} + +void daemon_limits_reclaim_slot(DaemonLimitRegistry* registry, int slot) { + if (!registry || slot < 0 || slot >= registry->max_slots) + return; + int previous = + atomic_exchange_explicit(®istry->slot_state[slot], SLOT_FREE, memory_order_acq_rel); + if (previous == SLOT_REGISTERED) { + int module = atomic_load(®istry->slot_module[slot]); + int host = atomic_load(®istry->slot_host[slot]); + if (module >= 0 && module < registry->module_count) { + int current = atomic_load(®istry->module_active[module]); + while (current > 0 && + !atomic_compare_exchange_weak(®istry->module_active[module], ¤t, current - 1)) + ; + } + if (host >= 0 && host < registry->host_slots) { + int current = atomic_load(®istry->host_active[host]); + while (current > 0 && + !atomic_compare_exchange_weak(®istry->host_active[host], ¤t, current - 1)) + ; + } + } + atomic_store(®istry->slot_pid[slot], 0); +} + +void daemon_limits_reclaim_pid(DaemonLimitRegistry* registry, long pid) { + if (!registry || pid <= 0) + return; + for (int i = 0; i < registry->max_slots; i++) { + if (atomic_load(®istry->slot_state[i]) == SLOT_FREE) + continue; + if (atomic_load(®istry->slot_pid[i]) == (int)pid) { + daemon_limits_reclaim_slot(registry, i); + return; + } + } +} + +DaemonLimitResult daemon_limits_register(DaemonLimitRegistry* registry, int slot, int module_index, + const char* peer_ip, int module_cap) { + if (!registry || slot < 0 || slot >= registry->max_slots) + return DAEMON_LIMIT_UNAVAILABLE; + if (module_index < 0 || module_index >= registry->module_count) + return DAEMON_LIMIT_UNAVAILABLE; + if (atomic_load_explicit(®istry->slot_state[slot], memory_order_acquire) != SLOT_CLAIMED) + return DAEMON_LIMIT_UNAVAILABLE; + + int host = -1; + if (registry->per_host_cap > 0 || registry->lockout_threshold > 0) + host = host_intern(registry, peer_ip); + + int module_count = atomic_fetch_add(®istry->module_active[module_index], 1) + 1; + if (module_cap > 0 && module_count > module_cap) { + atomic_fetch_sub(®istry->module_active[module_index], 1); + return DAEMON_LIMIT_MODULE_FULL; + } + if (host >= 0) { + int host_count = atomic_fetch_add(®istry->host_active[host], 1) + 1; + if (registry->per_host_cap > 0 && host_count > registry->per_host_cap) { + atomic_fetch_sub(®istry->host_active[host], 1); + atomic_fetch_sub(®istry->module_active[module_index], 1); + return DAEMON_LIMIT_HOST_FULL; + } + } + atomic_store(®istry->slot_module[slot], module_index); + atomic_store(®istry->slot_host[slot], host); + atomic_store_explicit(®istry->slot_state[slot], SLOT_REGISTERED, memory_order_release); + return DAEMON_LIMIT_OK; +} + +bool daemon_limits_auth_locked(DaemonLimitRegistry* registry, const char* peer_ip, + int* seconds_remaining) { + if (!registry || registry->lockout_threshold <= 0 || registry->lockout_duration_sec <= 0) + return false; + int bucket = host_lookup(registry, peer_ip); + if (bucket < 0) + return false; + long long until = atomic_load(®istry->host_until[bucket]); + long long now = (long long)time(NULL); + if (until > now) { + if (seconds_remaining) + *seconds_remaining = (int)(until - now); + return true; + } + if (until != 0) { + /* The previous lockout has expired: clear the stale counter so the source + * gets a fresh allowance. */ + atomic_store(®istry->host_fail[bucket], 0); + atomic_store(®istry->host_until[bucket], 0); + } + return false; +} + +void daemon_limits_auth_record_failure(DaemonLimitRegistry* registry, const char* peer_ip) { + if (!registry || registry->lockout_threshold <= 0 || registry->lockout_duration_sec <= 0) + return; + int bucket = host_intern(registry, peer_ip); + if (bucket < 0) + return; + int failures = atomic_fetch_add(®istry->host_fail[bucket], 1) + 1; + if (failures >= registry->lockout_threshold) { + long long now = (long long)time(NULL); + atomic_store(®istry->host_until[bucket], now + (long long)registry->lockout_duration_sec); + } +} + +void daemon_limits_auth_record_success(DaemonLimitRegistry* registry, const char* peer_ip) { + if (!registry) + return; + int bucket = host_lookup(registry, peer_ip); + if (bucket < 0) + return; + atomic_store(®istry->host_fail[bucket], 0); + atomic_store(®istry->host_until[bucket], 0); +} diff --git a/src/shared/daemon_limits.h b/src/shared/daemon_limits.h new file mode 100644 index 0000000..47bfb0b --- /dev/null +++ b/src/shared/daemon_limits.h @@ -0,0 +1,102 @@ +#ifndef DAEMON_LIMITS_H +#define DAEMON_LIMITS_H + +#include +#include +#include + +/* Cross-process daemon connection registry. + * + * The daemon listener forks ONE child per accepted connection, so any + * per-module / per-source accounting must live in state shared across the + * forked children. This module owns a fixed-size registry carved out of an + * anonymous shared mapping (mmap(MAP_SHARED | MAP_ANONYMOUS)) created by the + * accept-loop PARENT before it forks; every child inherits the mapping (and the + * pointer to it) across fork(). + * + * Rules: + * - ONLY C11 atomics (atomic_*); never mtx_t/pthread locks, which can deadlock + * in a forked child if another thread held them at fork time. + * - No heap allocation after fork: the mapping is fixed-size and all access is + * atomic load/store/CAS over preallocated arrays. + * + * Slot lifecycle (the parent reclaims even when a child is SIGKILLed): + * FREE --(parent claim_slot)--> CLAIMED + * CLAIMED --(child register)--> REGISTERED + * any --(parent reclaim)--> FREE + * The child records its module index and per-source bucket into the slot before + * publishing REGISTERED; the parent's SIGCHLD handler matches the reaped pid to + * the slot and, when REGISTERED, decrements the module/per-source counters. + * A child killed before registering holds no counts, so reclaiming a CLAIMED + * slot only frees the slot. + * + * Per-source identity is the normalized numeric peer IP (IPv4-mapped IPv6 is + * already collapsed to IPv4 by utils_fd_peer_ip); it is interned into an + * open-addressed, linear-probing table keyed by a 64-bit hash. The same table + * also carries the cross-process auth-failure counter and lockout deadline. + */ + +typedef struct DaemonLimitRegistry DaemonLimitRegistry; + +/* Result of a per-connection admission check. */ +typedef enum { + DAEMON_LIMIT_OK = 0, /* admitted; slot is now REGISTERED */ + DAEMON_LIMIT_MODULE_FULL, /* module's `max connections` cap reached */ + DAEMON_LIMIT_HOST_FULL, /* global `max connections per host` cap reached */ + DAEMON_LIMIT_UNAVAILABLE, /* registry/slot unusable (caller fails open) */ +} DaemonLimitResult; + +/* Bounds for registry sizing. A slot is one concurrently live child. */ +#define DAEMON_LIMITS_MIN_SLOTS 16 +#define DAEMON_LIMITS_MAX_SLOTS 65536 +#define DAEMON_LIMITS_MAX_HOST_SLOTS 65536 +#define DAEMON_LIMITS_NO_SLOT (-1) + +/* Create the shared registry in the calling (parent) process. `max_slots` is + * the number of concurrently live children to track (clamped to + * [DAEMON_LIMITS_MIN_SLOTS, DAEMON_LIMITS_MAX_SLOTS]); `module_count` is the + * number of daemon modules (clamped to >= 1); `per_host_cap` and the lockout + * pair come from the daemon config (0 disables). Returns NULL on failure (e.g. + * mmap allocation); callers must degrade gracefully (global cap + ACLs still + * apply). */ +DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int per_host_cap, + int lockout_threshold, int lockout_duration_sec); + +/* Unmap the registry. Only the creating process may call this. */ +void daemon_limits_destroy(DaemonLimitRegistry* registry); + +/* Parent side: reserve a slot for the next fork. Returns the slot index or + * DAEMON_LIMITS_NO_SLOT when every slot is in use. */ +int daemon_limits_claim_slot(DaemonLimitRegistry* registry); +/* Parent side: record the forked child's pid in a claimed slot. */ +void daemon_limits_set_slot_pid(DaemonLimitRegistry* registry, int slot, long pid); +/* Parent side: release a slot, decrementing the module/per-source counters when + * the slot was actually REGISTERED. Idempotent. */ +void daemon_limits_reclaim_slot(DaemonLimitRegistry* registry, int slot); +/* Parent SIGCHLD side: reclaim the slot owned by `pid` (no-op when not found). */ +void daemon_limits_reclaim_pid(DaemonLimitRegistry* registry, long pid); + +/* Child side: admit the connection for `module_index` from `peer_ip`. Always + * tracks the module/per-source occupancy (so the parent's reclaim is + * symmetric); when `module_cap` > 0 it additionally enforces the per-module + * cap. Returns DAEMON_LIMIT_OK and publishes the slot, or a refusal reason. */ +DaemonLimitResult daemon_limits_register(DaemonLimitRegistry* registry, int slot, int module_index, + const char* peer_ip, int module_cap); + +/* Child side: true when `peer_ip` is currently locked out after too many failed + * authentications. `seconds_remaining` may be NULL. */ +bool daemon_limits_auth_locked(DaemonLimitRegistry* registry, const char* peer_ip, + int* seconds_remaining); +/* Child side: count one failed authentication for `peer_ip`; once the threshold + * is reached the source is locked out for the configured duration. */ +void daemon_limits_auth_record_failure(DaemonLimitRegistry* registry, const char* peer_ip); +/* Child side: clear the failure counter/lockout for a source that authenticated + * successfully (no-op when the source has no table entry). */ +void daemon_limits_auth_record_success(DaemonLimitRegistry* registry, const char* peer_ip); + +/* Pure helper: 64-bit FNV-1a hash of a numeric peer IP plus its family, used to + * index the per-source table. *ok is set false (and 0 returned) for a NULL or + * non-numeric address. Exposed for unit testing. */ +uint64_t daemon_limits_host_hash(const char* peer_ip, bool* ok); + +#endif diff --git a/tests/runner.c b/tests/runner.c index 9e12323..9549220 100644 --- a/tests/runner.c +++ b/tests/runner.c @@ -9,6 +9,7 @@ #include "test_credentials.h" #include "test_data.h" #include "test_daemon_conf.h" +#include "test_daemon_limits.h" #include "test_delay_updates.h" #include "test_delta.h" #include "test_file.h" @@ -85,6 +86,7 @@ int main() { RUN_TEST(test_client_cli); RUN_TEST(test_server); RUN_TEST(test_daemon_conf); + RUN_TEST(test_daemon_limits); RUN_TEST(test_motd); RUN_TEST(test_server_cli); RUN_TEST(test_fuzz_smoke); diff --git a/tests/test_daemon_limits.c b/tests/test_daemon_limits.c new file mode 100644 index 0000000..2815e0a --- /dev/null +++ b/tests/test_daemon_limits.c @@ -0,0 +1,194 @@ +#include "test_daemon_limits.h" +#include "daemon_limits.h" +#include "test_utils.h" +#include +#include +#include + +/* The per-source hash is a pure helper: numeric addresses hash to a nonzero, + * stable value and unparseable input reports failure. */ +static void test_daemon_limits_host_hash() { + bool ok = false; + uint64_t v4 = daemon_limits_host_hash("127.0.0.1", &ok); + EXPECT_TRUE(ok); + EXPECT_TRUE(v4 != 0); + EXPECT_EQ_INT((int)(daemon_limits_host_hash("127.0.0.1", NULL) == v4), 1); + + bool ok6 = false; + uint64_t v6 = daemon_limits_host_hash("2001:db8::1", &ok6); + EXPECT_TRUE(ok6); + EXPECT_TRUE(v6 != 0); + /* Distinct textual forms of different addresses must differ. */ + EXPECT_TRUE(v4 != v6); + + bool bad = true; + EXPECT_TRUE(daemon_limits_host_hash("not-an-ip", &bad) == 0); + EXPECT_FALSE(bad); + bad = true; + EXPECT_TRUE(daemon_limits_host_hash(NULL, &bad) == 0); + EXPECT_FALSE(bad); + bad = true; + EXPECT_TRUE(daemon_limits_host_hash("", &bad) == 0); + EXPECT_FALSE(bad); +} + +/* Slot reservation is a plain parent-side resource: claim until exhausted, + * reclaim, then claim again. */ +static void test_daemon_limits_slots() { + DaemonLimitRegistry* registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 2, 0, 0, 0); + EXPECT_NOT_NULL(registry); + int slots[DAEMON_LIMITS_MIN_SLOTS]; + for (int i = 0; i < DAEMON_LIMITS_MIN_SLOTS; i++) { + slots[i] = daemon_limits_claim_slot(registry); + EXPECT_EQ_INT(slots[i], i); + } + EXPECT_EQ_INT(daemon_limits_claim_slot(registry), DAEMON_LIMITS_NO_SLOT); + daemon_limits_reclaim_slot(registry, slots[3]); + int reclaimed = daemon_limits_claim_slot(registry); + EXPECT_EQ_INT(reclaimed, slots[3]); + daemon_limits_destroy(registry); +} + +/* Per-module accounting: the cap is enforced across slots and a reclaimed slot + * frees a module count. */ +static void test_daemon_limits_module_cap() { + DaemonLimitRegistry* registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 2, 0, 0, 0); + EXPECT_NOT_NULL(registry); + + int slot0 = daemon_limits_claim_slot(registry); + int slot1 = daemon_limits_claim_slot(registry); + int slot2 = daemon_limits_claim_slot(registry); + int slot3 = daemon_limits_claim_slot(registry); + EXPECT_TRUE(slot0 >= 0 && slot1 >= 0 && slot2 >= 0 && slot3 >= 0); + + EXPECT_EQ_INT(daemon_limits_register(registry, slot0, 0, "10.0.0.1", 2), DAEMON_LIMIT_OK); + EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.2", 2), DAEMON_LIMIT_OK); + EXPECT_EQ_INT(daemon_limits_register(registry, slot2, 0, "10.0.0.3", 2), + DAEMON_LIMIT_MODULE_FULL); + /* A different module has its own counter. */ + EXPECT_EQ_INT(daemon_limits_register(registry, slot2, 1, "10.0.0.3", 2), DAEMON_LIMIT_OK); + /* A module cap of 0 is unlimited. */ + EXPECT_EQ_INT(daemon_limits_register(registry, slot3, 0, "10.0.0.3", 0), DAEMON_LIMIT_OK); + + daemon_limits_reclaim_slot(registry, slot0); + daemon_limits_reclaim_slot(registry, slot1); + int slot4 = daemon_limits_claim_slot(registry); + EXPECT_TRUE(slot4 >= 0); + EXPECT_EQ_INT(daemon_limits_register(registry, slot4, 0, "10.0.0.4", 2), DAEMON_LIMIT_OK); + + daemon_limits_destroy(registry); +} + +/* Per-source accounting: the same peer hits the cap, a different peer does not. */ +static void test_daemon_limits_host_cap() { + DaemonLimitRegistry* registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 1, 1, 0, 0); + EXPECT_NOT_NULL(registry); + + int slot0 = daemon_limits_claim_slot(registry); + int slot1 = daemon_limits_claim_slot(registry); + int slot2 = daemon_limits_claim_slot(registry); + EXPECT_TRUE(slot0 >= 0 && slot1 >= 0 && slot2 >= 0); + + EXPECT_EQ_INT(daemon_limits_register(registry, slot0, 0, "10.0.0.1", 0), DAEMON_LIMIT_OK); + EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.1", 0), DAEMON_LIMIT_HOST_FULL); + EXPECT_EQ_INT(daemon_limits_register(registry, slot2, 0, "10.0.0.2", 0), DAEMON_LIMIT_OK); + /* Reclaiming the first source frees its per-host allowance. */ + daemon_limits_reclaim_slot(registry, slot0); + EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.1", 0), DAEMON_LIMIT_OK); + + daemon_limits_destroy(registry); +} + +/* The pid-indexed reclaim is what the parent's SIGCHLD handler uses: a dead + * child's module/source counts must be released. */ +static void test_daemon_limits_reclaim_pid() { + DaemonLimitRegistry* registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 1, 1, 0, 0); + EXPECT_NOT_NULL(registry); + + int slot0 = daemon_limits_claim_slot(registry); + int slot1 = daemon_limits_claim_slot(registry); + EXPECT_TRUE(slot0 >= 0 && slot1 >= 0); + daemon_limits_set_slot_pid(registry, slot0, 4242); + EXPECT_EQ_INT(daemon_limits_register(registry, slot0, 0, "10.0.0.1", 1), DAEMON_LIMIT_OK); + /* Cap (module 1) and per-host (1) are both saturated. */ + EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.1", 1), + DAEMON_LIMIT_MODULE_FULL); + + daemon_limits_reclaim_pid(registry, 4242); + EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.1", 1), DAEMON_LIMIT_OK); + /* Reclaiming an unknown pid is a no-op. */ + daemon_limits_reclaim_pid(registry, 999999); + + daemon_limits_destroy(registry); +} + +/* Cross-process lockout: failures counted in the shared mapping lock the source + * out after the threshold; a success clears it; threshold 0 disables it. */ +static void test_daemon_limits_auth_lockout() { + DaemonLimitRegistry* registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 1, 0, 2, 300); + EXPECT_NOT_NULL(registry); + + int remaining = 0; + EXPECT_FALSE(daemon_limits_auth_locked(registry, "10.0.0.1", &remaining)); + daemon_limits_auth_record_failure(registry, "10.0.0.1"); + EXPECT_FALSE(daemon_limits_auth_locked(registry, "10.0.0.1", &remaining)); + daemon_limits_auth_record_failure(registry, "10.0.0.1"); + EXPECT_TRUE(daemon_limits_auth_locked(registry, "10.0.0.1", &remaining)); + EXPECT_TRUE(remaining > 0 && remaining <= 300); + /* Another source is unaffected. */ + EXPECT_FALSE(daemon_limits_auth_locked(registry, "10.0.0.2", &remaining)); + /* A successful authentication clears the lockout. */ + daemon_limits_auth_record_success(registry, "10.0.0.1"); + EXPECT_FALSE(daemon_limits_auth_locked(registry, "10.0.0.1", &remaining)); + daemon_limits_destroy(registry); + + /* threshold 0 disables the lockout entirely. */ + registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 1, 0, 0, 300); + EXPECT_NOT_NULL(registry); + for (int i = 0; i < 50; i++) + daemon_limits_auth_record_failure(registry, "10.0.0.1"); + EXPECT_FALSE(daemon_limits_auth_locked(registry, "10.0.0.1", &remaining)); + daemon_limits_destroy(registry); +} + +/* The registry must be visible across fork(): a child's registration is seen by + * the parent, and the parent's pid reclaim releases it. */ +static void test_daemon_limits_fork_shared() { + if (is_running_under_valgrind()) + return; /* fork + shared mapping is slow/noisy under valgrind */ + DaemonLimitRegistry* registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 1, 0, 0, 0); + EXPECT_NOT_NULL(registry); + + int slot0 = daemon_limits_claim_slot(registry); + EXPECT_TRUE(slot0 >= 0); + pid_t pid = fork(); + if (pid == 0) { + if (daemon_limits_register(registry, slot0, 0, "10.0.0.1", 1) != DAEMON_LIMIT_OK) + _exit(1); + _exit(0); + } + EXPECT_TRUE(pid > 0); + daemon_limits_set_slot_pid(registry, slot0, (long)pid); + int status = 0; + EXPECT_TRUE(waitpid(pid, &status, 0) == pid); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + /* The child's module count is still held in the shared mapping. */ + int slot1 = daemon_limits_claim_slot(registry); + EXPECT_TRUE(slot1 >= 0); + EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.2", 1), + DAEMON_LIMIT_MODULE_FULL); + /* The parent reclaims the dead child's slot by pid. */ + daemon_limits_reclaim_pid(registry, (long)pid); + EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.2", 1), DAEMON_LIMIT_OK); + daemon_limits_destroy(registry); +} + +void test_daemon_limits() { + test_daemon_limits_host_hash(); + test_daemon_limits_slots(); + test_daemon_limits_module_cap(); + test_daemon_limits_host_cap(); + test_daemon_limits_reclaim_pid(); + test_daemon_limits_auth_lockout(); + test_daemon_limits_fork_shared(); +} diff --git a/tests/test_daemon_limits.h b/tests/test_daemon_limits.h new file mode 100644 index 0000000..67e905a --- /dev/null +++ b/tests/test_daemon_limits.h @@ -0,0 +1,6 @@ +#ifndef TEST_DAEMON_LIMITS_H +#define TEST_DAEMON_LIMITS_H + +void test_daemon_limits(); + +#endif From 0abaa62193d5f918d8b7c499c76b87bd5a6755dd Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:24:01 +0200 Subject: [PATCH 2/8] feat(daemon): parse per-host cap and auth lockout config keys Add global keys `max connections per host` (default 0 = unlimited), `auth lockout threshold` (default 10, 0 disables) and `auth lockout duration` (default 300 s, 0 disables). Module `max connections` now accepts 0 as unlimited. Bound the number of [module] sections (DAEMON_CONF_MAX_MODULES) so the shared registry's per-module counter array stays fixed-size; absent keys keep their defaults so old configs still load. --- src/shared/daemon_conf.c | 43 ++++++++++++++++++++++++++- src/shared/daemon_conf.h | 56 +++++++++++++++++++++++++---------- tests/test_daemon_conf.c | 63 ++++++++++++++++++++++++++++++++++++---- 3 files changed, 140 insertions(+), 22 deletions(-) diff --git a/src/shared/daemon_conf.c b/src/shared/daemon_conf.c index 01a759a..300e196 100644 --- a/src/shared/daemon_conf.c +++ b/src/shared/daemon_conf.c @@ -190,6 +190,26 @@ static bool store_max_connections(int* slot, const char* value, const char* modu return true; } +/* Parse a non-negative concurrency cap where 0 means unlimited/disabled + * (per-module `max connections`, `max connections per host`, + * `auth lockout threshold`). Negative/garbage/oversized values are rejected. */ +static bool store_optional_cap(int* slot, const char* value, int max_value, const char* key, + const char* module_name, char* err, size_t err_size) { + char* end = NULL; + errno = 0; + long n = strtol(value, &end, 10); + if (*value == '\0' || errno != 0 || *end != '\0' || n < 0 || n > max_value) { + if (module_name) + set_error(err, err_size, "module '%s': invalid '%s' '%s' (must be 0-%d)", module_name, key, + value, max_value); + else + set_error(err, err_size, "invalid '%s' '%s' (must be 0-%d)", key, value, max_value); + return false; + } + *slot = (int)n; + return true; +} + /* Parse an `auth failure delay` value: 0 (disabled) through the configured cap. */ static bool store_auth_failure_delay(int* slot, const char* value, char* err, size_t err_size) { char* end = NULL; @@ -227,6 +247,9 @@ DaemonConf* daemon_conf_create(void) { conf->global.port = DAEMON_CONF_DEFAULT_PORT; conf->global.max_connections = DAEMON_CONF_DEFAULT_MAX_CONNECTIONS; conf->global.auth_failure_delay_ms = DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS; + conf->global.max_connections_per_host = DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST; + conf->global.auth_lockout_threshold = DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD; + conf->global.auth_lockout_duration_sec = DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC; return conf; } @@ -312,8 +335,20 @@ static bool apply_global_key(DaemonConf* conf, char* key, const char* value, boo } if (key_equals(key, "max connections")) return store_max_connections(&conf->global.max_connections, value, NULL, err, err_size); + if (key_equals(key, "max connections per host")) + return store_optional_cap(&conf->global.max_connections_per_host, value, + DAEMON_CONF_MAX_CONCURRENCY_LIMIT, "max connections per host", NULL, + err, err_size); if (key_equals(key, "auth failure delay")) return store_auth_failure_delay(&conf->global.auth_failure_delay_ms, value, err, err_size); + if (key_equals(key, "auth lockout threshold")) + return store_optional_cap(&conf->global.auth_lockout_threshold, value, + DAEMON_CONF_MAX_CONCURRENCY_LIMIT, "auth lockout threshold", NULL, + err, err_size); + if (key_equals(key, "auth lockout duration")) + return store_optional_cap(&conf->global.auth_lockout_duration_sec, value, + DAEMON_CONF_MAX_AUTH_LOCKOUT_DURATION_SEC, "auth lockout duration", + NULL, err, err_size); if (key_equals(key, "hosts allow")) return store_host_list(&conf->global.hosts_allow, &conf->global.hosts_allow_count, value, "hosts allow", NULL, replace_hosts, err, err_size); @@ -400,7 +435,8 @@ static bool apply_module_key(DaemonModule* module, char* key, char* value, char* return true; } if (key_equals(key, "max connections")) - return store_max_connections(&module->max_connections, value, module->name, err, err_size); + return store_optional_cap(&module->max_connections, value, DAEMON_CONF_MAX_CONCURRENCY_LIMIT, + "max connections", module->name, err, err_size); if (key_equals(key, "hosts allow")) return store_host_list(&module->hosts_allow, &module->hosts_allow_count, value, "hosts allow", false, module->name, err, err_size); @@ -444,6 +480,11 @@ static int open_module(DaemonConf* conf, int* current_module, const char* name, set_error(err, err_size, "duplicate module '%s'", name); return -1; } + if (conf->module_count >= DAEMON_CONF_MAX_MODULES) { + set_error(err, err_size, "too many modules (limit %d); module '%s' rejected", + DAEMON_CONF_MAX_MODULES, name); + return -1; + } DaemonModule* grown = realloc(conf->modules, (size_t)(conf->module_count + 1) * sizeof(DaemonModule)); if (!grown) { diff --git a/src/shared/daemon_conf.h b/src/shared/daemon_conf.h index 3b790a7..d04399e 100644 --- a/src/shared/daemon_conf.h +++ b/src/shared/daemon_conf.h @@ -52,11 +52,10 @@ typedef struct DaemonModule { activities. Without it the daemon refuses all of them. */ char** auth_users; /* `auth users = a,b`; Wave B credential list */ int auth_user_count; - /* `max connections = N` (optional per-module cap). 0 means "not set" - * (inherit the global cap). Parsed, stored, and validated, but NOT enforced - * per-module: connections are counted in the accept-loop parent before the - * client's module is known, so only the global cap is enforced (see - * transport_tcp.c and the Daemon Mode notes in RSYNC_COMPAT.md). */ + /* `max connections = N` (optional per-module cap). 0 means unlimited. The + * per-connection child records the selected module in the shared registry + * (daemon_limits.c) once the config frame names it, so the cap is enforced + * across all forked children; the parent reclaims the slot on SIGCHLD. */ int max_connections; char** hosts_allow; /* `hosts allow = a,b`; host access allow patterns */ int hosts_allow_count; @@ -67,14 +66,25 @@ typedef struct DaemonModule { /* Global (pre-module) scalar keys. `motd file` is parsed and stored but has * no wire effect yet (MOTD display is Wave C). */ typedef struct DaemonConfGlobals { - int port; /* `port`, default DAEMON_CONF_DEFAULT_PORT (873) */ - char* motd_file; /* `motd file`, may be NULL */ - char* address; /* `address` (optional bind address), may be NULL */ - int max_connections; /* `max connections`, default - DAEMON_CONF_DEFAULT_MAX_CONNECTIONS (100) */ - int auth_failure_delay_ms; /* `auth failure delay`, milliseconds; default - DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS */ - char** hosts_allow; /* `hosts allow`; global host access allow patterns */ + int port; /* `port`, default DAEMON_CONF_DEFAULT_PORT (873) */ + char* motd_file; /* `motd file`, may be NULL */ + char* address; /* `address` (optional bind address), may be NULL */ + int max_connections; /* `max connections`, default + DAEMON_CONF_DEFAULT_MAX_CONNECTIONS (100) */ + int auth_failure_delay_ms; /* `auth failure delay`, milliseconds; default + DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS */ + int max_connections_per_host; /* `max connections per host`, concurrent cap per + source IP; default + DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST (0 = + unlimited) */ + int auth_lockout_threshold; /* `auth lockout threshold`, failed attempts from + one source before lockout; default + DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD (0 + disables) */ + int auth_lockout_duration_sec; /* `auth lockout duration`, seconds; default + DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC + (0 disables) */ + char** hosts_allow; /* `hosts allow`; global host access allow patterns */ int hosts_allow_count; char** hosts_deny; /* `hosts deny`; global host access deny patterns */ int hosts_deny_count; @@ -92,11 +102,26 @@ typedef struct DaemonConf { #define DAEMON_CONF_DEFAULT_MAX_CONNECTIONS 100 /* Default `auth failure delay` in milliseconds (0 disables the throttle). */ #define DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS 500 +/* Default `max connections per host` (0 = unlimited). */ +#define DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST 0 +/* Default cross-process auth lockout: 10 failed attempts from one source lock + * it out for 300 s (0 disables either knob). */ +#define DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD 10 +#define DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC 300 +/* Upper bound on a `max connections per host` or `auth lockout threshold` + * value, so a typo cannot size the shared registry absurdly. */ +#define DAEMON_CONF_MAX_CONCURRENCY_LIMIT 1000000 +/* Upper bound on `auth lockout duration` (7 days). */ +#define DAEMON_CONF_MAX_AUTH_LOCKOUT_DURATION_SEC 604800 /* Largest accepted `auth failure delay`, so a typo cannot pin a connection * child in nanosleep for an absurd time. */ /* Bounded well below the socket I/O timeout so a failed-auth child cannot hold * a connection slot for long enough to amplify connection-cap exhaustion. */ #define DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS 5000 +/* Upper bound on the number of [module] sections, so the shared registry's + * per-module counter array stays fixed-size. The parser rejects the next + * section past this bound. */ +#define DAEMON_CONF_MAX_MODULES 256 /* Longest accepted config line (excluding the trailing newline). Longer lines * are rejected rather than buffered unboundedly. */ #define DAEMON_CONF_MAX_LINE 4096 @@ -129,8 +154,9 @@ bool daemon_module_name_valid(const char* name); /* Parse one --dparam=KEY=VALUE (or "--dparam KEY=VALUE") override string and * apply it to the global keys only. Keys are case-insensitive and limited to * the global keys defined by the grammar (port, motd file, address, - * max connections, auth failure delay, hosts allow, hosts deny). Returns 0 on - * success, -1 on error (err filled). */ + * max connections, max connections per host, auth failure delay, + * auth lockout threshold, auth lockout duration, hosts allow, hosts deny). + * Returns 0 on success, -1 on error (err filled). */ int daemon_conf_apply_dparam(DaemonConf* conf, const char* assignment, char* err, size_t err_size); /* Host access-control matching (pure; no I/O). `daemon_host_pattern_match` diff --git a/tests/test_daemon_conf.c b/tests/test_daemon_conf.c index e0020c9..a9113a5 100644 --- a/tests/test_daemon_conf.c +++ b/tests/test_daemon_conf.c @@ -33,6 +33,11 @@ static void test_daemon_conf_create_defaults() { EXPECT_NULL(conf->global.address); EXPECT_EQ_INT(conf->global.max_connections, DAEMON_CONF_DEFAULT_MAX_CONNECTIONS); EXPECT_EQ_INT(conf->global.auth_failure_delay_ms, DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS); + EXPECT_EQ_INT(conf->global.max_connections_per_host, + DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST); + EXPECT_EQ_INT(conf->global.auth_lockout_threshold, DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD); + EXPECT_EQ_INT(conf->global.auth_lockout_duration_sec, + DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC); EXPECT_EQ_INT(conf->global.hosts_allow_count, 0); EXPECT_EQ_INT(conf->global.hosts_deny_count, 0); EXPECT_EQ_INT(conf->module_count, 0); @@ -316,6 +321,12 @@ static void test_daemon_conf_dparam_override() { EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "max connections=7", err, sizeof(err)), 0); EXPECT_EQ_INT(conf->global.max_connections, 7); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "max connections per host=3", err, sizeof(err)), 0); + EXPECT_EQ_INT(conf->global.max_connections_per_host, 3); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "auth lockout threshold=5", err, sizeof(err)), 0); + EXPECT_EQ_INT(conf->global.auth_lockout_threshold, 5); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "auth lockout duration=120", err, sizeof(err)), 0); + EXPECT_EQ_INT(conf->global.auth_lockout_duration_sec, 120); EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "AUTH FAILURE DELAY=1500", err, sizeof(err)), 0); EXPECT_EQ_INT(conf->global.auth_failure_delay_ms, 1500); EXPECT_EQ_INT( @@ -390,6 +401,9 @@ static void test_daemon_conf_limits_and_hosts_parse() { char err[256]; EXPECT_EQ_INT(write_conf("max connections = 25\n" "auth failure delay = 0\n" + "max connections per host = 4\n" + "auth lockout threshold = 3\n" + "auth lockout duration = 60\n" "hosts allow = 10.0.0.0/8, 192.168.1.0/24\n" "hosts deny = 192.168.0.1 2001:db8::/32\n" "\n" @@ -405,6 +419,9 @@ static void test_daemon_conf_limits_and_hosts_parse() { EXPECT_NOT_NULL(conf); EXPECT_EQ_INT(conf->global.max_connections, 25); EXPECT_EQ_INT(conf->global.auth_failure_delay_ms, 0); + EXPECT_EQ_INT(conf->global.max_connections_per_host, 4); + EXPECT_EQ_INT(conf->global.auth_lockout_threshold, 3); + EXPECT_EQ_INT(conf->global.auth_lockout_duration_sec, 60); EXPECT_EQ_INT(conf->global.hosts_allow_count, 2); EXPECT_EQ_STR(conf->global.hosts_allow[0], "10.0.0.0/8"); EXPECT_EQ_STR(conf->global.hosts_allow[1], "192.168.1.0/24"); @@ -419,11 +436,14 @@ static void test_daemon_conf_limits_and_hosts_parse() { daemon_conf_free(conf); const char* bad_values[] = { - "max connections = 0\n", "max connections = -1\n", - "max connections = abc\n", "auth failure delay = -1\n", - "auth failure delay = 70000\n", "auth failure delay = soon\n", - "hosts allow = 10.0.0.0/99\n", "hosts deny = 2001:db8::/129\n", - "hosts allow = *.example.com\n", "hosts deny = not-an-ip\n", + "max connections = 0\n", "max connections = -1\n", + "max connections = abc\n", "auth failure delay = -1\n", + "auth failure delay = 70000\n", "auth failure delay = soon\n", + "max connections per host = -1\n", "max connections per host = lots\n", + "auth lockout threshold = -2\n", "auth lockout threshold = many\n", + "auth lockout duration = -1\n", "auth lockout duration = forever\n", + "hosts allow = 10.0.0.0/99\n", "hosts deny = 2001:db8::/129\n", + "hosts allow = *.example.com\n", "hosts deny = not-an-ip\n", }; for (size_t i = 0; i < sizeof(bad_values) / sizeof(bad_values[0]); i++) { EXPECT_EQ_INT(write_conf(bad_values[i], &path), 0); @@ -434,7 +454,8 @@ static void test_daemon_conf_limits_and_hosts_parse() { /* The same strictness applies inside a module section. */ const char* bad_module[] = { - "[m]\npath = /x\nmax connections = 0\n", + "[m]\npath = /x\nmax connections = -1\n", + "[m]\npath = /x\nmax connections = abc\n", "[m]\npath = /x\nhosts allow = 10.0.0.0/40\n", "[m]\npath = /x\nhosts deny = 999.1.1.1/8\n", }; @@ -446,6 +467,14 @@ static void test_daemon_conf_limits_and_hosts_parse() { EXPECT_TRUE(strstr(err, "invalid") != NULL); } + /* Module `max connections = 0` is now valid and means unlimited. */ + EXPECT_EQ_INT(write_conf("[m]\npath = /x\nmax connections = 0\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_EQ_INT(conf->modules[0].max_connections, 0); + daemon_conf_free(conf); + /* An empty hosts list is not an error (no patterns are added). */ EXPECT_EQ_INT(write_conf("hosts allow = \n[m]\npath = /x\n", &path), 0); conf = daemon_conf_load(path, err, sizeof(err)); @@ -509,6 +538,27 @@ static void test_daemon_module_name_valid() { } } +static void test_daemon_conf_module_count_capped() { + size_t cap = DAEMON_CONF_MAX_MODULES; + size_t len = (cap + 8) * 32; + char* body = malloc(len); + EXPECT_NOT_NULL(body); + body[0] = '\0'; + for (size_t i = 0; i < cap + 1; i++) { + char line[48]; + snprintf(line, sizeof(line), "[m%zu]\npath = /x\n", i); + strcat(body, line); + } + char* path; + EXPECT_EQ_INT(write_conf(body, &path), 0); + free(body); + char err[256]; + const DaemonConf* conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "too many modules") != NULL); +} + void test_daemon_conf() { test_daemon_conf_create_defaults(); test_daemon_conf_full_parse(); @@ -525,6 +575,7 @@ void test_daemon_conf() { test_daemon_conf_dparam_override(); test_daemon_conf_auth_users_validated(); test_daemon_conf_limits_and_hosts_parse(); + test_daemon_conf_module_count_capped(); test_daemon_hosts_allowed(); test_daemon_module_name_valid(); } \ No newline at end of file From 4c17122b008aa7dae550c4235d6787cf12427820 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:24:05 +0200 Subject: [PATCH 3/8] feat(daemon): enforce per-module/per-host caps and shared auth lockout Wire the shared registry into the accept loop (parent claims a slot before fork, blocks SIGCHLD across fork+pid publication, and reclaims the dead child's slot from the SIGCHLD handler so per-module/per-source counts are released even on SIGKILL). The connection child records the selected module and normalized peer IP once the config frame names them: an over-cap module or source is refused at the config gate with an audit log, and a source that exceeded the auth-failure threshold is refused before a SCRAM challenge (the counter is shared across children and cleared on success). The existing global cap and host ACLs are untouched. --- src/server/server.c | 114 +++++++++++++++++++++++++++++-- src/shared/transport_tcp.c | 52 +++++++++++++- src/shared/transport_tcp.h | 13 ++++ tests/integration/test_daemon.py | 79 ++++++++++++++++++++- 4 files changed, 248 insertions(+), 10 deletions(-) diff --git a/src/server/server.c b/src/server/server.c index 494a840..6f3862c 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -2,6 +2,7 @@ #include "charset.h" #include "credentials.h" #include "daemon_conf.h" +#include "daemon_limits.h" #include "delay_updates.h" #include "file.h" #include "identity.h" @@ -56,6 +57,12 @@ static DaemonConf* g_daemon_conf = NULL; * such a module exists. */ static CredentialStore* g_credentials = NULL; +/* Cross-process connection registry (per-module and per-source caps plus the + * shared auth lockout), created once in main BEFORE the accept loop forks and + * shared read-only-by-pointer with every connection child. NULL outside daemon + * mode or when the mapping could not be allocated (global cap + ACLs remain). */ +static DaemonLimitRegistry* g_daemon_limits = NULL; + /* Opaque context threaded through to the config-frame gate: the connection's * SSL object (NULL over plaintext) so the gate can warn when a credential * exchange is not encrypted, plus the super-mode override the gate decides on. @@ -292,6 +299,56 @@ static const DaemonModule* module_gate_lookup_module(const Config* config, const return module; } +/* Index of `module` within the loaded config's module array (the registry's + * per-module counter key). Returns -1 when it cannot be resolved. */ +static int daemon_module_index(const DaemonModule* module) { + if (!g_daemon_conf || !module || module < g_daemon_conf->modules || + module >= g_daemon_conf->modules + g_daemon_conf->module_count) + return -1; + return (int)(module - g_daemon_conf->modules); +} + +/* Shared-registry admission: reserve this connection's slot for the selected + * module and the peer source IP. Enforces the per-module `max connections` and + * the global `max connections per host` across every forked child. Runs before + * auth/ownership so a client that is over a cap is refused before any work. + * The per-source cap is skipped when the peer cannot be classified (host ACLs + * fail closed separately); the module cap still applies. A missing registry + * (allocation failure / non-fork path) fails open -- the global cap and ACLs + * still bound the listener. */ +static const char* module_gate_check_limits(const Config* config, const DaemonModule* module, + ModuleGateContext* gate_ctx) { + if (!g_daemon_limits) + return NULL; + int slot = transport_tcp_current_slot(); + if (slot < 0) + return NULL; /* not on the forked accept-loop path (e.g. --stdio) */ + int module_index = daemon_module_index(module); + if (module_index < 0) + return NULL; + const char* peer = (gate_ctx && gate_ctx->has_peer_ip) ? gate_ctx->peer_ip : ""; + DaemonLimitResult result = + daemon_limits_register(g_daemon_limits, slot, module_index, peer, module->max_connections); + switch (result) { + case DAEMON_LIMIT_OK: + return NULL; + case DAEMON_LIMIT_MODULE_FULL: + log_message(LOG_LEVEL_ERROR, + "daemon module '%s': 'max connections' cap (%d) reached; refusing %s", + config->module, module->max_connections, peer[0] ? peer : "peer"); + return "requested daemon module is at its connection limit"; + case DAEMON_LIMIT_HOST_FULL: + log_message(LOG_LEVEL_ERROR, + "daemon: 'max connections per host' cap (%d) reached for %s; refusing module '%s'", + g_daemon_conf->global.max_connections_per_host, peer[0] ? peer : "peer", + config->module); + return "too many concurrent connections from this host"; + case DAEMON_LIMIT_UNAVAILABLE: + default: + return NULL; + } +} + /* Per-module client-chosen ownership / super-user policy (P7 Wave E hardening): * a daemon module refuses EVERY ownership-affecting request (--numeric-ids, * --chown, --usermap/--groupmap, --fake-super, --copy-as, explicit --super) @@ -396,6 +453,20 @@ static ModuleAuthResult module_gate_authenticate(const Config* config, const Dae ModuleGateContext* gate_ctx, const char** error) { if (module->auth_user_count == 0) return MODULE_AUTH_ACCEPTED; + /* Cross-process lockout: a source that failed too many authentications is + * refused before the challenge is sent (the counter lives in the shared + * registry, so it spans every forked child and survives a child exit). */ + if (g_daemon_limits && gate_ctx && gate_ctx->has_peer_ip) { + int remaining = 0; + if (daemon_limits_auth_locked(g_daemon_limits, gate_ctx->peer_ip, &remaining)) { + log_message(LOG_LEVEL_ERROR, + "daemon module '%s': source %s is locked out after repeated authentication " + "failures (%d s remaining); refusing", + config->module, gate_ctx->peer_ip, remaining); + *error = "too many failed authentication attempts from this host; try again later"; + return MODULE_AUTH_REFUSED; + } + } /* Fail closed: no store -> refuse (server misconfiguration, STATUS_ERROR). */ if (g_credentials == NULL) { log_message(LOG_LEVEL_ERROR, @@ -450,10 +521,16 @@ static ModuleAuthResult module_gate_authenticate(const Config* config, const Dae "daemon module '%s': authentication failed for user '%s' from %s; refusing", config->module, escaped_user ? escaped_user : "(none)", peer); free(escaped_user); - /* Rate-limit online guessing per connection (no delay on success). */ + /* Count the failure in the shared registry (locks the source out once the + * configured threshold is reached) and rate-limit online guessing per + * connection (no delay on success). */ + if (g_daemon_limits && gate_ctx->has_peer_ip) + daemon_limits_auth_record_failure(g_daemon_limits, gate_ctx->peer_ip); daemon_auth_failure_delay(); return MODULE_AUTH_TERMINATED; } + if (g_daemon_limits && gate_ctx->has_peer_ip) + daemon_limits_auth_record_success(g_daemon_limits, gate_ctx->peer_ip); char* escaped_user = output_escape(config->auth_user, config->eight_bit_output); log_message(LOG_LEVEL_INFO, "daemon module '%s': user '%s' from %s authenticated", config->module, escaped_user ? escaped_user : "", @@ -561,6 +638,9 @@ static const char* server_module_gate(const Config* config, void* context) { log_message(LOG_LEVEL_DEBUG, "daemon module '%s': peer address unavailable", config->module); } error = module_gate_check_hosts(config, module, gate_ctx); + if (error) + return error; + error = module_gate_check_limits(config, module, gate_ctx); if (error) return error; error = module_gate_check_ownership(config, module, gate_ctx); @@ -880,7 +960,9 @@ static void print_server_usage(void) { printf(" fastsyncd.conf, else /etc/fastsyncd.conf)\n"); printf(" --dparam=KEY=VALUE Override one global config key on the command line\n"); printf(" (port, motd file, address, max connections,\n"); - printf(" auth failure delay, hosts allow, hosts deny)\n"); + printf(" max connections per host, auth failure delay,\n"); + printf(" auth lockout threshold, auth lockout duration,\n"); + printf(" hosts allow, hosts deny)\n"); printf(" --no-detach Stay in the foreground (default detaches to\n"); printf(" background when running --daemon)\n"); printf(" --password-file=FILE Credential store for modules that declare\n"); @@ -1096,11 +1178,10 @@ int main(int argc, char* argv[]) { "unless the module is intentionally open to the network", g_daemon_conf->modules[i].name); if (g_daemon_conf->modules[i].max_connections > 0) - log_message(LOG_LEVEL_WARNING, - "daemon module '%s': per-module 'max connections' is stored but not enforced " - "per module; the global 'max connections' cap (%d) applies to the whole " - "listener", - g_daemon_conf->modules[i].name, g_daemon_conf->global.max_connections); + log_message(LOG_LEVEL_INFO, + "daemon module '%s': per-module 'max connections' cap = %d (enforced " + "across all connection children)", + g_daemon_conf->modules[i].name, g_daemon_conf->modules[i].max_connections); } /* Daemon credential store (Wave B). --password-file and --early-input * feed the same store, loaded BEFORE the listener forks so every @@ -1144,6 +1225,21 @@ int main(int argc, char* argv[]) { module->name, module->auth_users[j]); } } + /* Shared cross-process registry for the per-module / per-source caps and + * the auth lockout. Created HERE in the parent before any accept-loop + * fork; every connection child inherits the mapping. A failure degrades to + * "registry disabled" (the global cap and host ACLs still apply) rather + * than refusing to start. */ + g_daemon_limits = daemon_limits_create((int)g_daemon_conf->global.max_connections, + g_daemon_conf->module_count, + g_daemon_conf->global.max_connections_per_host, + g_daemon_conf->global.auth_lockout_threshold, + g_daemon_conf->global.auth_lockout_duration_sec); + if (!g_daemon_limits) + log_message(LOG_LEVEL_WARNING, + "daemon: could not allocate the shared connection registry; per-module / " + "per-host caps and the cross-process auth lockout are disabled (the global " + "'max connections' cap and host ACLs still apply)"); } else { if (!configure_authorization(opts.destination_root)) { char* escaped = output_escape(opts.destination_root, false); @@ -1167,6 +1263,8 @@ int main(int argc, char* argv[]) { } if (g_daemon_conf) server_set_max_connections(g_server, (unsigned int)g_daemon_conf->global.max_connections); + if (g_daemon_limits) + server_set_limit_registry(g_server, g_daemon_limits); if (opts.use_tls) { if (!opts.tls_cert || !opts.tls_key || !opts.tls_ca || !opts.client_cn) { fprintf(stderr, "Error: --tls requires --cert, --key, --ca, and --client-cn\n"); @@ -1206,6 +1304,8 @@ int main(int argc, char* argv[]) { release_authorization(); out: + daemon_limits_destroy(g_daemon_limits); + g_daemon_limits = NULL; daemon_conf_free(g_daemon_conf); g_daemon_conf = NULL; credentials_free(g_credentials); diff --git a/src/shared/transport_tcp.c b/src/shared/transport_tcp.c index 2758cbe..e73dbce 100644 --- a/src/shared/transport_tcp.c +++ b/src/shared/transport_tcp.c @@ -1,4 +1,5 @@ #include "transport_tcp.h" +#include "daemon_limits.h" #include "log.h" #include "protocol.h" #include "utils.h" @@ -18,15 +19,25 @@ static volatile sig_atomic_t g_active_connections = 0; +/* Shared registry installed on the active server; the SIGCHLD handler needs a + * file-scope pointer so it can reclaim the dead child's slot. Set once by + * accept_loop before the fork loop (single-threaded parent). */ +static DaemonLimitRegistry* g_limit_registry = NULL; +/* Slot reserved by the parent for the connection child currently being forked. + * Written before fork(), read by the child (which inherits the value). */ +static int g_current_slot = DAEMON_LIMITS_NO_SLOT; + static void tcp_apply_socket_timeout(int fd); static void tcp_enable_nodelay_default(int fd, int family); static void sigchld_handler(int sig) { (void)sig; int saved_errno = errno; - while (waitpid(-1, NULL, WNOHANG) > 0) { + pid_t pid; + while ((pid = waitpid(-1, NULL, WNOHANG)) > 0) { if (g_active_connections > 0) g_active_connections--; + daemon_limits_reclaim_pid(g_limit_registry, (long)pid); } errno = saved_errno; } @@ -108,6 +119,7 @@ Server* server_create_ex(int port, const ServerBindOptions* bind_opts) { server->ssl_ctx = NULL; server->max_connections = 100; server->active_connections = 0; + server->limit_registry = NULL; return server; } @@ -121,6 +133,15 @@ void server_set_max_connections(Server* server, unsigned int max_connections) { server->max_connections = max_connections; } +void server_set_limit_registry(Server* server, struct DaemonLimitRegistry* registry) { + if (server) + server->limit_registry = registry; +} + +int transport_tcp_current_slot(void) { + return g_current_slot; +} + void server_delete(Server** server) { if (server == NULL || *server == NULL) return; @@ -140,6 +161,7 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil return; } signal(SIGCHLD, sigchld_handler); + g_limit_registry = server->limit_registry; while (1) { struct sockaddr_storage client_addr; socklen_t client_len = sizeof(client_addr); @@ -159,9 +181,31 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil close(fd); continue; } + int slot = DAEMON_LIMITS_NO_SLOT; + if (server->limit_registry) { + slot = daemon_limits_claim_slot(server->limit_registry); + if (slot == DAEMON_LIMITS_NO_SLOT) { + /* The global cap bounds live children, so this only happens when the + * fixed registry is smaller than the configured cap; fail closed. */ + log_message(LOG_LEVEL_WARNING, "Connection registry slots exhausted (max %u), rejecting %s", + server->max_connections, peer); + close(fd); + continue; + } + } log_message(LOG_LEVEL_INFO, "%s from %s", log_fmt, peer); + g_current_slot = slot; + /* Block SIGCHLD across fork() and the parent's pid publication: a child + * that exits immediately must not be reaped before its slot records its + * pid, which would leak the slot and its module/source counts. */ + sigset_t blocked; + sigset_t previous; + sigemptyset(&blocked); + sigaddset(&blocked, SIGCHLD); + sigprocmask(SIG_BLOCK, &blocked, &previous); pid_t pid = fork(); if (pid == 0) { + sigprocmask(SIG_SETMASK, &previous, NULL); /* Connection children must not run the parent's global cleanup(): it * frees state (credentials / daemon conf) that the child's worker * threads may still be reading and closes fd numbers the child could @@ -177,7 +221,13 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil _exit(0); } else if (pid > 0) { g_active_connections++; + if (server->limit_registry) + daemon_limits_set_slot_pid(server->limit_registry, slot, (long)pid); + } else if (server->limit_registry) { + /* fork() failed: release the reservation so the slot is not leaked. */ + daemon_limits_reclaim_slot(server->limit_registry, slot); } + sigprocmask(SIG_SETMASK, &previous, NULL); close(fd); } } diff --git a/src/shared/transport_tcp.h b/src/shared/transport_tcp.h index e37b879..c3862a6 100644 --- a/src/shared/transport_tcp.h +++ b/src/shared/transport_tcp.h @@ -7,6 +7,10 @@ #include #include +/* Cross-process daemon registry (daemon_limits.c). Only an opaque pointer is + * stored here so the transport layer does not depend on daemon config. */ +struct DaemonLimitRegistry; + typedef struct Server { struct sockaddr_storage address; unsigned int address_length; @@ -14,6 +18,7 @@ typedef struct Server { void* ssl_ctx; unsigned int max_connections; volatile unsigned int active_connections; + struct DaemonLimitRegistry* limit_registry; } Server; typedef struct Client { @@ -48,6 +53,14 @@ Server* server_create(int port); /* Override the listener's connection cap (the global daemon `max connections` * value). A non-positive value is ignored so the default cap stands. */ void server_set_max_connections(Server* server, unsigned int max_connections); +/* Install the shared per-module / per-source registry used by the accept loop + * to reserve a slot for each forked child. NULL disables the accounting (the + * global cap and ACLs still apply). */ +void server_set_limit_registry(Server* server, struct DaemonLimitRegistry* registry); +/* Slot reserved for the connection child currently running (set by the parent + * before fork, inherited by the child). Returns DAEMON_LIMITS_NO_SLOT (-1) + * outside the accept-loop child path. */ +int transport_tcp_current_slot(void); bool server_listen(Server* server, void (*handler)(int file_descriptor)); void server_accept_loop(Server* server, void (*child_fn)(int, void*), void* child_ctx, const char* log_fmt); diff --git a/tests/integration/test_daemon.py b/tests/integration/test_daemon.py index 98c963d..1ff1d9d 100644 --- a/tests/integration/test_daemon.py +++ b/tests/integration/test_daemon.py @@ -135,7 +135,7 @@ class DaemonManager: self._proc = None self._port = None - def start(self, config_path, port_override=None, extra_args=None): + def start(self, config_path, port_override=None, extra_args=None, log_path=None): self.stop() # When no override is given the daemon binds the config file's `port` # (the plain config-port path); with an override the --dparam path. @@ -146,7 +146,8 @@ class DaemonManager: cmd += ["--dparam", f"port={port_override}"] if extra_args: cmd += extra_args - log_path = os.path.join(TEST_DATA_DIR, "fastsyncd.log") + if log_path is None: + log_path = os.path.join(TEST_DATA_DIR, "fastsyncd.log") log = open(log_path, "w") self._proc = subprocess.Popen( cmd, stdout=log, stderr=log, stdin=subprocess.DEVNULL, start_new_session=True) @@ -1208,3 +1209,77 @@ class TestDaemonTLSAuth: d.stop() os.unlink(client_creds) shutil.rmtree(cert_dir, ignore_errors=True) + + +class TestDaemonConnectionLimits: + """Wave 8: cross-process per-module / per-source connection caps and the + shared auth lockout. Each test boots its own daemon with a unique port so + the shared (per-daemon) registry state is isolated from the module-scoped + `daemon` fixture.""" + + LOCKOUT_CONF = os.path.join(TEST_DATA_DIR, "fastsyncd_lockout.conf") + CAPS_CONF = os.path.join(TEST_DATA_DIR, "fastsyncd_caps.conf") + + @pytest.mark.ci + def test_auth_lockout_is_shared_across_children(self): + """`auth lockout threshold = 1`: the first failed authentication locks the + source out for the cooldown in the SHARED registry, so a subsequent + correct-password attempt (a different forked child) is refused before a + SCRAM challenge is even sent.""" + port = _find_free_port() + with open(self.LOCKOUT_CONF, "w") as f: + f.write("port = %d\n" + "auth lockout threshold = 1\n" + "auth lockout duration = 300\n" + "\n" + "[locked]\n" + "path = %s\n" + "auth users = alice\n" + % (port, AUTH_MODULE)) + d = DaemonManager() + log_path = os.path.join(TEST_DATA_DIR, f"fastsyncd_lockout_{os.getpid()}.log") + try: + d.start(self.LOCKOUT_CONF, port_override=port, extra_args=["--password-file", CRED_FILE], + log_path=log_path) + before = _tree_file_count(AUTH_MODULE) + log_before = os.path.getsize(log_path) if os.path.exists(log_path) else 0 + # First attempt: wrong password -> records failure #1 -> locks. + wrong = _push_with_creds("127.0.0.1::locked", port, "alice", WRONG_PASS) + assert wrong.returncode != 0 + # Second attempt: CORRECT password from the same source must still be + # refused by the shared lockout. + right = _push_with_creds("127.0.0.1::locked", port, "alice", ALICE_PASS) + assert right.returncode != 0, "the shared auth lockout must refuse after threshold" + assert _tree_file_count(AUTH_MODULE) == before, "a locked-out source wrote data" + time.sleep(0.3) + with open(log_path, "rb") as f: + f.seek(log_before) + tail = f.read().decode("utf-8", "replace") + assert "locked out" in tail, tail[-400:] + finally: + d.stop() + + def test_caps_keys_accepted_and_transfer_still_works(self): + """A daemon configured with the new keys (per-host cap, lockout threshold + and duration, per-module cap) starts and serves a normal transfer.""" + port = _find_free_port() + with open(self.CAPS_CONF, "w") as f: + f.write("port = %d\n" + "max connections per host = 5\n" + "auth lockout threshold = 3\n" + "auth lockout duration = 60\n" + "\n" + "[files]\n" + "path = %s\n" + "max connections = 2\n" + % (port, FILES_MODULE)) + d = DaemonManager() + try: + d.start(self.CAPS_CONF, port_override=port) + result = _push("127.0.0.1::files", port) + assert result.returncode == 0, result.stderr or result.stdout + received = get_dest_received_dir(FILES_MODULE, SOURCE_DIR) + _, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"missing: {missing[:5]}" + finally: + d.stop() From e1f8f75e7c794dc5ea5249755fcc519f4b7993e5 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:24:09 +0200 Subject: [PATCH 4/8] docs: document daemon per-module/per-host caps and shared auth lockout --- CHANGELOG.md | 12 ++++++++++++ README.md | 17 +++++++++++++---- RSYNC_COMPAT.md | 6 +++--- 3 files changed, 28 insertions(+), 7 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 66132a2..a641f4b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,18 @@ All notable changes to FastSync are documented here. Versions match `PROTOCOL_VERSION` (printed by `fastsync --version`); the client and server must run the same version because the handshake is strict. +## [Unreleased] + +### Security + +- Enforce the daemon's per-module `max connections` cap and add a global + `max connections per host` cap plus a cross-process `auth lockout` + (`auth lockout threshold` / `auth lockout duration`). Because the listener + forks one child per connection, the counters live in an anonymous shared + mapping created before the accept loop and reclaimed by the parent's + `SIGCHLD` handler, so the per-module, per-source and auth-failure state is + shared across every child (including after `SIGKILL`). + ## [2.20.0] - 2026-09-13 ### Security diff --git a/README.md b/README.md index 9820885..58d0d10 100644 --- a/README.md +++ b/README.md @@ -506,18 +506,27 @@ defaults to the current directory. | implicit global section, then `[module]` sections). Besides `port`, `motd file`, and `address`, the global section accepts: -- `max connections = N` — cap on concurrent connections, default 100. The +- `max connections = N` — global cap on concurrent connections, default 100. The listener enforces it; `0`, negative, and non-numeric values are parse errors. +- `max connections per host = N` — cap on concurrent connections from a single + source IP, default 0 (unlimited). Enforced across all forked connection + children through a shared registry. - `auth failure delay = MS` — milliseconds to sleep after a failed authentication, default 500. `0` disables it and the value is capped at 60000, so online password guessing is rate-limited per connection. Successful auths are never delayed. +- `auth lockout threshold = N` — number of failed authentications from one source + IP before that source is locked out, default 10; `0` disables the lockout. The + failure counter is shared across every connection child, so the lockout holds + even when the next attempt is handled by a different forked child. +- `auth lockout duration = SECONDS` — how long a locked-out source is refused + (default 300). A locked-out client is refused before any SCRAM challenge is + sent; a successful authentication clears the counter. - `hosts allow` / `hosts deny` — comma- and/or whitespace-separated host access patterns. -A `[module]` may also set `max connections` (parsed and validated but not -enforced per module — the global cap applies to the whole listener) and its own -`hosts allow`/`hosts deny`. +A `[module]` may also set `max connections` (0 = unlimited; enforced per module +across all connection children) and its own `hosts allow`/`hosts deny`. Host patterns are `*` (match all), IPv4/IPv6 literals, or IPv4/IPv6 CIDR (`10.0.0.0/8`, `2001:db8::/32`). Hostnames are not resolved, so hostname globs diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index ea171bc..f0d4ba2 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -627,7 +627,7 @@ now transmits targets (the prior behavior was broken/partial); its status moved |------|-------------------|-----------------|-------| | `--daemon` | Run as rsync daemon | ✅ Implemented | Wave A: a real persistent listener. `fastsync-server --daemon --config FILE` (plus `--no-detach` to stay foreground; without it the listener detaches to the background after binding) reads a FastSync-native module config file and serves each connection confined to the requested module's `path` root (never a client-chosen root; every client-chosen-ownership/super-user request (`--numeric-ids`/`--chown`/`--usermap`/`--groupmap`/`--fake-super`/`--copy-as`/explicit `--super`) is refused unless the module opts in with `client owner = yes`, and the operator `--no-super` veto is honored). TCP/TLS via the existing `--tls` stack; plaintext still requires `--allow-unauthenticated` (same secure default as the standalone server). Client destinations use rsync's `host::module/path` form. Wire/protocol: the config frame gained a trailing daemon-module string and `PROTOCOL_VERSION` was bumped **2.14.0 → 2.15.0** (see the Daemon Mode notes below). Daemon mode is built in FastSync's own protocol/config grammar, not rsync's SMB/daemon option encoding | | `--config=FILE` | Alternate rsyncd.conf file | ✅ Implemented | Wave A: selects the daemon config file. Default when omitted (in `--daemon` mode): `~/.config/fastsync/fastsyncd.conf` if it exists, else `/etc/fastsyncd.conf`. The grammar is FastSync-native (documented in the Daemon Mode notes below) and strictly rejects unknown keys so a typo can never silently change what a module serves; requires `--daemon` | -| `--dparam=OVERRIDE` | Override global daemon config | ✅ Implemented | Wave A: overrides one global scalar from the command line (`--dparam port=8734` and `--dparam=KEY=VALUE` both work). Limited to the global keys the grammar defines (`port`, `motd file`, `address`, `max connections`, `auth failure delay`, `hosts allow`, `hosts deny`); keys are case-insensitive and unknown keys/invalid values are rejected. Requires `--daemon` | +| `--dparam=OVERRIDE` | Override global daemon config | ✅ Implemented | Wave A: overrides one global scalar from the command line (`--dparam port=8734` and `--dparam=KEY=VALUE` both work). Limited to the global keys the grammar defines (`port`, `motd file`, `address`, `max connections`, `max connections per host`, `auth failure delay`, `auth lockout threshold`, `auth lockout duration`, `hosts allow`, `hosts deny`); keys are case-insensitive and unknown keys/invalid values are rejected. Requires `--daemon` | | `--no-detach` | Don't detach from parent | ✅ Implemented | Wave A: with `--daemon`, keeps the listener in the foreground (what integration tests use). Without it the daemonizes (fork/setsid, stdio redirected to /dev/null) after the listening socket is bound. Requires `--daemon` | | `--password-file=FILE` | Read daemon password from file | ✅ Implemented | A7 daemon auth. Client: `--password-file` supplies `user:password` for a `host::module/path` destination (the username is taken from this file, so `user@host::module` stays rejected); the literal password is held client-side only for the SCRAM handshake and wiped at teardown. Server (`fastsync-server --daemon --password-file FILE`): the salted-PBKDF2 verifier store that modules with `auth users` are verified against. **Neither the password nor any replayable bearer value crosses the wire or is stored server-side** — the store holds a per-user salt plus derived keys, and the daemon proves the secret with a per-connection nonce challenge. The file must be private to its owner: both the client and server verify the exact inode they read (open-then-`fstat`, so the check cannot be raced) and refuse a `--password-file`/`--early-input` that is not owned by the current user or grants any group/other permission bit (mode 0600), mirroring the TLS private-key check. A process-substitution pipe (`--early-input <(vault ...)`) is still accepted when it satisfies those checks. See the Daemon Mode notes below for the file formats and the plaintext/TLS caveat | | `--early-input=FILE` | Use FILE for daemon early exec | ✅ Implemented | Server-only (requires `--daemon`): a second credential-store file, same new-format grammar as `--password-file`, read before the listener accepts connections (a secrets-manager / process-substitution source). Its entries layer over `--password-file`: byte-identical verifiers dedupe, a conflicting verifier for the same user is a startup error. A daemon whose modules declare `auth users` must be given at least one of the two, or it refuses to start (fail closed) | @@ -635,9 +635,9 @@ now transmits targets (the prior behavior was broken/partial); its status moved **Daemon Mode notes (Wave A protocol 2.15.0; A7 auth protocol 2.19.0; MOTD no bump):** FastSync daemon mode is supported in FastSync's own protocol/config grammar, not rsync's SMB/daemon option encoding. -- **Config grammar** (`fastsyncd.conf`): line-based; an implicit global section first, then `[module]` sections. Keys are case-insensitive, values are trimmed and may be wrapped in one layer of double quotes (`path = "/srv/my dir"`). `#` and `;` at the start of a line (after leading whitespace) are full-line comments; inline comments and `\` continuations are not supported. Lines are bounded (4096 chars). Global keys: `port` (default 873), `motd file` (the daemon sends its bounded, escaped content to a client after the module gate/auth accepts, unless the client passes `--no-motd`), `address` (optional bind address), `max connections` (positive integer cap on concurrent connections, default 100; 0/negative/garbage is a parse error), `auth failure delay` (milliseconds to sleep after a failed authentication, default 500; 0 disables, capped at 5000), `hosts allow` and `hosts deny` (comma- and/or whitespace-separated host access patterns — see the host access control note below). Module keys: `path` (required; the daemon-side authorized root for that module), `read only` (yes/no/true/false/1/0, default no), `client owner` (yes/no/true/false/1/0, default no; opts the module into client-chosen ownership — see below), `auth users` (comma list), `max connections` (optional per-module cap; parsed and stored but **not enforced** — the global cap applies to the whole listener), `hosts allow`/`hosts deny` (per-module host access lists). **Unknown keys and malformed lines are parse-and-reject errors** (never silently ignored), so a typo cannot change what a module serves. +- **Config grammar** (`fastsyncd.conf`): line-based; an implicit global section first, then `[module]` sections. Keys are case-insensitive, values are trimmed and may be wrapped in one layer of double quotes (`path = "/srv/my dir"`). `#` and `;` at the start of a line (after leading whitespace) are full-line comments; inline comments and `\` continuations are not supported. Lines are bounded (4096 chars), and at most 256 `[module]` sections are accepted. Global keys: `port` (default 873), `motd file` (the daemon sends its bounded, escaped content to a client after the module gate/auth accepts, unless the client passes `--no-motd`), `address` (optional bind address), `max connections` (positive integer cap on concurrent connections, default 100; 0/negative/garbage is a parse error), `max connections per host` (concurrent-connection cap per source IP, default 0 = unlimited), `auth failure delay` (milliseconds to sleep after a failed authentication, default 500; 0 disables, capped at 5000), `auth lockout threshold` (failed authentications from one source before lockout, default 10; 0 disables), `auth lockout duration` (seconds a locked-out source is refused, default 300), `hosts allow` and `hosts deny` (comma- and/or whitespace-separated host access patterns — see the host access control note below). Module keys: `path` (required; the daemon-side authorized root for that module), `read only` (yes/no/true/false/1/0, default no), `client owner` (yes/no/true/false/1/0, default no; opts the module into client-chosen ownership — see below), `auth users` (comma list), `max connections` (optional per-module cap, 0 = unlimited; enforced across all connection children), `hosts allow`/`hosts deny` (per-module host access lists). **Unknown keys and malformed lines are parse-and-reject errors** (never silently ignored), so a typo cannot change what a module serves. - **Host access control (`hosts allow`/`hosts deny`):** both keys accept a comma- and/or whitespace-separated list of patterns and may appear globally and/or per module (multiple config-file lines append; a `--dparam` override replaces). Supported patterns are `*` (match all), an IPv4 or IPv6 literal (`10.0.0.1`, `2001:db8::1`), and an IPv4/IPv6 CIDR (`10.0.0.0/8`, `2001:db8::/32`). Hostname patterns are **not** supported: because the peer is always a numeric address and no reverse DNS is performed, a hostname/glob pattern would silently never match, so it is rejected at load time (fail-closed) instead of being accepted as a dead rule. An IPv4 peer on a dual-stack IPv6 listener is normalized from its `::ffff:a.b.c.d` form so IPv4 patterns match it. rsync-like semantics: a matching `hosts deny` rejects; if any `hosts allow` entries exist, a peer matching none of them is rejected; deny takes precedence over allow. The daemon enforces the global list first, then the selected module's list, **before authentication** in `server_module_gate`, with an audit log line naming the peer, the module and the outcome. The numeric peer address is obtained with `getpeername`+`inet_ntop` (`utils_fd_peer_ip`, handling both address families); when it cannot be obtained a module with any ACL fails closed (refused), while an ACL-free module continues and logs at debug. A malformed pattern (e.g. an out-of-range CIDR prefix) is a parse error at load time. -- **Connection cap and auth throttle:** the global `max connections` key (default 100) is plumbed into the listener (`transport_tcp.c`), which rejects a connection once the accept-loop parent's active-child count reaches it; the IPv4/IPv6 peer is logged for every accepted connection. The optional per-module `max connections` key is parsed and validated but **not enforced** (connections are counted in the parent before the client's module is known); the daemon logs a startup warning for any module that sets it. On a failed authentication the per-connection child sleeps the global `auth failure delay` (default 500 ms, 0 disables, capped at 5000) via `nanosleep` before the connection closes, rate-limiting online guessing without delaying a success. +- **Connection caps, shared registry and auth lockout:** the global `max connections` key (default 100) is plumbed into the listener (`transport_tcp.c`), which rejects a connection once the accept-loop parent's active-child count reaches it; the IPv4/IPv6 peer is logged for every accepted connection. Because the listener forks one child per connection, the per-module `max connections` cap, the global `max connections per host` cap, and the auth-failure counter live in a fixed-size registry carved from an anonymous shared mapping (`daemon_limits.c`, `mmap(MAP_SHARED|MAP_ANONYMOUS)`) created by the parent before the accept loop, so every forked child shares the same counters (C11 atomics only — never a pthread lock, which can deadlock in a forked child). The parent reserves a registry slot per accepted connection and the child records the selected module and source IP once known; the parent's `SIGCHLD` handler reclaims the slot when the child dies (including `SIGKILL`), decrementing the per-module and per-source counts. The per-module cap (0 = unlimited) is enforced after the module lookup and before auth; per-source identity reuses the normalized numeric peer address (`utils_fd_peer_ip`, IPv4-mapped IPv6 collapsed to IPv4). A failed authentication increments the shared per-source failure count and, once `auth lockout threshold` (default 10; 0 disables) is reached, the source is refused for `auth lockout duration` seconds (default 300) before any challenge is sent, even when the next attempt is handled by a different forked child; a successful authentication clears the counter. On a failed authentication the per-connection child still sleeps the global `auth failure delay` (default 500 ms, 0 disables, capped at 5000) via `nanosleep`, rate-limiting online guessing without delaying a success. A missing registry (allocation failure) degrades to the global cap and host ACLs rather than refusing to start. - **Module selection & confinement:** the client requests a module with an rsync-style `host::module[/path]` destination. The module name crosses the wire as a trailing string on the config frame (bumping `PROTOCOL_VERSION` 2.14.0 → 2.15.0; the bump is required because the config-frame layout changed and the strict same-version handshake is what prevents a peer from desynchronizing on the new trailing field). The daemon looks the module up in ITS OWN config and uses the module's `path` as the authorized root through the exact same `configure_authorization` confinement the standalone server applies to `--destination-root` (`file_open_secure_parent`, `has_path_traversal`, `path_is_within`); the client never supplies the root, every client-chosen-ownership/super-user request is refused unless the module declares `client owner = yes` (the daemon's per-module opt-in, see below), and the operator `--no-super` veto forces super-user activities off for every daemon connection. The client's `/path` part is relative inside the module and is rejected if absolute or if it contains `..`. Unknown modules are refused before any data moves (the run fails cleanly at the config handshake). An absolute destination and a module request against a non-daemon server are also refused. - **`client owner` (client-chosen-ownership opt-in):** by default a daemon module refuses every request that would let the client pick an owner or ask for super-user activities — `--numeric-ids`, `--chown`, `--usermap`/`--groupmap`, `--fake-super`, `--copy-as`, and an explicit `--super` — at the config handshake (before `STATUS_OK`), because a daemon has no per-module opt-in for client-chosen ownership and any anonymous client could otherwise force arbitrary owner ids inside the module root. `client owner = yes` opts a single module in, allowing those requests within that module's root (the standalone listener and the SSH `--stdio` server always honor them for their single operator-authorized root). Without the opt-in the daemon also forces super-user **device** activity off for that connection — char/block device-node creation (`--devices`) and `--write-devices` — even under the default `AUTO` mode, so a non-opted module can never be made to `mknod` or write a raw device; those entries are skipped (not refused) so an ordinary `-a` push still succeeds without device nodes. The opt-in does **not** lift the privilege requirement: `--copy-as` still needs a root receiver, and the operator `--no-super` veto still forces super-user activities off for every connection. The daemon logs a prominent startup warning for each `client owner = yes` module so the operator's deliberate choice is visible. - **`read only` safe default:** every network transfer FastSync currently supports is a push that writes under the module root, so a `read only` module refuses the connection (clear server log "module is read only"; the client exits non-zero, nothing is transferred). A future pull/list operation can be opened up when it exists; the knob is already stored. From bd43448af2ef80afd38b2396351815b04ed6b727 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:50:53 +0200 Subject: [PATCH 5/8] fix(daemon): bound per-source table lifetime and recompute occupancy The per-source host table only grew: once its fixed open-addressed table filled, host_intern returned -1 and the per-host cap plus the shared auth lockout silently failed open forever. Add a bounded-lifetime eviction policy: track a per-bucket last-use time and, when no empty bucket exists, atomically repurpose the first bucket that has no active connection and either has an expired lockout or has been idle, resetting its counters. Warn (rate-limited) on the genuine fail-open path. A child SIGKILLed mid-registration could also leak a module/host count because the parent only decremented on a REGISTERED slot. Make the slot table the source of truth: after the SIGCHLD reap the parent recomputes module_active[]/host_active[] from the surviving REGISTERED slots (atomics only, async-signal-safe) so any leaked increment is erased. Also clamp module_count to DAEMON_LIMITS_MAX_MODULES and use one helper for the sizing/register host-tracking condition (a lockout threshold with duration 0 is a no-op and must not intern hosts). --- src/shared/daemon_limits.c | 183 +++++++++++++++++++++++++++++-------- src/shared/daemon_limits.h | 59 ++++++++++-- src/shared/transport_tcp.c | 48 ++++++++-- tests/test_daemon_limits.c | 78 ++++++++++++++++ 4 files changed, 315 insertions(+), 53 deletions(-) diff --git a/src/shared/daemon_limits.c b/src/shared/daemon_limits.c index 2ba7e4e..edb5819 100644 --- a/src/shared/daemon_limits.c +++ b/src/shared/daemon_limits.c @@ -1,4 +1,6 @@ #include "daemon_limits.h" +#include "daemon_conf.h" +#include "log.h" #include #include #include @@ -8,6 +10,12 @@ #include #include +/* The two module-count bounds must agree: the daemon config parser never + * produces more than DAEMON_CONF_MAX_MODULES modules, so the shared registry's + * per-module counter array is sized from the same bound. */ +_Static_assert(DAEMON_LIMITS_MAX_MODULES == DAEMON_CONF_MAX_MODULES, + "daemon_limits module bound must match daemon_conf"); + /* Slot lifecycle states (stored in slot_state). */ enum { SLOT_FREE = 0, @@ -27,6 +35,7 @@ struct DaemonLimitRegistry { int lockout_threshold; int lockout_duration_sec; size_t map_size; + _Atomic long long host_full_warn; /* last "table full" warning epoch */ _Atomic int* slot_state; _Atomic int* slot_pid; _Atomic int* slot_module; @@ -35,7 +44,8 @@ struct DaemonLimitRegistry { _Atomic uint64_t* host_key; /* 0 == empty bucket */ _Atomic int* host_active; _Atomic int* host_fail; - _Atomic long long* host_until; /* epoch seconds the lockout expires */ + _Atomic long long* host_until; /* epoch seconds the lockout expires */ + _Atomic long long* host_last_use; /* epoch seconds the bucket was last touched */ }; static size_t round_up(size_t n, size_t align) { @@ -88,7 +98,17 @@ uint64_t daemon_limits_host_hash(const char* peer_ip, bool* ok) { return hash; } -/* Find the bucket holding `peer_ip`, or -1 when it has no entry. */ +/* True when the registry must maintain per-source buckets: either the per-host + * cap is configured, or the auth lockout is (threshold AND duration > 0). A + * lockout threshold without a duration is a no-op, so it must not size or intern + * the table. create(), register() and the lockout paths all agree on this. */ +static bool registry_tracks_hosts(const DaemonLimitRegistry* registry) { + return registry->per_host_cap > 0 || + (registry->lockout_threshold > 0 && registry->lockout_duration_sec > 0); +} + +/* Find the bucket holding `peer_ip`, or -1 when it has no entry. Finding a + * bucket refreshes its last-use time so the eviction policy sees it as live. */ static int host_lookup(DaemonLimitRegistry* registry, const char* peer_ip) { bool ok = false; uint64_t key = daemon_limits_host_hash(peer_ip, &ok); @@ -99,39 +119,112 @@ static int host_lookup(DaemonLimitRegistry* registry, const char* peer_ip) { for (size_t i = 0; i < (size_t)registry->host_slots; i++) { size_t idx = (start + i) & mask; uint64_t current = atomic_load_explicit(®istry->host_key[idx], memory_order_acquire); - if (current == key) + if (current == key) { + atomic_store_explicit(®istry->host_last_use[idx], (long long)time(NULL), + memory_order_relaxed); return (int)idx; + } if (current == 0) return -1; /* no tombstones: an empty bucket ends the probe chain */ } return -1; } +/* A bucket with no live connection may be repurposed: immediately when its + * lockout deadline has already passed (the review's "expired" case), or after an + * idle window when it holds no pending lockout. A bucket with a future lockout + * deadline is retained so the lockout actually lasts its configured duration. */ +static bool host_bucket_reclaimable(DaemonLimitRegistry* registry, size_t idx, long long now) { + if (atomic_load_explicit(®istry->host_active[idx], memory_order_relaxed) != 0) + return false; + long long until = atomic_load_explicit(®istry->host_until[idx], memory_order_relaxed); + if (until != 0) + return until <= now; + long long last_use = atomic_load_explicit(®istry->host_last_use[idx], memory_order_relaxed); + return last_use == 0 || now - last_use >= DAEMON_LIMITS_HOST_EVICT_IDLE_SEC; +} + +/* Emit at most one "per-source table full" warning per + * DAEMON_LIMITS_HOST_FULL_WARN_SEC across all forked children. Called from a + * normal (non-signal) child path, so logging is safe here. */ +static void host_warn_table_full(DaemonLimitRegistry* registry, long long now) { + long long last = atomic_load_explicit(®istry->host_full_warn, memory_order_relaxed); + if (last != 0 && now - last < DAEMON_LIMITS_HOST_FULL_WARN_SEC) + return; + if (atomic_compare_exchange_strong_explicit(®istry->host_full_warn, &last, now, + memory_order_relaxed, memory_order_relaxed)) { + log_message(LOG_LEVEL_WARNING, + "daemon: per-source registry is full (%d slots) and no bucket can be reclaimed; " + "'max connections per host' and the auth lockout are temporarily not enforced for " + "new sources (the per-module cap and host ACLs still apply)", + registry->host_slots); + } +} + /* Find or insert the bucket for `peer_ip`. Insertion is a lock-free CAS so two - * forked children racing on the same source converge on one bucket. Returns -1 - * when the table is full or the address is unparseable (callers fail open: the - * global/module caps and ACLs still apply). */ + * forked children racing on the same source converge on one bucket. + * + * When the probe finds no empty bucket it reclaims, via a key CAS, the first + * bucket that is reclaimable (expired lockout or idle, and no active + * connection) and resets its counters. This bounds the table's lifetime so it + * cannot fill permanently and stay fail-open. Returns -1 only when the address + * is unparseable or the table is genuinely full of live/locked buckets + * (callers fail open: the global/module caps and ACLs still apply). */ static int host_intern(DaemonLimitRegistry* registry, const char* peer_ip) { bool ok = false; uint64_t key = daemon_limits_host_hash(peer_ip, &ok); if (!ok) return -1; + long long now = (long long)time(NULL); size_t mask = (size_t)registry->host_slots - 1; size_t start = (size_t)(key & mask); - for (size_t i = 0; i < (size_t)registry->host_slots; i++) { - size_t idx = (start + i) & mask; - uint64_t current = atomic_load_explicit(®istry->host_key[idx], memory_order_acquire); - if (current == key) - return (int)idx; - if (current == 0) { - uint64_t expected = 0; - if (atomic_compare_exchange_strong_explicit(®istry->host_key[idx], &expected, key, - memory_order_acq_rel, memory_order_acquire)) - return (int)idx; - if (atomic_load_explicit(®istry->host_key[idx], memory_order_acquire) == key) + /* A couple of passes bound the work: the first normally claims/seeds a bucket; + * a lost eviction CAS retries once against the freshly observed table. */ + for (int pass = 0; pass < 2; pass++) { + int evict = -1; + uint64_t evict_key = 0; + for (size_t i = 0; i < (size_t)registry->host_slots; i++) { + size_t idx = (start + i) & mask; + uint64_t current = atomic_load_explicit(®istry->host_key[idx], memory_order_acquire); + if (current == key) { + atomic_store_explicit(®istry->host_last_use[idx], now, memory_order_relaxed); return (int)idx; + } + if (current == 0) { + uint64_t expected = 0; + if (atomic_compare_exchange_strong_explicit(®istry->host_key[idx], &expected, key, + memory_order_acq_rel, memory_order_acquire)) { + atomic_store_explicit(®istry->host_last_use[idx], now, memory_order_relaxed); + return (int)idx; + } + if (atomic_load_explicit(®istry->host_key[idx], memory_order_acquire) == key) { + atomic_store_explicit(®istry->host_last_use[idx], now, memory_order_relaxed); + return (int)idx; + } + continue; /* another child won this empty bucket; keep probing */ + } + if (evict < 0 && host_bucket_reclaimable(registry, idx, now)) { + evict = (int)idx; + evict_key = current; + } } + if (evict >= 0) { + uint64_t expected = evict_key; + if (atomic_compare_exchange_strong_explicit(®istry->host_key[evict], &expected, key, + memory_order_acq_rel, memory_order_acquire)) { + /* The bucket now belongs to the new source; clear the evicted source's + * stale lockout/failure state. */ + atomic_store_explicit(®istry->host_active[evict], 0, memory_order_relaxed); + atomic_store_explicit(®istry->host_fail[evict], 0, memory_order_relaxed); + atomic_store_explicit(®istry->host_until[evict], 0, memory_order_relaxed); + atomic_store_explicit(®istry->host_last_use[evict], now, memory_order_relaxed); + return evict; + } + continue; /* lost the race; re-probe with fresh observations */ + } + break; /* no free and no reclaimable bucket: genuinely full */ } + host_warn_table_full(registry, now); return -1; } @@ -143,6 +236,8 @@ DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int p max_slots = DAEMON_LIMITS_MAX_SLOTS; if (module_count < 1) module_count = 1; + if (module_count > DAEMON_LIMITS_MAX_MODULES) + module_count = DAEMON_LIMITS_MAX_MODULES; if (per_host_cap < 0) per_host_cap = 0; if (lockout_threshold < 0) @@ -167,7 +262,7 @@ DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int p size_t module_bytes = round_up((size_t)module_count * sizeof(_Atomic int), 16); size_t host_key_bytes = round_up((size_t)host_slots * sizeof(_Atomic uint64_t), 16); size_t host_int_bytes = round_up((size_t)host_slots * sizeof(_Atomic int), 16) * 2; - size_t host_until_bytes = round_up((size_t)host_slots * sizeof(_Atomic long long), 16); + size_t host_until_bytes = round_up((size_t)host_slots * sizeof(_Atomic long long), 16) * 2; size_t total = header + slot_bytes + module_bytes + host_key_bytes + host_int_bytes + host_until_bytes + 16; @@ -205,6 +300,8 @@ DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int p cursor += (size_t)host_slots * sizeof(_Atomic int); cursor = (unsigned char*)round_up((size_t)(uintptr_t)cursor, 16); registry->host_until = (atomic_llong*)cursor; + cursor += (size_t)host_slots * sizeof(_Atomic long long); + registry->host_last_use = (atomic_llong*)cursor; for (int i = 0; i < max_slots; i++) { atomic_store(®istry->slot_module[i], -1); @@ -243,25 +340,12 @@ void daemon_limits_set_slot_pid(DaemonLimitRegistry* registry, int slot, long pi void daemon_limits_reclaim_slot(DaemonLimitRegistry* registry, int slot) { if (!registry || slot < 0 || slot >= registry->max_slots) return; - int previous = - atomic_exchange_explicit(®istry->slot_state[slot], SLOT_FREE, memory_order_acq_rel); - if (previous == SLOT_REGISTERED) { - int module = atomic_load(®istry->slot_module[slot]); - int host = atomic_load(®istry->slot_host[slot]); - if (module >= 0 && module < registry->module_count) { - int current = atomic_load(®istry->module_active[module]); - while (current > 0 && - !atomic_compare_exchange_weak(®istry->module_active[module], ¤t, current - 1)) - ; - } - if (host >= 0 && host < registry->host_slots) { - int current = atomic_load(®istry->host_active[host]); - while (current > 0 && - !atomic_compare_exchange_weak(®istry->host_active[host], ¤t, current - 1)) - ; - } - } - atomic_store(®istry->slot_pid[slot], 0); + atomic_exchange_explicit(®istry->slot_state[slot], SLOT_FREE, memory_order_acq_rel); + atomic_store_explicit(®istry->slot_pid[slot], 0, memory_order_relaxed); + /* The module/host occupancy arrays are derived from the slot table; do not + * decrement here or a SIGKILL between a child's increment and its REGISTERED + * publish would leak a count. Callers that need the derived counts call + * daemon_limits_recompute. */ } void daemon_limits_reclaim_pid(DaemonLimitRegistry* registry, long pid) { @@ -277,6 +361,29 @@ void daemon_limits_reclaim_pid(DaemonLimitRegistry* registry, long pid) { } } +void daemon_limits_recompute(DaemonLimitRegistry* registry) { + if (!registry) + return; + /* Zero the derived arrays, then re-derive solely from the REGISTERED slots. + * A child that was SIGKILLed after incrementing a counter but before + * publishing REGISTERED is not counted, and its leaked increment is erased by + * the zeroing, so the leak cannot persist. */ + for (int m = 0; m < registry->module_count; m++) + atomic_store_explicit(®istry->module_active[m], 0, memory_order_relaxed); + for (int h = 0; h < registry->host_slots; h++) + atomic_store_explicit(®istry->host_active[h], 0, memory_order_relaxed); + for (int i = 0; i < registry->max_slots; i++) { + if (atomic_load_explicit(®istry->slot_state[i], memory_order_acquire) != SLOT_REGISTERED) + continue; + int module = atomic_load_explicit(®istry->slot_module[i], memory_order_relaxed); + if (module >= 0 && module < registry->module_count) + atomic_fetch_add_explicit(®istry->module_active[module], 1, memory_order_relaxed); + int host = atomic_load_explicit(®istry->slot_host[i], memory_order_relaxed); + if (host >= 0 && host < registry->host_slots) + atomic_fetch_add_explicit(®istry->host_active[host], 1, memory_order_relaxed); + } +} + DaemonLimitResult daemon_limits_register(DaemonLimitRegistry* registry, int slot, int module_index, const char* peer_ip, int module_cap) { if (!registry || slot < 0 || slot >= registry->max_slots) @@ -287,7 +394,7 @@ DaemonLimitResult daemon_limits_register(DaemonLimitRegistry* registry, int slot return DAEMON_LIMIT_UNAVAILABLE; int host = -1; - if (registry->per_host_cap > 0 || registry->lockout_threshold > 0) + if (registry_tracks_hosts(registry)) host = host_intern(registry, peer_ip); int module_count = atomic_fetch_add(®istry->module_active[module_index], 1) + 1; diff --git a/src/shared/daemon_limits.h b/src/shared/daemon_limits.h index 47bfb0b..b6d89d2 100644 --- a/src/shared/daemon_limits.h +++ b/src/shared/daemon_limits.h @@ -34,6 +34,20 @@ * already collapsed to IPv4 by utils_fd_peer_ip); it is interned into an * open-addressed, linear-probing table keyed by a 64-bit hash. The same table * also carries the cross-process auth-failure counter and lockout deadline. + * + * Per-source table lifetime: a bucket's key is never cleared back to empty (that + * would break every later probe chain that passed through it). Instead the + * table has a bounded-lifetime eviction policy: when no empty bucket exists, the + * first bucket that is reclaimable -- no active connection AND (its lockout + * deadline has passed OR it has been idle for + * DAEMON_LIMITS_HOST_EVICT_IDLE_SEC) -- is atomically repurposed for the new + * source via a CAS of its key, and its counters are reset. The table therefore + * cannot fill permanently, and a full table degrades to fail-open for the + * per-source cap/lockout of new sources (the per-module cap and host ACLs still + * apply) instead of staying fail-open forever. A rate-limited warning is logged + * on the fail-open path. The eviction race with a concurrent + * registration/reclaim on the same bucket is benign: it can at worst lose one + * source's counter (fail-open), never corrupt memory or the module caps. */ typedef struct DaemonLimitRegistry DaemonLimitRegistry; @@ -51,13 +65,27 @@ typedef enum { #define DAEMON_LIMITS_MAX_SLOTS 65536 #define DAEMON_LIMITS_MAX_HOST_SLOTS 65536 #define DAEMON_LIMITS_NO_SLOT (-1) +/* Upper bound on `module_count`, matching daemon_conf.h's DAEMON_CONF_MAX_MODULES + * (asserted in daemon_limits.c) so a caller can never size the per-module counter + * array larger than the config parser can produce. */ +#define DAEMON_LIMITS_MAX_MODULES 256 + +/* Per-source table lifetime: a bucket with no active connection and no pending + * lockout is reclaimable once it has been idle this long, so a flood of distinct + * sources cannot pin the table full forever. A bucket whose lockout deadline + * has passed is reclaimable immediately (independent of this idle window). */ +#define DAEMON_LIMITS_HOST_EVICT_IDLE_SEC 300 +/* Minimum spacing between "per-source table is full" warnings, so a table-full + * attack cannot flood the log. */ +#define DAEMON_LIMITS_HOST_FULL_WARN_SEC 60 /* Create the shared registry in the calling (parent) process. `max_slots` is * the number of concurrently live children to track (clamped to * [DAEMON_LIMITS_MIN_SLOTS, DAEMON_LIMITS_MAX_SLOTS]); `module_count` is the - * number of daemon modules (clamped to >= 1); `per_host_cap` and the lockout - * pair come from the daemon config (0 disables). Returns NULL on failure (e.g. - * mmap allocation); callers must degrade gracefully (global cap + ACLs still + * number of daemon modules (clamped to + * [1, DAEMON_LIMITS_MAX_MODULES]); `per_host_cap` and the lockout pair come + * from the daemon config (0 disables). Returns NULL on failure (e.g. mmap + * allocation); callers must degrade gracefully (global cap + ACLs still * apply). */ DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int per_host_cap, int lockout_threshold, int lockout_duration_sec); @@ -70,16 +98,33 @@ void daemon_limits_destroy(DaemonLimitRegistry* registry); int daemon_limits_claim_slot(DaemonLimitRegistry* registry); /* Parent side: record the forked child's pid in a claimed slot. */ void daemon_limits_set_slot_pid(DaemonLimitRegistry* registry, int slot, long pid); -/* Parent side: release a slot, decrementing the module/per-source counters when - * the slot was actually REGISTERED. Idempotent. */ +/* Parent side: release a slot. The slot becomes FREE; the module/per-source + * occupancy arrays are DERIVED state and are only refreshed by + * daemon_limits_recompute, which callers must invoke afterwards when they rely + * on the derived counts (the SIGCHLD handler batches one recompute for the whole + * reap). Idempotent. */ void daemon_limits_reclaim_slot(DaemonLimitRegistry* registry, int slot); -/* Parent SIGCHLD side: reclaim the slot owned by `pid` (no-op when not found). */ +/* Parent SIGCHLD side: release the slot owned by `pid` (no-op when not found). + * Like reclaim_slot this does not touch the derived occupancy arrays; call + * daemon_limits_recompute after a batch of releases. */ void daemon_limits_reclaim_pid(DaemonLimitRegistry* registry, long pid); +/* Parent side (async-signal-safe; atomics only, no malloc/log): rebuild + * module_active[] / host_active[] from scratch by scanning the REGISTERED slots. + * The slot table is the single source of truth, so this self-heals any + * count leaked by a child that was SIGKILLed mid-registration (it zeroes the + * arrays and re-derives them). Bounded by max_slots + host_slots. A + * registration racing this call can be transiently undercounted until the next + * recompute, which can only relax a cap briefly -- never corrupt memory. */ +void daemon_limits_recompute(DaemonLimitRegistry* registry); + /* Child side: admit the connection for `module_index` from `peer_ip`. Always * tracks the module/per-source occupancy (so the parent's reclaim is * symmetric); when `module_cap` > 0 it additionally enforces the per-module - * cap. Returns DAEMON_LIMIT_OK and publishes the slot, or a refusal reason. */ + * cap. A NULL/empty or non-numeric `peer_ip` skips the per-source track (the + * callers use that to exempt a trusted loopback peer from the per-host cap; the + * per-module cap still applies). Returns DAEMON_LIMIT_OK and publishes the + * slot, or a refusal reason. */ DaemonLimitResult daemon_limits_register(DaemonLimitRegistry* registry, int slot, int module_index, const char* peer_ip, int module_cap); diff --git a/src/shared/transport_tcp.c b/src/shared/transport_tcp.c index e73dbce..9fedfb7 100644 --- a/src/shared/transport_tcp.c +++ b/src/shared/transport_tcp.c @@ -9,6 +9,7 @@ #include #include #include +#include #include #include #include @@ -39,9 +40,25 @@ static void sigchld_handler(int sig) { g_active_connections--; daemon_limits_reclaim_pid(g_limit_registry, (long)pid); } + /* Re-derive the occupancy counters once for the whole reap batch. The slot + * table is the source of truth, so this self-heals any count leaked by a child + * SIGKILLed mid-registration. Atomics only: async-signal-safe. */ + if (g_limit_registry) + daemon_limits_recompute(g_limit_registry); errno = saved_errno; } +/* Reset a signal to its default action with sigaction (preferred over + * signal(3), whose semantics are implementation-defined). Used in the forked + * child before it can spawn any thread. */ +static void reset_signal_default(int sig) { + struct sigaction action; + memset(&action, 0, sizeof(action)); + action.sa_handler = SIG_DFL; + sigemptyset(&action.sa_mask); + sigaction(sig, &action, NULL); +} + /* Map a listen socket's address to its numeric port for logging, independent * of whether it is an IPv4 or IPv6 sockaddr. */ static unsigned short server_address_port(const struct sockaddr_storage* addr) { @@ -160,7 +177,16 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil log_perror("Could not listen on port!"); return; } - signal(SIGCHLD, sigchld_handler); + /* SIGCHLD via sigaction (not signal(3)); SA_RESTART keeps accept(2) from + * failing with EINTR, and SA_NOCLDSTOP only notifies on child exit. The + * accept loop is single-threaded at this point, so installing here cannot race + * a worker thread. */ + struct sigaction chld_action; + memset(&chld_action, 0, sizeof(chld_action)); + chld_action.sa_handler = sigchld_handler; + sigemptyset(&chld_action.sa_mask); + chld_action.sa_flags = SA_RESTART | SA_NOCLDSTOP; + sigaction(SIGCHLD, &chld_action, NULL); g_limit_registry = server->limit_registry; while (1) { struct sockaddr_storage client_addr; @@ -197,15 +223,21 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil g_current_slot = slot; /* Block SIGCHLD across fork() and the parent's pid publication: a child * that exits immediately must not be reaped before its slot records its - * pid, which would leak the slot and its module/source counts. */ + * pid, which would leak the slot and its module/source counts. Use + * pthread_sigmask rather than sigprocmask so the behavior is well defined + * even if this process ever gains threads: the mask is per-thread, the fork + * copies only the calling thread, and the child inherits this thread's + * blocked mask until it restores `previous` below. No thread exists yet at + * this point, and none is created before the mask is restored, so the + * critical window is race-free. */ sigset_t blocked; sigset_t previous; sigemptyset(&blocked); sigaddset(&blocked, SIGCHLD); - sigprocmask(SIG_BLOCK, &blocked, &previous); + pthread_sigmask(SIG_BLOCK, &blocked, &previous); pid_t pid = fork(); if (pid == 0) { - sigprocmask(SIG_SETMASK, &previous, NULL); + pthread_sigmask(SIG_SETMASK, &previous, NULL); /* Connection children must not run the parent's global cleanup(): it * frees state (credentials / daemon conf) that the child's worker * threads may still be reading and closes fd numbers the child could @@ -213,9 +245,9 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil * terminates the child directly; SIGCHLD is reset too since a child * must never reap the parent's children. This runs before the child * spawns any thread, so it cannot race one. */ - signal(SIGINT, SIG_DFL); - signal(SIGTERM, SIG_DFL); - signal(SIGCHLD, SIG_DFL); + reset_signal_default(SIGINT); + reset_signal_default(SIGTERM); + reset_signal_default(SIGCHLD); close(server->file_descriptor); child_fn(fd, child_ctx); _exit(0); @@ -227,7 +259,7 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil /* fork() failed: release the reservation so the slot is not leaked. */ daemon_limits_reclaim_slot(server->limit_registry, slot); } - sigprocmask(SIG_SETMASK, &previous, NULL); + pthread_sigmask(SIG_SETMASK, &previous, NULL); close(fd); } } diff --git a/tests/test_daemon_limits.c b/tests/test_daemon_limits.c index 2815e0a..8f36bb0 100644 --- a/tests/test_daemon_limits.c +++ b/tests/test_daemon_limits.c @@ -2,7 +2,9 @@ #include "daemon_limits.h" #include "test_utils.h" #include +#include #include +#include #include /* The per-source hash is a pure helper: numeric addresses hash to a nonzero, @@ -72,6 +74,7 @@ static void test_daemon_limits_module_cap() { daemon_limits_reclaim_slot(registry, slot0); daemon_limits_reclaim_slot(registry, slot1); + daemon_limits_recompute(registry); int slot4 = daemon_limits_claim_slot(registry); EXPECT_TRUE(slot4 >= 0); EXPECT_EQ_INT(daemon_limits_register(registry, slot4, 0, "10.0.0.4", 2), DAEMON_LIMIT_OK); @@ -94,6 +97,7 @@ static void test_daemon_limits_host_cap() { EXPECT_EQ_INT(daemon_limits_register(registry, slot2, 0, "10.0.0.2", 0), DAEMON_LIMIT_OK); /* Reclaiming the first source frees its per-host allowance. */ daemon_limits_reclaim_slot(registry, slot0); + daemon_limits_recompute(registry); EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.1", 0), DAEMON_LIMIT_OK); daemon_limits_destroy(registry); @@ -115,6 +119,7 @@ static void test_daemon_limits_reclaim_pid() { DAEMON_LIMIT_MODULE_FULL); daemon_limits_reclaim_pid(registry, 4242); + daemon_limits_recompute(registry); EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.1", 1), DAEMON_LIMIT_OK); /* Reclaiming an unknown pid is a no-op. */ daemon_limits_reclaim_pid(registry, 999999); @@ -179,16 +184,89 @@ static void test_daemon_limits_fork_shared() { DAEMON_LIMIT_MODULE_FULL); /* The parent reclaims the dead child's slot by pid. */ daemon_limits_reclaim_pid(registry, (long)pid); + daemon_limits_recompute(registry); EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.2", 1), DAEMON_LIMIT_OK); daemon_limits_destroy(registry); } +/* The occupancy arrays are derived from the slot table: recompute rebuilds them + * and is the self-heal path the SIGCHLD handler uses after a child dies. */ +static void test_daemon_limits_recompute() { + DaemonLimitRegistry* registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 2, 1, 0, 0); + EXPECT_NOT_NULL(registry); + int slot0 = daemon_limits_claim_slot(registry); + int slot1 = daemon_limits_claim_slot(registry); + int slot2 = daemon_limits_claim_slot(registry); + EXPECT_TRUE(slot0 >= 0 && slot1 >= 0 && slot2 >= 0); + EXPECT_EQ_INT(daemon_limits_register(registry, slot0, 0, "10.0.0.1", 0), DAEMON_LIMIT_OK); + EXPECT_EQ_INT(daemon_limits_register(registry, slot1, 0, "10.0.0.2", 0), DAEMON_LIMIT_OK); + + /* Recompute is idempotent and re-derives the same counts from REGISTERED + * slots (a CLAIMED slot is never counted). */ + daemon_limits_recompute(registry); + daemon_limits_recompute(registry); + EXPECT_EQ_INT(daemon_limits_register(registry, slot2, 0, "10.0.0.3", 2), + DAEMON_LIMIT_MODULE_FULL); + + /* Freeing a slot and recomputing releases its module/per-source count. */ + daemon_limits_reclaim_slot(registry, slot0); + daemon_limits_recompute(registry); + EXPECT_EQ_INT(daemon_limits_register(registry, slot2, 0, "10.0.0.3", 2), DAEMON_LIMIT_OK); + daemon_limits_destroy(registry); +} + +/* The per-source table has a bounded lifetime. When every bucket is occupied + * but not yet reclaimable, a new source is fail-open: the per-host cap is not + * enforced and the probe must terminate. Once the occupied buckets' lockouts + * expire (or they go idle), a new source reclaims a bucket and enforcement comes + * back. This covers the "table never evicts -> cap silently fails open forever" + * review finding. */ +static void test_daemon_limits_host_table_eviction() { + char ip[32]; + + /* Part A: all buckets locked out with a long deadline and no active + * connection are not reclaimable yet. A new source cannot be interned, so the + * per-host cap is documented fail-open (both connections admitted) -- and the + * bounded probe returns instead of looping forever. */ + DaemonLimitRegistry* registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 1, 1, 1, 300); + EXPECT_NOT_NULL(registry); + for (int i = 0; i < 64; i++) { + snprintf(ip, sizeof(ip), "10.0.0.%d", i + 1); + daemon_limits_auth_record_failure(registry, ip); + } + int a = daemon_limits_claim_slot(registry); + int b = daemon_limits_claim_slot(registry); + EXPECT_TRUE(a >= 0 && b >= 0); + EXPECT_EQ_INT(daemon_limits_register(registry, a, 0, "10.9.9.9", 0), DAEMON_LIMIT_OK); + EXPECT_EQ_INT(daemon_limits_register(registry, b, 0, "10.9.9.9", 0), DAEMON_LIMIT_OK); + daemon_limits_destroy(registry); + + /* Part B: with an already-expired lockout every bucket is reclaimable, so a + * new source reclaims one and the per-host cap is enforced again. */ + registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 1, 1, 1, 1); + EXPECT_NOT_NULL(registry); + for (int i = 0; i < 64; i++) { + snprintf(ip, sizeof(ip), "10.0.0.%d", i + 1); + daemon_limits_auth_record_failure(registry, ip); + } + struct timespec pause = {2, 0}; + nanosleep(&pause, NULL); + int c = daemon_limits_claim_slot(registry); + int d = daemon_limits_claim_slot(registry); + EXPECT_TRUE(c >= 0 && d >= 0); + EXPECT_EQ_INT(daemon_limits_register(registry, c, 0, "10.9.9.9", 0), DAEMON_LIMIT_OK); + EXPECT_EQ_INT(daemon_limits_register(registry, d, 0, "10.9.9.9", 0), DAEMON_LIMIT_HOST_FULL); + daemon_limits_destroy(registry); +} + void test_daemon_limits() { test_daemon_limits_host_hash(); test_daemon_limits_slots(); test_daemon_limits_module_cap(); test_daemon_limits_host_cap(); test_daemon_limits_reclaim_pid(); + test_daemon_limits_recompute(); test_daemon_limits_auth_lockout(); + test_daemon_limits_host_table_eviction(); test_daemon_limits_fork_shared(); } From 25909110ac5ded4083a9380f0c1739956913ea59 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:50:58 +0200 Subject: [PATCH 6/8] fix(daemon): exempt trusted loopback peers from per-host limits Every client on loopback shares the 127.0.0.1 identity, so counting them against 'max connections per host' or the default-on auth lockout lets one local client deny service to all the others (and makes a shared-NAT/proxy address a natural DoS vector for remote clients). Use utils_fd_peer_is_local (fail-closed) in the daemon gate to exempt a provably local peer from the per-source cap and the auth lockout while keeping the per-module and global caps. Remote peers are unchanged. Document the shared-NAT/proxy identity limitation and the loopback exemption in README/RSYNC_COMPAT/CHANGELOG, update the integration test to assert the exemption, and fix the README 'auth failure delay' cap (5000, not 60000). --- CHANGELOG.md | 9 ++++++++- README.md | 22 +++++++++++++++++++++- RSYNC_COMPAT.md | 2 +- src/server/server.c | 30 ++++++++++++++++++++++++------ tests/integration/test_daemon.py | 27 +++++++++++++++------------ 5 files changed, 69 insertions(+), 21 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a641f4b..fe26d5a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,7 +14,14 @@ run the same version because the handshake is strict. forks one child per connection, the counters live in an anonymous shared mapping created before the accept loop and reclaimed by the parent's `SIGCHLD` handler, so the per-module, per-source and auth-failure state is - shared across every child (including after `SIGKILL`). + shared across every child (including after `SIGKILL`). The per-source table + now has a bounded lifetime (expired-lockout/idle entries are reclaimed, with a + rate-limited warning when it is genuinely full), and the occupancy counters are + re-derived from the shared slot table on every child exit so a child killed + mid-registration cannot leak a count. Trusted loopback peers are exempt from the + per-host cap and the auth lockout (they share one address); clients behind a + shared NAT/proxy still share a single per-host budget and lockout, which is + documented. ## [2.20.0] - 2026-09-13 diff --git a/README.md b/README.md index 58d0d10..4e7733c 100644 --- a/README.md +++ b/README.md @@ -512,7 +512,7 @@ and `address`, the global section accepts: source IP, default 0 (unlimited). Enforced across all forked connection children through a shared registry. - `auth failure delay = MS` — milliseconds to sleep after a failed - authentication, default 500. `0` disables it and the value is capped at 60000, + authentication, default 500. `0` disables it and the value is capped at 5000, so online password guessing is rate-limited per connection. Successful auths are never delayed. - `auth lockout threshold = N` — number of failed authentications from one source @@ -528,6 +528,26 @@ and `address`, the global section accepts: A `[module]` may also set `max connections` (0 = unlimited; enforced per module across all connection children) and its own `hosts allow`/`hosts deny`. +The per-host cap and the shared auth lockout identify a source by its numeric +peer IP. **Loopback peers (127.0.0.0/8, IPv6 `::1`) are exempt**: every local +client shares that one address, so counting or locking them out would let one +local process deny service to all the others. The per-module and global +`max connections` caps still apply to loopback. Because the key is the peer IP, +`max connections per host` and `auth lockout` also cannot distinguish clients +behind the same NAT, proxy, or reverse-proxy address — they share one budget and +one lockout counter, so an over-aggressive lockout can affect unrelated users +behind that address. Prefer TLS client certificates (`--client-cn`) plus +`hosts allow`/`hosts deny` for per-client policy when clients share an address, +and size `auth lockout threshold` accordingly. + +The shared per-source table has a bounded lifetime: an entry with no live +connection is reclaimed once its lockout has expired, or after it has been idle +(300 s). If every entry is still live or locked, a new source is admitted without +per-host accounting (fail open) and a rate-limited warning is logged; the +per-module cap and host ACLs still apply. The occupancy counters are re-derived +from the shared slot table after every child exit, so a child killed mid-transfer +(or mid-registration) cannot leak a slot or an occupancy count. + Host patterns are `*` (match all), IPv4/IPv6 literals, or IPv4/IPv6 CIDR (`10.0.0.0/8`, `2001:db8::/32`). Hostnames are not resolved, so hostname globs are rejected at parse time rather than silently never matching. A matching diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index f0d4ba2..df03870 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -637,7 +637,7 @@ now transmits targets (the prior behavior was broken/partial); its status moved - **Config grammar** (`fastsyncd.conf`): line-based; an implicit global section first, then `[module]` sections. Keys are case-insensitive, values are trimmed and may be wrapped in one layer of double quotes (`path = "/srv/my dir"`). `#` and `;` at the start of a line (after leading whitespace) are full-line comments; inline comments and `\` continuations are not supported. Lines are bounded (4096 chars), and at most 256 `[module]` sections are accepted. Global keys: `port` (default 873), `motd file` (the daemon sends its bounded, escaped content to a client after the module gate/auth accepts, unless the client passes `--no-motd`), `address` (optional bind address), `max connections` (positive integer cap on concurrent connections, default 100; 0/negative/garbage is a parse error), `max connections per host` (concurrent-connection cap per source IP, default 0 = unlimited), `auth failure delay` (milliseconds to sleep after a failed authentication, default 500; 0 disables, capped at 5000), `auth lockout threshold` (failed authentications from one source before lockout, default 10; 0 disables), `auth lockout duration` (seconds a locked-out source is refused, default 300), `hosts allow` and `hosts deny` (comma- and/or whitespace-separated host access patterns — see the host access control note below). Module keys: `path` (required; the daemon-side authorized root for that module), `read only` (yes/no/true/false/1/0, default no), `client owner` (yes/no/true/false/1/0, default no; opts the module into client-chosen ownership — see below), `auth users` (comma list), `max connections` (optional per-module cap, 0 = unlimited; enforced across all connection children), `hosts allow`/`hosts deny` (per-module host access lists). **Unknown keys and malformed lines are parse-and-reject errors** (never silently ignored), so a typo cannot change what a module serves. - **Host access control (`hosts allow`/`hosts deny`):** both keys accept a comma- and/or whitespace-separated list of patterns and may appear globally and/or per module (multiple config-file lines append; a `--dparam` override replaces). Supported patterns are `*` (match all), an IPv4 or IPv6 literal (`10.0.0.1`, `2001:db8::1`), and an IPv4/IPv6 CIDR (`10.0.0.0/8`, `2001:db8::/32`). Hostname patterns are **not** supported: because the peer is always a numeric address and no reverse DNS is performed, a hostname/glob pattern would silently never match, so it is rejected at load time (fail-closed) instead of being accepted as a dead rule. An IPv4 peer on a dual-stack IPv6 listener is normalized from its `::ffff:a.b.c.d` form so IPv4 patterns match it. rsync-like semantics: a matching `hosts deny` rejects; if any `hosts allow` entries exist, a peer matching none of them is rejected; deny takes precedence over allow. The daemon enforces the global list first, then the selected module's list, **before authentication** in `server_module_gate`, with an audit log line naming the peer, the module and the outcome. The numeric peer address is obtained with `getpeername`+`inet_ntop` (`utils_fd_peer_ip`, handling both address families); when it cannot be obtained a module with any ACL fails closed (refused), while an ACL-free module continues and logs at debug. A malformed pattern (e.g. an out-of-range CIDR prefix) is a parse error at load time. -- **Connection caps, shared registry and auth lockout:** the global `max connections` key (default 100) is plumbed into the listener (`transport_tcp.c`), which rejects a connection once the accept-loop parent's active-child count reaches it; the IPv4/IPv6 peer is logged for every accepted connection. Because the listener forks one child per connection, the per-module `max connections` cap, the global `max connections per host` cap, and the auth-failure counter live in a fixed-size registry carved from an anonymous shared mapping (`daemon_limits.c`, `mmap(MAP_SHARED|MAP_ANONYMOUS)`) created by the parent before the accept loop, so every forked child shares the same counters (C11 atomics only — never a pthread lock, which can deadlock in a forked child). The parent reserves a registry slot per accepted connection and the child records the selected module and source IP once known; the parent's `SIGCHLD` handler reclaims the slot when the child dies (including `SIGKILL`), decrementing the per-module and per-source counts. The per-module cap (0 = unlimited) is enforced after the module lookup and before auth; per-source identity reuses the normalized numeric peer address (`utils_fd_peer_ip`, IPv4-mapped IPv6 collapsed to IPv4). A failed authentication increments the shared per-source failure count and, once `auth lockout threshold` (default 10; 0 disables) is reached, the source is refused for `auth lockout duration` seconds (default 300) before any challenge is sent, even when the next attempt is handled by a different forked child; a successful authentication clears the counter. On a failed authentication the per-connection child still sleeps the global `auth failure delay` (default 500 ms, 0 disables, capped at 5000) via `nanosleep`, rate-limiting online guessing without delaying a success. A missing registry (allocation failure) degrades to the global cap and host ACLs rather than refusing to start. +- **Connection caps, shared registry and auth lockout:** the global `max connections` key (default 100) is plumbed into the listener (`transport_tcp.c`), which rejects a connection once the accept-loop parent's active-child count reaches it; the IPv4/IPv6 peer is logged for every accepted connection. Because the listener forks one child per connection, the per-module `max connections` cap, the global `max connections per host` cap, and the auth-failure counter live in a fixed-size registry carved from an anonymous shared mapping (`daemon_limits.c`, `mmap(MAP_SHARED|MAP_ANONYMOUS)`) created by the parent before the accept loop, so every forked child shares the same counters (C11 atomics only — never a pthread lock, which can deadlock in a forked child). The parent reserves a registry slot per accepted connection and the child records the selected module and source IP once known; the parent's `SIGCHLD` handler reclaims the slot when the child dies (including `SIGKILL`) and re-derives the per-module and per-source occupancy counts from the surviving REGISTERED slots, so a child killed mid-registration cannot leak a count. The per-source table has a bounded lifetime: an entry with no live connection is reclaimed after its lockout expires or it has been idle (300 s); if the table is genuinely full the per-source cap/lockout fails open for new sources (per-module cap and ACLs still apply) with a rate-limited warning. The per-module cap (0 = unlimited) is enforced after the module lookup and before auth; per-source identity reuses the normalized numeric peer address (`utils_fd_peer_ip`, IPv4-mapped IPv6 collapsed to IPv4), and a trusted loopback peer (127.0.0.0/8 / `::1`, `utils_fd_peer_is_local`) is exempt from the per-source cap and the auth lockout because all local clients share one address (the per-module/global caps still apply). Clients behind a shared NAT/proxy address likewise share one per-source budget and lockout counter. A failed authentication increments the shared per-source failure count and, once `auth lockout threshold` (default 10; 0 disables) is reached, the source is refused for `auth lockout duration` seconds (default 300) before any challenge is sent, even when the next attempt is handled by a different forked child; a successful authentication clears the counter. On a failed authentication the per-connection child still sleeps the global `auth failure delay` (default 500 ms, 0 disables, capped at 5000) via `nanosleep`, rate-limiting online guessing without delaying a success. A missing registry (allocation failure) degrades to the global cap and host ACLs rather than refusing to start. - **Module selection & confinement:** the client requests a module with an rsync-style `host::module[/path]` destination. The module name crosses the wire as a trailing string on the config frame (bumping `PROTOCOL_VERSION` 2.14.0 → 2.15.0; the bump is required because the config-frame layout changed and the strict same-version handshake is what prevents a peer from desynchronizing on the new trailing field). The daemon looks the module up in ITS OWN config and uses the module's `path` as the authorized root through the exact same `configure_authorization` confinement the standalone server applies to `--destination-root` (`file_open_secure_parent`, `has_path_traversal`, `path_is_within`); the client never supplies the root, every client-chosen-ownership/super-user request is refused unless the module declares `client owner = yes` (the daemon's per-module opt-in, see below), and the operator `--no-super` veto forces super-user activities off for every daemon connection. The client's `/path` part is relative inside the module and is rejected if absolute or if it contains `..`. Unknown modules are refused before any data moves (the run fails cleanly at the config handshake). An absolute destination and a module request against a non-daemon server are also refused. - **`client owner` (client-chosen-ownership opt-in):** by default a daemon module refuses every request that would let the client pick an owner or ask for super-user activities — `--numeric-ids`, `--chown`, `--usermap`/`--groupmap`, `--fake-super`, `--copy-as`, and an explicit `--super` — at the config handshake (before `STATUS_OK`), because a daemon has no per-module opt-in for client-chosen ownership and any anonymous client could otherwise force arbitrary owner ids inside the module root. `client owner = yes` opts a single module in, allowing those requests within that module's root (the standalone listener and the SSH `--stdio` server always honor them for their single operator-authorized root). Without the opt-in the daemon also forces super-user **device** activity off for that connection — char/block device-node creation (`--devices`) and `--write-devices` — even under the default `AUTO` mode, so a non-opted module can never be made to `mknod` or write a raw device; those entries are skipped (not refused) so an ordinary `-a` push still succeeds without device nodes. The opt-in does **not** lift the privilege requirement: `--copy-as` still needs a root receiver, and the operator `--no-super` veto still forces super-user activities off for every connection. The daemon logs a prominent startup warning for each `client owner = yes` module so the operator's deliberate choice is visible. - **`read only` safe default:** every network transfer FastSync currently supports is a push that writes under the module root, so a `read only` module refuses the connection (clear server log "module is read only"; the client exits non-zero, nothing is transferred). A future pull/list operation can be opened up when it exists; the knob is already stored. diff --git a/src/server/server.c b/src/server/server.c index 6f3862c..6785b12 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -82,6 +82,13 @@ typedef struct ModuleGateContext { * not classify the peer; an ACL-configured module then fails closed. */ bool has_peer_ip; char peer_ip[INET6_ADDRSTRLEN]; + /* True when the peer is provably loopback (utils_fd_peer_is_local, fail + * closed). A trusted local/SSH peer is exempt from the per-host cap and the + * cross-process auth lockout: every loopback client shares the 127.0.0.1 + * identity, so counting/locking them out would let one local client deny + * service to (or leak lockout state about) all the others. The per-module and + * global caps still apply. */ + bool is_local; } ModuleGateContext; /* Server half of the SCRAM challenge/response (A7 remediation, protocol @@ -326,7 +333,11 @@ static const char* module_gate_check_limits(const Config* config, const DaemonMo int module_index = daemon_module_index(module); if (module_index < 0) return NULL; - const char* peer = (gate_ctx && gate_ctx->has_peer_ip) ? gate_ctx->peer_ip : ""; + /* A trusted loopback peer is exempt from the per-source cap: pass an + * unparseable peer so the registry skips per-source tracking, while the + * per-module cap below is still enforced. Remote peers are tracked normally. */ + const char* peer = + (!gate_ctx || gate_ctx->is_local || !gate_ctx->has_peer_ip) ? "" : gate_ctx->peer_ip; DaemonLimitResult result = daemon_limits_register(g_daemon_limits, slot, module_index, peer, module->max_connections); switch (result) { @@ -455,8 +466,10 @@ static ModuleAuthResult module_gate_authenticate(const Config* config, const Dae return MODULE_AUTH_ACCEPTED; /* Cross-process lockout: a source that failed too many authentications is * refused before the challenge is sent (the counter lives in the shared - * registry, so it spans every forked child and survives a child exit). */ - if (g_daemon_limits && gate_ctx && gate_ctx->has_peer_ip) { + * registry, so it spans every forked child and survives a child exit). A + * trusted loopback peer is exempt: all local clients share the 127.0.0.1 + * identity, so a lockout would let one deny the others. */ + if (g_daemon_limits && gate_ctx && gate_ctx->has_peer_ip && !gate_ctx->is_local) { int remaining = 0; if (daemon_limits_auth_locked(g_daemon_limits, gate_ctx->peer_ip, &remaining)) { log_message(LOG_LEVEL_ERROR, @@ -523,13 +536,14 @@ static ModuleAuthResult module_gate_authenticate(const Config* config, const Dae free(escaped_user); /* Count the failure in the shared registry (locks the source out once the * configured threshold is reached) and rate-limit online guessing per - * connection (no delay on success). */ - if (g_daemon_limits && gate_ctx->has_peer_ip) + * connection (no delay on success). A loopback peer is exempt from the + * shared counter. */ + if (g_daemon_limits && gate_ctx->has_peer_ip && !gate_ctx->is_local) daemon_limits_auth_record_failure(g_daemon_limits, gate_ctx->peer_ip); daemon_auth_failure_delay(); return MODULE_AUTH_TERMINATED; } - if (g_daemon_limits && gate_ctx->has_peer_ip) + if (g_daemon_limits && gate_ctx->has_peer_ip && !gate_ctx->is_local) daemon_limits_auth_record_success(g_daemon_limits, gate_ctx->peer_ip); char* escaped_user = output_escape(config->auth_user, config->eight_bit_output); log_message(LOG_LEVEL_INFO, "daemon module '%s': user '%s' from %s authenticated", config->module, @@ -636,6 +650,9 @@ static const char* server_module_gate(const Config* config, void* context) { utils_fd_peer_ip(gate_ctx->fd, gate_ctx->peer_ip, sizeof(gate_ctx->peer_ip)); if (!gate_ctx->has_peer_ip) log_message(LOG_LEVEL_DEBUG, "daemon module '%s': peer address unavailable", config->module); + /* utils_fd_peer_is_local is fail-closed (getpeername must succeed and report + * a loopback peer), so "cannot tell" is never treated as trusted. */ + gate_ctx->is_local = utils_fd_peer_is_local(gate_ctx->fd); } error = module_gate_check_hosts(config, module, gate_ctx); if (error) @@ -670,6 +687,7 @@ void handler(int file_descriptor) { gate_ctx.super_mode_override = -1; gate_ctx.has_peer_ip = false; gate_ctx.peer_ip[0] = '\0'; + gate_ctx.is_local = false; /* All teardown state starts empty so the single `done` epilogue is safe to * reach from any error path (including before the config frame arrives). */ Config* config = NULL; diff --git a/tests/integration/test_daemon.py b/tests/integration/test_daemon.py index 1ff1d9d..a7870e9 100644 --- a/tests/integration/test_daemon.py +++ b/tests/integration/test_daemon.py @@ -1221,11 +1221,14 @@ class TestDaemonConnectionLimits: CAPS_CONF = os.path.join(TEST_DATA_DIR, "fastsyncd_caps.conf") @pytest.mark.ci - def test_auth_lockout_is_shared_across_children(self): - """`auth lockout threshold = 1`: the first failed authentication locks the - source out for the cooldown in the SHARED registry, so a subsequent - correct-password attempt (a different forked child) is refused before a - SCRAM challenge is even sent.""" + def test_auth_lockout_exempts_trusted_loopback(self): + """`auth lockout threshold = 1`: a trusted loopback peer is EXEMPT from + the shared lockout because every local client shares the 127.0.0.1 + identity, so a single wrong password must not lock out correct-password + attempts (that would be a local denial of service). The shared + per-source lockout machinery itself is covered by the daemon_limits unit + tests; this locks in the loopback policy and the absence of a stale + "locked out" log line.""" port = _find_free_port() with open(self.LOCKOUT_CONF, "w") as f: f.write("port = %d\n" @@ -1241,21 +1244,21 @@ class TestDaemonConnectionLimits: try: d.start(self.LOCKOUT_CONF, port_override=port, extra_args=["--password-file", CRED_FILE], log_path=log_path) - before = _tree_file_count(AUTH_MODULE) log_before = os.path.getsize(log_path) if os.path.exists(log_path) else 0 - # First attempt: wrong password -> records failure #1 -> locks. + # First attempt: wrong password -> a failure is logged, but a loopback + # peer is not counted toward the lockout. wrong = _push_with_creds("127.0.0.1::locked", port, "alice", WRONG_PASS) assert wrong.returncode != 0 - # Second attempt: CORRECT password from the same source must still be - # refused by the shared lockout. + # Second attempt: the correct password from the same local source must + # still be accepted (no lockout), which also runs the SCRAM handshake + # to completion in a fresh forked child. right = _push_with_creds("127.0.0.1::locked", port, "alice", ALICE_PASS) - assert right.returncode != 0, "the shared auth lockout must refuse after threshold" - assert _tree_file_count(AUTH_MODULE) == before, "a locked-out source wrote data" + assert right.returncode == 0, (right.stderr or right.stdout) time.sleep(0.3) with open(log_path, "rb") as f: f.seek(log_before) tail = f.read().decode("utf-8", "replace") - assert "locked out" in tail, tail[-400:] + assert "locked out" not in tail, tail[-400:] finally: d.stop() From 0a7f5faea6e4739ce88ac7fb119e199c9fdca0aa Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:51:01 +0200 Subject: [PATCH 7/8] test: replace strcat with a bounds-checked append in daemon-conf test --- tests/test_daemon_conf.c | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/tests/test_daemon_conf.c b/tests/test_daemon_conf.c index a9113a5..b6f54bc 100644 --- a/tests/test_daemon_conf.c +++ b/tests/test_daemon_conf.c @@ -543,11 +543,19 @@ static void test_daemon_conf_module_count_capped() { size_t len = (cap + 8) * 32; char* body = malloc(len); EXPECT_NOT_NULL(body); + size_t used = 0; body[0] = '\0'; for (size_t i = 0; i < cap + 1; i++) { char line[48]; - snprintf(line, sizeof(line), "[m%zu]\npath = /x\n", i); - strcat(body, line); + int n = snprintf(line, sizeof(line), "[m%zu]\npath = /x\n", i); + if (n < 0 || (size_t)n >= sizeof(line) || used + (size_t)n >= len) { + free(body); + EXPECT_TRUE(0 && "module-count test buffer overflow"); + return; + } + memcpy(body + used, line, (size_t)n); + used += (size_t)n; + body[used] = '\0'; } char* path; EXPECT_EQ_INT(write_conf(body, &path), 0); From 2ec17e821c62b5da48cd0e8c7adc442c4f77c0a9 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 11:11:23 +0200 Subject: [PATCH 8/8] fix(daemon): harden bounded per-source registry races Stamp host_last_use before publishing a bucket key and treat an unstamped (last_use == 0) bucket as live, so a just-claimed bucket can no longer be stolen by a concurrent reclaimer. After a successful eviction CAS, re-scan for the interned key and, when an earlier bucket already holds it, zero the duplicate's active count and return the canonical bucket, preventing orphaned per-host counts and cap overshoot under full-table concurrency. Add a message-carrying EXPECT_FAIL primitive and use it for the daemon-conf buffer-overflow guard, and add a fork-based test that records auth failures from forked children and asserts the parent observes the shared lockout. --- src/shared/daemon_limits.c | 39 ++++++++++++++++++++++++++++++++++---- tests/test_daemon_conf.c | 2 +- tests/test_daemon_limits.c | 39 ++++++++++++++++++++++++++++++++++++++ tests/test_utils.h | 8 ++++++++ 4 files changed, 83 insertions(+), 5 deletions(-) diff --git a/src/shared/daemon_limits.c b/src/shared/daemon_limits.c index edb5819..067173e 100644 --- a/src/shared/daemon_limits.c +++ b/src/shared/daemon_limits.c @@ -141,7 +141,11 @@ static bool host_bucket_reclaimable(DaemonLimitRegistry* registry, size_t idx, l if (until != 0) return until <= now; long long last_use = atomic_load_explicit(®istry->host_last_use[idx], memory_order_relaxed); - return last_use == 0 || now - last_use >= DAEMON_LIMITS_HOST_EVICT_IDLE_SEC; + /* A bucket whose key is published but whose last_use has not yet been stamped + * (last_use == 0) must be treated as live: reclaiming it here would steal a + * bucket a racing child just claimed. The claim path also stamps last_use + * before publishing the key, so this window cannot persist. */ + return last_use != 0 && now - last_use >= DAEMON_LIMITS_HOST_EVICT_IDLE_SEC; } /* Emit at most one "per-source table full" warning per @@ -191,14 +195,18 @@ static int host_intern(DaemonLimitRegistry* registry, const char* peer_ip) { return (int)idx; } if (current == 0) { + /* Stamp last_use *before* publishing the key so a reclaimer racing the + * claim can never observe a claimed bucket with last_use == 0 and + * evict it. A pre-stamp is harmless if the CAS loses: the bucket is + * either still empty (never inspected for reclaim) or has just been + * taken by another source that wants a fresh timestamp anyway. */ + atomic_store_explicit(®istry->host_last_use[idx], now, memory_order_relaxed); uint64_t expected = 0; if (atomic_compare_exchange_strong_explicit(®istry->host_key[idx], &expected, key, memory_order_acq_rel, memory_order_acquire)) { - atomic_store_explicit(®istry->host_last_use[idx], now, memory_order_relaxed); return (int)idx; } if (atomic_load_explicit(®istry->host_key[idx], memory_order_acquire) == key) { - atomic_store_explicit(®istry->host_last_use[idx], now, memory_order_relaxed); return (int)idx; } continue; /* another child won this empty bucket; keep probing */ @@ -209,6 +217,9 @@ static int host_intern(DaemonLimitRegistry* registry, const char* peer_ip) { } } if (evict >= 0) { + /* Refresh the timestamp before the key changes hands so the reused bucket + * is not seen as immediately idle by a racing reclaimer. */ + atomic_store_explicit(®istry->host_last_use[evict], now, memory_order_relaxed); uint64_t expected = evict_key; if (atomic_compare_exchange_strong_explicit(®istry->host_key[evict], &expected, key, memory_order_acq_rel, memory_order_acquire)) { @@ -217,7 +228,27 @@ static int host_intern(DaemonLimitRegistry* registry, const char* peer_ip) { atomic_store_explicit(®istry->host_active[evict], 0, memory_order_relaxed); atomic_store_explicit(®istry->host_fail[evict], 0, memory_order_relaxed); atomic_store_explicit(®istry->host_until[evict], 0, memory_order_relaxed); - atomic_store_explicit(®istry->host_last_use[evict], now, memory_order_relaxed); + /* Two children can race to intern the same brand-new key into different + * eviction targets, leaving the table with duplicate buckets for `key`. + * Re-scan for the first (canonical) bucket holding `key`; when it + * precedes `evict`, drop our duplicate's occupancy and hand back the + * canonical bucket so per-source counts are not orphaned on the + * duplicate. The duplicate keeps its key, so no tombstone hole is + * created and probe chains stay intact; it ages out normally. */ + for (size_t i = 0; i < (size_t)registry->host_slots; i++) { + size_t candidate = (start + i) & mask; + uint64_t found = + atomic_load_explicit(®istry->host_key[candidate], memory_order_acquire); + if (found == key) { + if (candidate != (size_t)evict) { + atomic_store_explicit(®istry->host_active[evict], 0, memory_order_relaxed); + return (int)candidate; + } + break; + } + if (found == 0) + break; /* the key is present at `evict`, so this cannot happen first */ + } return evict; } continue; /* lost the race; re-probe with fresh observations */ diff --git a/tests/test_daemon_conf.c b/tests/test_daemon_conf.c index b6f54bc..3cddf0f 100644 --- a/tests/test_daemon_conf.c +++ b/tests/test_daemon_conf.c @@ -550,7 +550,7 @@ static void test_daemon_conf_module_count_capped() { int n = snprintf(line, sizeof(line), "[m%zu]\npath = /x\n", i); if (n < 0 || (size_t)n >= sizeof(line) || used + (size_t)n >= len) { free(body); - EXPECT_TRUE(0 && "module-count test buffer overflow"); + EXPECT_FAIL("module-count test buffer overflow"); return; } memcpy(body + used, line, (size_t)n); diff --git a/tests/test_daemon_limits.c b/tests/test_daemon_limits.c index 8f36bb0..2b04b7b 100644 --- a/tests/test_daemon_limits.c +++ b/tests/test_daemon_limits.c @@ -189,6 +189,44 @@ static void test_daemon_limits_fork_shared() { daemon_limits_destroy(registry); } +/* Cross-process auth lockout: failures recorded by forked children against the + * shared mmap must lock the source out for the parent. This is the + * cross-process path the integration test can no longer cover because trusted + * loopback peers are exempt from the per-host limits. */ +static void test_daemon_limits_fork_auth_lockout() { + if (is_running_under_valgrind()) + return; /* fork + shared mapping is slow/noisy under valgrind */ + DaemonLimitRegistry* registry = daemon_limits_create(DAEMON_LIMITS_MIN_SLOTS, 1, 0, 2, 300); + EXPECT_NOT_NULL(registry); + + int remaining = 0; + EXPECT_FALSE(daemon_limits_auth_locked(registry, "10.0.0.1", &remaining)); + + /* One failure from each of two children reaches the threshold of 2 in the + * shared mapping; atomics only, no mtx/malloc, so fork-safe. */ + for (int i = 0; i < 2; i++) { + pid_t pid = fork(); + if (pid == 0) { + daemon_limits_auth_record_failure(registry, "10.0.0.1"); + _exit(0); + } + EXPECT_TRUE(pid > 0); + int status = 0; + EXPECT_TRUE(waitpid(pid, &status, 0) == pid); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } + + /* The parent observes the lockout the children established. */ + EXPECT_TRUE(daemon_limits_auth_locked(registry, "10.0.0.1", &remaining)); + EXPECT_TRUE(remaining > 0 && remaining <= 300); + /* A different source is unaffected across processes. */ + EXPECT_FALSE(daemon_limits_auth_locked(registry, "10.0.0.2", &remaining)); + /* The parent clears the shared lockout on a successful authentication. */ + daemon_limits_auth_record_success(registry, "10.0.0.1"); + EXPECT_FALSE(daemon_limits_auth_locked(registry, "10.0.0.1", &remaining)); + daemon_limits_destroy(registry); +} + /* The occupancy arrays are derived from the slot table: recompute rebuilds them * and is the self-heal path the SIGCHLD handler uses after a child dies. */ static void test_daemon_limits_recompute() { @@ -269,4 +307,5 @@ void test_daemon_limits() { test_daemon_limits_auth_lockout(); test_daemon_limits_host_table_eviction(); test_daemon_limits_fork_shared(); + test_daemon_limits_fork_auth_lockout(); } diff --git a/tests/test_utils.h b/tests/test_utils.h index 67b6bca..4d5f3d3 100644 --- a/tests/test_utils.h +++ b/tests/test_utils.h @@ -112,4 +112,12 @@ extern bool current_test_failed; } \ } while (0) +/* Unconditional test failure carrying an explanatory message. */ +#define EXPECT_FAIL(message) \ + do { \ + printf(" \033[1;31m[FAIL]\033[0m %s:%d: %s\n", __FILE__, __LINE__, (message)); \ + current_test_failed = true; \ + return; \ + } while (0) + #endif