From 25909110ac5ded4083a9380f0c1739956913ea59 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 10:50:58 +0200 Subject: [PATCH] fix(daemon): exempt trusted loopback peers from per-host limits Every client on loopback shares the 127.0.0.1 identity, so counting them against 'max connections per host' or the default-on auth lockout lets one local client deny service to all the others (and makes a shared-NAT/proxy address a natural DoS vector for remote clients). Use utils_fd_peer_is_local (fail-closed) in the daemon gate to exempt a provably local peer from the per-source cap and the auth lockout while keeping the per-module and global caps. Remote peers are unchanged. Document the shared-NAT/proxy identity limitation and the loopback exemption in README/RSYNC_COMPAT/CHANGELOG, update the integration test to assert the exemption, and fix the README 'auth failure delay' cap (5000, not 60000). --- CHANGELOG.md | 9 ++++++++- README.md | 22 +++++++++++++++++++++- RSYNC_COMPAT.md | 2 +- src/server/server.c | 30 ++++++++++++++++++++++++------ tests/integration/test_daemon.py | 27 +++++++++++++++------------ 5 files changed, 69 insertions(+), 21 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a641f4b..fe26d5a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,7 +14,14 @@ run the same version because the handshake is strict. forks one child per connection, the counters live in an anonymous shared mapping created before the accept loop and reclaimed by the parent's `SIGCHLD` handler, so the per-module, per-source and auth-failure state is - shared across every child (including after `SIGKILL`). + shared across every child (including after `SIGKILL`). The per-source table + now has a bounded lifetime (expired-lockout/idle entries are reclaimed, with a + rate-limited warning when it is genuinely full), and the occupancy counters are + re-derived from the shared slot table on every child exit so a child killed + mid-registration cannot leak a count. Trusted loopback peers are exempt from the + per-host cap and the auth lockout (they share one address); clients behind a + shared NAT/proxy still share a single per-host budget and lockout, which is + documented. ## [2.20.0] - 2026-09-13 diff --git a/README.md b/README.md index 58d0d10..4e7733c 100644 --- a/README.md +++ b/README.md @@ -512,7 +512,7 @@ and `address`, the global section accepts: source IP, default 0 (unlimited). Enforced across all forked connection children through a shared registry. - `auth failure delay = MS` — milliseconds to sleep after a failed - authentication, default 500. `0` disables it and the value is capped at 60000, + authentication, default 500. `0` disables it and the value is capped at 5000, so online password guessing is rate-limited per connection. Successful auths are never delayed. - `auth lockout threshold = N` — number of failed authentications from one source @@ -528,6 +528,26 @@ and `address`, the global section accepts: A `[module]` may also set `max connections` (0 = unlimited; enforced per module across all connection children) and its own `hosts allow`/`hosts deny`. +The per-host cap and the shared auth lockout identify a source by its numeric +peer IP. **Loopback peers (127.0.0.0/8, IPv6 `::1`) are exempt**: every local +client shares that one address, so counting or locking them out would let one +local process deny service to all the others. The per-module and global +`max connections` caps still apply to loopback. Because the key is the peer IP, +`max connections per host` and `auth lockout` also cannot distinguish clients +behind the same NAT, proxy, or reverse-proxy address — they share one budget and +one lockout counter, so an over-aggressive lockout can affect unrelated users +behind that address. Prefer TLS client certificates (`--client-cn`) plus +`hosts allow`/`hosts deny` for per-client policy when clients share an address, +and size `auth lockout threshold` accordingly. + +The shared per-source table has a bounded lifetime: an entry with no live +connection is reclaimed once its lockout has expired, or after it has been idle +(300 s). If every entry is still live or locked, a new source is admitted without +per-host accounting (fail open) and a rate-limited warning is logged; the +per-module cap and host ACLs still apply. The occupancy counters are re-derived +from the shared slot table after every child exit, so a child killed mid-transfer +(or mid-registration) cannot leak a slot or an occupancy count. + Host patterns are `*` (match all), IPv4/IPv6 literals, or IPv4/IPv6 CIDR (`10.0.0.0/8`, `2001:db8::/32`). Hostnames are not resolved, so hostname globs are rejected at parse time rather than silently never matching. A matching diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index f0d4ba2..df03870 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -637,7 +637,7 @@ now transmits targets (the prior behavior was broken/partial); its status moved - **Config grammar** (`fastsyncd.conf`): line-based; an implicit global section first, then `[module]` sections. Keys are case-insensitive, values are trimmed and may be wrapped in one layer of double quotes (`path = "/srv/my dir"`). `#` and `;` at the start of a line (after leading whitespace) are full-line comments; inline comments and `\` continuations are not supported. Lines are bounded (4096 chars), and at most 256 `[module]` sections are accepted. Global keys: `port` (default 873), `motd file` (the daemon sends its bounded, escaped content to a client after the module gate/auth accepts, unless the client passes `--no-motd`), `address` (optional bind address), `max connections` (positive integer cap on concurrent connections, default 100; 0/negative/garbage is a parse error), `max connections per host` (concurrent-connection cap per source IP, default 0 = unlimited), `auth failure delay` (milliseconds to sleep after a failed authentication, default 500; 0 disables, capped at 5000), `auth lockout threshold` (failed authentications from one source before lockout, default 10; 0 disables), `auth lockout duration` (seconds a locked-out source is refused, default 300), `hosts allow` and `hosts deny` (comma- and/or whitespace-separated host access patterns — see the host access control note below). Module keys: `path` (required; the daemon-side authorized root for that module), `read only` (yes/no/true/false/1/0, default no), `client owner` (yes/no/true/false/1/0, default no; opts the module into client-chosen ownership — see below), `auth users` (comma list), `max connections` (optional per-module cap, 0 = unlimited; enforced across all connection children), `hosts allow`/`hosts deny` (per-module host access lists). **Unknown keys and malformed lines are parse-and-reject errors** (never silently ignored), so a typo cannot change what a module serves. - **Host access control (`hosts allow`/`hosts deny`):** both keys accept a comma- and/or whitespace-separated list of patterns and may appear globally and/or per module (multiple config-file lines append; a `--dparam` override replaces). Supported patterns are `*` (match all), an IPv4 or IPv6 literal (`10.0.0.1`, `2001:db8::1`), and an IPv4/IPv6 CIDR (`10.0.0.0/8`, `2001:db8::/32`). Hostname patterns are **not** supported: because the peer is always a numeric address and no reverse DNS is performed, a hostname/glob pattern would silently never match, so it is rejected at load time (fail-closed) instead of being accepted as a dead rule. An IPv4 peer on a dual-stack IPv6 listener is normalized from its `::ffff:a.b.c.d` form so IPv4 patterns match it. rsync-like semantics: a matching `hosts deny` rejects; if any `hosts allow` entries exist, a peer matching none of them is rejected; deny takes precedence over allow. The daemon enforces the global list first, then the selected module's list, **before authentication** in `server_module_gate`, with an audit log line naming the peer, the module and the outcome. The numeric peer address is obtained with `getpeername`+`inet_ntop` (`utils_fd_peer_ip`, handling both address families); when it cannot be obtained a module with any ACL fails closed (refused), while an ACL-free module continues and logs at debug. A malformed pattern (e.g. an out-of-range CIDR prefix) is a parse error at load time. -- **Connection caps, shared registry and auth lockout:** the global `max connections` key (default 100) is plumbed into the listener (`transport_tcp.c`), which rejects a connection once the accept-loop parent's active-child count reaches it; the IPv4/IPv6 peer is logged for every accepted connection. Because the listener forks one child per connection, the per-module `max connections` cap, the global `max connections per host` cap, and the auth-failure counter live in a fixed-size registry carved from an anonymous shared mapping (`daemon_limits.c`, `mmap(MAP_SHARED|MAP_ANONYMOUS)`) created by the parent before the accept loop, so every forked child shares the same counters (C11 atomics only — never a pthread lock, which can deadlock in a forked child). The parent reserves a registry slot per accepted connection and the child records the selected module and source IP once known; the parent's `SIGCHLD` handler reclaims the slot when the child dies (including `SIGKILL`), decrementing the per-module and per-source counts. The per-module cap (0 = unlimited) is enforced after the module lookup and before auth; per-source identity reuses the normalized numeric peer address (`utils_fd_peer_ip`, IPv4-mapped IPv6 collapsed to IPv4). A failed authentication increments the shared per-source failure count and, once `auth lockout threshold` (default 10; 0 disables) is reached, the source is refused for `auth lockout duration` seconds (default 300) before any challenge is sent, even when the next attempt is handled by a different forked child; a successful authentication clears the counter. On a failed authentication the per-connection child still sleeps the global `auth failure delay` (default 500 ms, 0 disables, capped at 5000) via `nanosleep`, rate-limiting online guessing without delaying a success. A missing registry (allocation failure) degrades to the global cap and host ACLs rather than refusing to start. +- **Connection caps, shared registry and auth lockout:** the global `max connections` key (default 100) is plumbed into the listener (`transport_tcp.c`), which rejects a connection once the accept-loop parent's active-child count reaches it; the IPv4/IPv6 peer is logged for every accepted connection. Because the listener forks one child per connection, the per-module `max connections` cap, the global `max connections per host` cap, and the auth-failure counter live in a fixed-size registry carved from an anonymous shared mapping (`daemon_limits.c`, `mmap(MAP_SHARED|MAP_ANONYMOUS)`) created by the parent before the accept loop, so every forked child shares the same counters (C11 atomics only — never a pthread lock, which can deadlock in a forked child). The parent reserves a registry slot per accepted connection and the child records the selected module and source IP once known; the parent's `SIGCHLD` handler reclaims the slot when the child dies (including `SIGKILL`) and re-derives the per-module and per-source occupancy counts from the surviving REGISTERED slots, so a child killed mid-registration cannot leak a count. The per-source table has a bounded lifetime: an entry with no live connection is reclaimed after its lockout expires or it has been idle (300 s); if the table is genuinely full the per-source cap/lockout fails open for new sources (per-module cap and ACLs still apply) with a rate-limited warning. The per-module cap (0 = unlimited) is enforced after the module lookup and before auth; per-source identity reuses the normalized numeric peer address (`utils_fd_peer_ip`, IPv4-mapped IPv6 collapsed to IPv4), and a trusted loopback peer (127.0.0.0/8 / `::1`, `utils_fd_peer_is_local`) is exempt from the per-source cap and the auth lockout because all local clients share one address (the per-module/global caps still apply). Clients behind a shared NAT/proxy address likewise share one per-source budget and lockout counter. A failed authentication increments the shared per-source failure count and, once `auth lockout threshold` (default 10; 0 disables) is reached, the source is refused for `auth lockout duration` seconds (default 300) before any challenge is sent, even when the next attempt is handled by a different forked child; a successful authentication clears the counter. On a failed authentication the per-connection child still sleeps the global `auth failure delay` (default 500 ms, 0 disables, capped at 5000) via `nanosleep`, rate-limiting online guessing without delaying a success. A missing registry (allocation failure) degrades to the global cap and host ACLs rather than refusing to start. - **Module selection & confinement:** the client requests a module with an rsync-style `host::module[/path]` destination. The module name crosses the wire as a trailing string on the config frame (bumping `PROTOCOL_VERSION` 2.14.0 → 2.15.0; the bump is required because the config-frame layout changed and the strict same-version handshake is what prevents a peer from desynchronizing on the new trailing field). The daemon looks the module up in ITS OWN config and uses the module's `path` as the authorized root through the exact same `configure_authorization` confinement the standalone server applies to `--destination-root` (`file_open_secure_parent`, `has_path_traversal`, `path_is_within`); the client never supplies the root, every client-chosen-ownership/super-user request is refused unless the module declares `client owner = yes` (the daemon's per-module opt-in, see below), and the operator `--no-super` veto forces super-user activities off for every daemon connection. The client's `/path` part is relative inside the module and is rejected if absolute or if it contains `..`. Unknown modules are refused before any data moves (the run fails cleanly at the config handshake). An absolute destination and a module request against a non-daemon server are also refused. - **`client owner` (client-chosen-ownership opt-in):** by default a daemon module refuses every request that would let the client pick an owner or ask for super-user activities — `--numeric-ids`, `--chown`, `--usermap`/`--groupmap`, `--fake-super`, `--copy-as`, and an explicit `--super` — at the config handshake (before `STATUS_OK`), because a daemon has no per-module opt-in for client-chosen ownership and any anonymous client could otherwise force arbitrary owner ids inside the module root. `client owner = yes` opts a single module in, allowing those requests within that module's root (the standalone listener and the SSH `--stdio` server always honor them for their single operator-authorized root). Without the opt-in the daemon also forces super-user **device** activity off for that connection — char/block device-node creation (`--devices`) and `--write-devices` — even under the default `AUTO` mode, so a non-opted module can never be made to `mknod` or write a raw device; those entries are skipped (not refused) so an ordinary `-a` push still succeeds without device nodes. The opt-in does **not** lift the privilege requirement: `--copy-as` still needs a root receiver, and the operator `--no-super` veto still forces super-user activities off for every connection. The daemon logs a prominent startup warning for each `client owner = yes` module so the operator's deliberate choice is visible. - **`read only` safe default:** every network transfer FastSync currently supports is a push that writes under the module root, so a `read only` module refuses the connection (clear server log "module is read only"; the client exits non-zero, nothing is transferred). A future pull/list operation can be opened up when it exists; the knob is already stored. diff --git a/src/server/server.c b/src/server/server.c index 6f3862c..6785b12 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -82,6 +82,13 @@ typedef struct ModuleGateContext { * not classify the peer; an ACL-configured module then fails closed. */ bool has_peer_ip; char peer_ip[INET6_ADDRSTRLEN]; + /* True when the peer is provably loopback (utils_fd_peer_is_local, fail + * closed). A trusted local/SSH peer is exempt from the per-host cap and the + * cross-process auth lockout: every loopback client shares the 127.0.0.1 + * identity, so counting/locking them out would let one local client deny + * service to (or leak lockout state about) all the others. The per-module and + * global caps still apply. */ + bool is_local; } ModuleGateContext; /* Server half of the SCRAM challenge/response (A7 remediation, protocol @@ -326,7 +333,11 @@ static const char* module_gate_check_limits(const Config* config, const DaemonMo int module_index = daemon_module_index(module); if (module_index < 0) return NULL; - const char* peer = (gate_ctx && gate_ctx->has_peer_ip) ? gate_ctx->peer_ip : ""; + /* A trusted loopback peer is exempt from the per-source cap: pass an + * unparseable peer so the registry skips per-source tracking, while the + * per-module cap below is still enforced. Remote peers are tracked normally. */ + const char* peer = + (!gate_ctx || gate_ctx->is_local || !gate_ctx->has_peer_ip) ? "" : gate_ctx->peer_ip; DaemonLimitResult result = daemon_limits_register(g_daemon_limits, slot, module_index, peer, module->max_connections); switch (result) { @@ -455,8 +466,10 @@ static ModuleAuthResult module_gate_authenticate(const Config* config, const Dae return MODULE_AUTH_ACCEPTED; /* Cross-process lockout: a source that failed too many authentications is * refused before the challenge is sent (the counter lives in the shared - * registry, so it spans every forked child and survives a child exit). */ - if (g_daemon_limits && gate_ctx && gate_ctx->has_peer_ip) { + * registry, so it spans every forked child and survives a child exit). A + * trusted loopback peer is exempt: all local clients share the 127.0.0.1 + * identity, so a lockout would let one deny the others. */ + if (g_daemon_limits && gate_ctx && gate_ctx->has_peer_ip && !gate_ctx->is_local) { int remaining = 0; if (daemon_limits_auth_locked(g_daemon_limits, gate_ctx->peer_ip, &remaining)) { log_message(LOG_LEVEL_ERROR, @@ -523,13 +536,14 @@ static ModuleAuthResult module_gate_authenticate(const Config* config, const Dae free(escaped_user); /* Count the failure in the shared registry (locks the source out once the * configured threshold is reached) and rate-limit online guessing per - * connection (no delay on success). */ - if (g_daemon_limits && gate_ctx->has_peer_ip) + * connection (no delay on success). A loopback peer is exempt from the + * shared counter. */ + if (g_daemon_limits && gate_ctx->has_peer_ip && !gate_ctx->is_local) daemon_limits_auth_record_failure(g_daemon_limits, gate_ctx->peer_ip); daemon_auth_failure_delay(); return MODULE_AUTH_TERMINATED; } - if (g_daemon_limits && gate_ctx->has_peer_ip) + if (g_daemon_limits && gate_ctx->has_peer_ip && !gate_ctx->is_local) daemon_limits_auth_record_success(g_daemon_limits, gate_ctx->peer_ip); char* escaped_user = output_escape(config->auth_user, config->eight_bit_output); log_message(LOG_LEVEL_INFO, "daemon module '%s': user '%s' from %s authenticated", config->module, @@ -636,6 +650,9 @@ static const char* server_module_gate(const Config* config, void* context) { utils_fd_peer_ip(gate_ctx->fd, gate_ctx->peer_ip, sizeof(gate_ctx->peer_ip)); if (!gate_ctx->has_peer_ip) log_message(LOG_LEVEL_DEBUG, "daemon module '%s': peer address unavailable", config->module); + /* utils_fd_peer_is_local is fail-closed (getpeername must succeed and report + * a loopback peer), so "cannot tell" is never treated as trusted. */ + gate_ctx->is_local = utils_fd_peer_is_local(gate_ctx->fd); } error = module_gate_check_hosts(config, module, gate_ctx); if (error) @@ -670,6 +687,7 @@ void handler(int file_descriptor) { gate_ctx.super_mode_override = -1; gate_ctx.has_peer_ip = false; gate_ctx.peer_ip[0] = '\0'; + gate_ctx.is_local = false; /* All teardown state starts empty so the single `done` epilogue is safe to * reach from any error path (including before the config frame arrives). */ Config* config = NULL; diff --git a/tests/integration/test_daemon.py b/tests/integration/test_daemon.py index 1ff1d9d..a7870e9 100644 --- a/tests/integration/test_daemon.py +++ b/tests/integration/test_daemon.py @@ -1221,11 +1221,14 @@ class TestDaemonConnectionLimits: CAPS_CONF = os.path.join(TEST_DATA_DIR, "fastsyncd_caps.conf") @pytest.mark.ci - def test_auth_lockout_is_shared_across_children(self): - """`auth lockout threshold = 1`: the first failed authentication locks the - source out for the cooldown in the SHARED registry, so a subsequent - correct-password attempt (a different forked child) is refused before a - SCRAM challenge is even sent.""" + def test_auth_lockout_exempts_trusted_loopback(self): + """`auth lockout threshold = 1`: a trusted loopback peer is EXEMPT from + the shared lockout because every local client shares the 127.0.0.1 + identity, so a single wrong password must not lock out correct-password + attempts (that would be a local denial of service). The shared + per-source lockout machinery itself is covered by the daemon_limits unit + tests; this locks in the loopback policy and the absence of a stale + "locked out" log line.""" port = _find_free_port() with open(self.LOCKOUT_CONF, "w") as f: f.write("port = %d\n" @@ -1241,21 +1244,21 @@ class TestDaemonConnectionLimits: try: d.start(self.LOCKOUT_CONF, port_override=port, extra_args=["--password-file", CRED_FILE], log_path=log_path) - before = _tree_file_count(AUTH_MODULE) log_before = os.path.getsize(log_path) if os.path.exists(log_path) else 0 - # First attempt: wrong password -> records failure #1 -> locks. + # First attempt: wrong password -> a failure is logged, but a loopback + # peer is not counted toward the lockout. wrong = _push_with_creds("127.0.0.1::locked", port, "alice", WRONG_PASS) assert wrong.returncode != 0 - # Second attempt: CORRECT password from the same source must still be - # refused by the shared lockout. + # Second attempt: the correct password from the same local source must + # still be accepted (no lockout), which also runs the SCRAM handshake + # to completion in a fresh forked child. right = _push_with_creds("127.0.0.1::locked", port, "alice", ALICE_PASS) - assert right.returncode != 0, "the shared auth lockout must refuse after threshold" - assert _tree_file_count(AUTH_MODULE) == before, "a locked-out source wrote data" + assert right.returncode == 0, (right.stderr or right.stdout) time.sleep(0.3) with open(log_path, "rb") as f: f.seek(log_before) tail = f.read().decode("utf-8", "replace") - assert "locked out" in tail, tail[-400:] + assert "locked out" not in tail, tail[-400:] finally: d.stop()