fix(daemon): bound per-source table lifetime and recompute occupancy
The per-source host table only grew: once its fixed open-addressed table filled, host_intern returned -1 and the per-host cap plus the shared auth lockout silently failed open forever. Add a bounded-lifetime eviction policy: track a per-bucket last-use time and, when no empty bucket exists, atomically repurpose the first bucket that has no active connection and either has an expired lockout or has been idle, resetting its counters. Warn (rate-limited) on the genuine fail-open path. A child SIGKILLed mid-registration could also leak a module/host count because the parent only decremented on a REGISTERED slot. Make the slot table the source of truth: after the SIGCHLD reap the parent recomputes module_active[]/host_active[] from the surviving REGISTERED slots (atomics only, async-signal-safe) so any leaked increment is erased. Also clamp module_count to DAEMON_LIMITS_MAX_MODULES and use one helper for the sizing/register host-tracking condition (a lockout threshold with duration 0 is a no-op and must not intern hosts).
This commit is contained in:
@@ -9,6 +9,7 @@
|
||||
#include <netinet/in.h>
|
||||
#include <netinet/tcp.h>
|
||||
#include <openssl/ssl.h>
|
||||
#include <pthread.h>
|
||||
#include <signal.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
@@ -39,9 +40,25 @@ static void sigchld_handler(int sig) {
|
||||
g_active_connections--;
|
||||
daemon_limits_reclaim_pid(g_limit_registry, (long)pid);
|
||||
}
|
||||
/* Re-derive the occupancy counters once for the whole reap batch. The slot
|
||||
* table is the source of truth, so this self-heals any count leaked by a child
|
||||
* SIGKILLed mid-registration. Atomics only: async-signal-safe. */
|
||||
if (g_limit_registry)
|
||||
daemon_limits_recompute(g_limit_registry);
|
||||
errno = saved_errno;
|
||||
}
|
||||
|
||||
/* Reset a signal to its default action with sigaction (preferred over
|
||||
* signal(3), whose semantics are implementation-defined). Used in the forked
|
||||
* child before it can spawn any thread. */
|
||||
static void reset_signal_default(int sig) {
|
||||
struct sigaction action;
|
||||
memset(&action, 0, sizeof(action));
|
||||
action.sa_handler = SIG_DFL;
|
||||
sigemptyset(&action.sa_mask);
|
||||
sigaction(sig, &action, NULL);
|
||||
}
|
||||
|
||||
/* Map a listen socket's address to its numeric port for logging, independent
|
||||
* of whether it is an IPv4 or IPv6 sockaddr. */
|
||||
static unsigned short server_address_port(const struct sockaddr_storage* addr) {
|
||||
@@ -160,7 +177,16 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil
|
||||
log_perror("Could not listen on port!");
|
||||
return;
|
||||
}
|
||||
signal(SIGCHLD, sigchld_handler);
|
||||
/* SIGCHLD via sigaction (not signal(3)); SA_RESTART keeps accept(2) from
|
||||
* failing with EINTR, and SA_NOCLDSTOP only notifies on child exit. The
|
||||
* accept loop is single-threaded at this point, so installing here cannot race
|
||||
* a worker thread. */
|
||||
struct sigaction chld_action;
|
||||
memset(&chld_action, 0, sizeof(chld_action));
|
||||
chld_action.sa_handler = sigchld_handler;
|
||||
sigemptyset(&chld_action.sa_mask);
|
||||
chld_action.sa_flags = SA_RESTART | SA_NOCLDSTOP;
|
||||
sigaction(SIGCHLD, &chld_action, NULL);
|
||||
g_limit_registry = server->limit_registry;
|
||||
while (1) {
|
||||
struct sockaddr_storage client_addr;
|
||||
@@ -197,15 +223,21 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil
|
||||
g_current_slot = slot;
|
||||
/* Block SIGCHLD across fork() and the parent's pid publication: a child
|
||||
* that exits immediately must not be reaped before its slot records its
|
||||
* pid, which would leak the slot and its module/source counts. */
|
||||
* pid, which would leak the slot and its module/source counts. Use
|
||||
* pthread_sigmask rather than sigprocmask so the behavior is well defined
|
||||
* even if this process ever gains threads: the mask is per-thread, the fork
|
||||
* copies only the calling thread, and the child inherits this thread's
|
||||
* blocked mask until it restores `previous` below. No thread exists yet at
|
||||
* this point, and none is created before the mask is restored, so the
|
||||
* critical window is race-free. */
|
||||
sigset_t blocked;
|
||||
sigset_t previous;
|
||||
sigemptyset(&blocked);
|
||||
sigaddset(&blocked, SIGCHLD);
|
||||
sigprocmask(SIG_BLOCK, &blocked, &previous);
|
||||
pthread_sigmask(SIG_BLOCK, &blocked, &previous);
|
||||
pid_t pid = fork();
|
||||
if (pid == 0) {
|
||||
sigprocmask(SIG_SETMASK, &previous, NULL);
|
||||
pthread_sigmask(SIG_SETMASK, &previous, NULL);
|
||||
/* Connection children must not run the parent's global cleanup(): it
|
||||
* frees state (credentials / daemon conf) that the child's worker
|
||||
* threads may still be reading and closes fd numbers the child could
|
||||
@@ -213,9 +245,9 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil
|
||||
* terminates the child directly; SIGCHLD is reset too since a child
|
||||
* must never reap the parent's children. This runs before the child
|
||||
* spawns any thread, so it cannot race one. */
|
||||
signal(SIGINT, SIG_DFL);
|
||||
signal(SIGTERM, SIG_DFL);
|
||||
signal(SIGCHLD, SIG_DFL);
|
||||
reset_signal_default(SIGINT);
|
||||
reset_signal_default(SIGTERM);
|
||||
reset_signal_default(SIGCHLD);
|
||||
close(server->file_descriptor);
|
||||
child_fn(fd, child_ctx);
|
||||
_exit(0);
|
||||
@@ -227,7 +259,7 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil
|
||||
/* fork() failed: release the reservation so the slot is not leaked. */
|
||||
daemon_limits_reclaim_slot(server->limit_registry, slot);
|
||||
}
|
||||
sigprocmask(SIG_SETMASK, &previous, NULL);
|
||||
pthread_sigmask(SIG_SETMASK, &previous, NULL);
|
||||
close(fd);
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user