Collect each directory's entries up front and process extraneous subdirectories first (descending name, depth-first), then extraneous files (descending name), then descend into kept subdirectories in ascending order. Emit a trailing slash for deleted directories in observers/dry-run output. This matches rsync's delete order for --delete-before/--delete-after/--delete-delay and for dry-run listings.
276 lines
16 KiB
C
276 lines
16 KiB
C
#ifndef UTILS_H
|
|
#define UTILS_H
|
|
|
|
#include "array_list.h"
|
|
#include "filter.h"
|
|
#include <stddef.h>
|
|
#include <stdbool.h>
|
|
#include <stdio.h>
|
|
#include <sys/socket.h>
|
|
#include <sys/types.h>
|
|
|
|
/* Small open-addressing string hash set used to turn quadratic membership
|
|
* scans into O(path length) exact-match lookups (the --delete keep-set and the
|
|
* --files-from allow-set). Keys are hashed with xxHash64 (seed 0); collisions
|
|
* are resolved by linear probing over a power-of-two table that grows at 75%
|
|
* load. Keys are always borrowed from the caller and must outlive the set; the
|
|
* set never copies or owns keys, so indexing M entries costs O(M) memory. The
|
|
* set is not thread-safe for mutation, but a fully built set supports
|
|
* concurrent read-only lookups. */
|
|
typedef struct {
|
|
const char* key; /* NULL marks an empty slot */
|
|
} StrHashSetSlot;
|
|
|
|
typedef struct {
|
|
StrHashSetSlot* slots;
|
|
size_t capacity; /* power of two, zero before init */
|
|
size_t size;
|
|
} StrHashSet;
|
|
|
|
/* Initialize an empty set sized for roughly `hint` entries. Returns false on
|
|
* allocation failure. */
|
|
bool str_hash_set_init(StrHashSet* set, size_t hint);
|
|
void str_hash_set_free(StrHashSet* set);
|
|
/* Insert a borrowed key (must outlive the set). A duplicate is ignored.
|
|
* Returns false on allocation failure. */
|
|
bool str_hash_set_insert_ref(StrHashSet* set, const char* key);
|
|
/* Look up a NUL-terminated key / a key of `len` bytes. */
|
|
bool str_hash_set_lookup(const StrHashSet* set, const char* key);
|
|
bool str_hash_set_lookup_n(const StrHashSet* set, const char* key, size_t len);
|
|
|
|
/* Sorted, non-owning view of NUL-terminated strings. Built from borrowed
|
|
* pointers (qsort), so indexing M entries costs O(M) memory and O(M log M)
|
|
* time; exact membership and ancestor-prefix existence are binary searches
|
|
* that never materialize a prefix copy. */
|
|
typedef struct {
|
|
const char** items; /* sorted with strcmp; borrowed, never freed */
|
|
size_t count;
|
|
} StrSortedArray;
|
|
|
|
/* Build `array` over the borrowed `items`. Only the pointer array is copied,
|
|
* never the strings. Returns false on allocation failure. */
|
|
bool str_sorted_array_build(StrSortedArray* array, const char* const* items, size_t count);
|
|
void str_sorted_array_free(StrSortedArray* array);
|
|
/* True when some item equals `key`. */
|
|
bool str_sorted_array_contains(const StrSortedArray* array, const char* key);
|
|
/* True when some item starts with `key` followed by '/' (i.e. `key` is a proper
|
|
* ancestor directory of an item). Allocates nothing. */
|
|
bool str_sorted_array_has_child_prefix(const StrSortedArray* array, const char* key);
|
|
|
|
/* Read-only membership index over exact relative paths. `exact` answers
|
|
* O(path length) equality; `sorted` answers whether any indexed path lies
|
|
* strictly below a query directory. Both borrow their keys from the caller and
|
|
* no ancestor prefix is stored as a separate string, so an index over M entries
|
|
* is O(M) memory regardless of path depth. Not thread-safe to build, but safe
|
|
* for concurrent read-only queries once built. */
|
|
typedef struct {
|
|
StrHashSet exact;
|
|
StrSortedArray sorted;
|
|
} PathIndex;
|
|
|
|
/* Build an index borrowing `entries` (which must outlive the index). Returns
|
|
* false on allocation failure, freeing any partial state. */
|
|
bool path_index_build(PathIndex* index, const char* const* entries, size_t count);
|
|
void path_index_free(PathIndex* index);
|
|
/* True when `path` is an indexed entry. */
|
|
bool path_index_contains(const PathIndex* index, const char* path);
|
|
/* Length-bounded form of path_index_contains (`path` need not be terminated). */
|
|
bool path_index_contains_n(const PathIndex* index, const char* path, size_t len);
|
|
/* True when some indexed entry lies strictly below `path` (starts with
|
|
* `path` + '/'). */
|
|
bool path_index_has_descendant(const PathIndex* index, const char* path);
|
|
|
|
char* str_dup(const char* string);
|
|
char* output_escape(const char* string, bool eight_bit_output);
|
|
|
|
/* Resolve the first supported name from a rsync algorithm-preference
|
|
* environment variable (RSYNC_COMPRESS_LIST / RSYNC_CHECKSUM_LIST). `resolve`
|
|
* maps a case-insensitive name to an algorithm id (>= 0) or -1 for an unknown
|
|
* name. rsync's syntax is a whitespace-separated list (comma/colon are NOT
|
|
* separators); the client-side half ends at '&'. Unknown entries are skipped
|
|
* and the first resolvable one wins. *specified is set true when the variable
|
|
* holds at least one non-blank character. Returns the first resolvable id, or
|
|
* -1 when the variable is unset/blank or names no supported algorithm. */
|
|
int env_choice_first(const char* env_name, int (*resolve)(const char*), bool* specified);
|
|
|
|
/* Upper bound on one line/token read from a local list file (--files-from,
|
|
* --exclude-from/--include-from, .rsync-filter). Mirrors MAX_STRING_SIZE and
|
|
* stops a hostile multi-gigabyte line from forcing unbounded allocation. */
|
|
#define UTILS_MAX_LINE_LEN (64 * 1024)
|
|
/* Read one `delim`-terminated record from `stream` into *line (grown as needed
|
|
* and NUL-terminated), refusing to consume/allocate more than `max_len` bytes
|
|
* of content. Returns the number of bytes stored (delimiter included, matching
|
|
* getdelim), 0 at end of file, or -1 on error (errno is EFBIG when the record
|
|
* exceeds `max_len`, ENOMEM on allocation failure). *line and *cap are updated
|
|
* as the buffer grows and the caller owns *line. */
|
|
ssize_t utils_getdelim_bounded(FILE* stream, char** line, size_t* cap, int delim, size_t max_len);
|
|
char* path_cat(const char* path1, const char* path2);
|
|
bool glob_match(const char* pattern, const char* str);
|
|
/* Result of a bounded extra-file deletion run. */
|
|
typedef enum {
|
|
/* Every extra entry was removed (or there were none). */
|
|
DELETE_WALK_OK = 0,
|
|
/* The numeric cap for this run was reached before every extra was removed.
|
|
The walker removed exactly the entries the cap allowed and skipped (without
|
|
removing) the rest, matching rsync's partial --max-delete behavior. */
|
|
DELETE_WALK_LIMIT_REACHED,
|
|
/* A traversal or unlink failure aborted the deletion (partial removal is
|
|
possible, mirroring the delete pass). */
|
|
DELETE_WALK_ERROR
|
|
} DeleteWalkResult;
|
|
/* One protected entry for the delete walker. When top_level_only is true the
|
|
prefix is skipped only as a DIRECT child of dest_root (the --delay-updates
|
|
staging directory, which must not hide genuine extras inside a nested
|
|
destination directory that happens to share the staging name); otherwise the
|
|
prefix is skipped at any depth (the --compare-dest/--copy-dest/--link-dest
|
|
basis trees, and the sender-side protected filter-excluded prefixes, which
|
|
are never destination content). */
|
|
typedef struct {
|
|
const char* prefix;
|
|
bool top_level_only;
|
|
} DeleteSkipEntry;
|
|
/* True when child_rel is, or lies below, one of the protected entries (a prefix
|
|
"a" protects "a" and "a/b/c" but not "ab"; top_level_only entries protect
|
|
only DIRECT children of the destination root, i.e. child_rel has no '/'). */
|
|
bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips,
|
|
int skip_count);
|
|
/* One destination-directory entry collected up front so the delete walkers can
|
|
reproduce rsync's traversal order instead of readdir() order. rsync processes
|
|
a directory's extraneous subdirectories first (descending name, depth-first),
|
|
then its extraneous files (descending name), and only afterwards descends into
|
|
its kept subdirectories (ascending name). */
|
|
typedef struct {
|
|
char* name;
|
|
bool is_dir;
|
|
} DeleteDirEntry;
|
|
/* Collect the entries of the directory open on `dirfd` (excluding "." and ".."),
|
|
stat'ing each with AT_SYMLINK_NOFOLLOW. On success *out is a malloc'd array of
|
|
*count entries whose names the caller frees with delete_dir_entries_free().
|
|
Returns false on an allocation/readdir failure; a vanished entry (ENOENT) is
|
|
skipped, any other stat failure is reported through *operation_ok while the
|
|
walk continues. */
|
|
bool delete_dir_entries_collect(int dirfd, DeleteDirEntry** out, size_t* count, bool* operation_ok);
|
|
void delete_dir_entries_free(DeleteDirEntry* entries, size_t count);
|
|
/* Sort comparators: `_desc` orders subdirectories before files and each group by
|
|
descending name (rsync's extraneous-entry order); `_asc` orders plain ascending
|
|
name (rsync's kept-subdirectory order). */
|
|
int delete_dir_entry_cmp_desc(const void* a, const void* b);
|
|
int delete_dir_entry_cmp_asc(const void* a, const void* b);
|
|
/* Remove files/dirs/symlinks under dest_root that are not listed in manifest
|
|
without ever descending into a protected prefix (see DeleteSkipEntry). When
|
|
`synced_dirs` is non-NULL, extras are only removed directly inside a directory
|
|
whose destination-relative path is an exact entry in that list (the receive
|
|
root is the "." sentinel); directories outside the synchronized set are still
|
|
descended into so kept content below a listed directory is preserved, but
|
|
nothing in them is removed. A NULL `synced_dirs` keeps the legacy behavior of
|
|
treating the whole destination tree as deletable. `max_delete` caps the
|
|
number of removed entries (SIZE_MAX = unlimited): the walker removes up to the
|
|
cap and returns DELETE_WALK_LIMIT_REACHED when more extras remained.
|
|
`deleted_out`/`skipped_out` optionally receive the number of entries removed
|
|
and the number skipped because of the cap. */
|
|
DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest,
|
|
const ArrayList* synced_dirs, size_t max_delete,
|
|
const DeleteSkipEntry* skips, int skip_count,
|
|
const FilterRuleList* protect_rules, size_t* deleted_out,
|
|
size_t* skipped_out);
|
|
|
|
/* Optional per-deletion observer: called for each destination-relative path
|
|
actually removed (a file, symlink, or directory), in removal order, so the
|
|
receiver can stream rsync's `--info=del`/`--info=remove` lines. */
|
|
typedef void (*DeletePathObserver)(void* context, const char* rel_path);
|
|
|
|
/* `delete_extras_limited_observed` is delete_extras_limited with an optional
|
|
* observer; the observer is invoked only for entries truly removed. When
|
|
* `protect_rules` is non-NULL its receiver-side verdict is evaluated for every
|
|
* candidate extra: a first-match PROTECT leaves the entry (and, for a
|
|
* directory, its whole subtree) in place, while RISK/NONE fall through to the
|
|
* ordinary skip-prefix/keep-set logic. */
|
|
DeleteWalkResult delete_extras_limited_observed(const char* dest_root, const ArrayList* manifest,
|
|
const ArrayList* synced_dirs, size_t max_delete,
|
|
const DeleteSkipEntry* skips, int skip_count,
|
|
const FilterRuleList* protect_rules,
|
|
size_t* deleted_out, size_t* skipped_out,
|
|
DeletePathObserver observer,
|
|
void* observer_context);
|
|
/* Read-only companion to delete_extras_limited: walk the destination exactly as
|
|
the delete pass would and APPEND (strdup'd) destination-relative paths that
|
|
WOULD be removed, without touching disk. Used for -n/--dry-run --delete
|
|
would-delete reporting. Returns true on a clean walk; the caller owns the
|
|
strings appended to `out` and receives their count in *count_out. */
|
|
bool delete_extras_list(const char* dest_root, const ArrayList* manifest,
|
|
const ArrayList* synced_dirs, const DeleteSkipEntry* skips, int skip_count,
|
|
const FilterRuleList* protect_rules, ArrayList* out, size_t* count_out);
|
|
bool delete_extras(const char* dest_root, const ArrayList* manifest);
|
|
/* Open the existing destination directory at `dest_root`, confined to the
|
|
authorized root with an O_NOFOLLOW component walk (the same confinement the
|
|
deletion walker uses for its root). Returns a new fd the caller owns, or -1
|
|
on error (including a destination that does not exist). */
|
|
int utils_open_authorized_destination(const char* dest_root);
|
|
bool utils_set_authorized_root(int fd, const char* canonical_path);
|
|
/* The fd-only compatibility form is fail-closed for path-based operations;
|
|
* callers should use utils_set_authorized_root with the canonical identity. */
|
|
void utils_set_authorized_root_fd(int fd);
|
|
/* Read accessors for the process-wide authorized root, so every secure-walk
|
|
* site consumes the single shared state instead of keeping its own copy. The
|
|
* fd is caller-owned (see the setters): it is returned verbatim, never dup'd,
|
|
* and the caller that opened it is responsible for closing it. With no root
|
|
* configured the fd accessor returns -1 and the path accessor returns NULL.
|
|
*
|
|
* The pointer returned by utils_get_authorized_root_path() is borrowed into
|
|
* process-global state and is invalidated by the next
|
|
* utils_set_authorized_root() / utils_set_authorized_root_fd() call. The fd
|
|
* and path are stored separately and read independently, so the pair is NOT
|
|
* observed atomically together; the accessors are non-reentrant and callers
|
|
* must serialize configuration (the server installs the root before any worker
|
|
* threads spawn; see utils.c). */
|
|
int utils_get_authorized_root_fd(void);
|
|
const char* utils_get_authorized_root_path(void);
|
|
/* True when `path` is `root` itself or lies directly beneath it: a lexical
|
|
* prefix test requiring the byte after `root` to be '\0' or '/'. Both `root`
|
|
* and `path` must be absolute canonical paths free of "."/".." components (the
|
|
* callers guarantee this); this is containment by string, not by resolved
|
|
* symlinks. Shared by the utils and file secure-walk root confinement. */
|
|
bool path_is_within_root(const char* root, const char* path);
|
|
/* Non-allocating transfer-relative view of `path`: strip any leading '/' and
|
|
* then a `root` prefix (leading/trailing slashes tolerated), returning a
|
|
* borrowed pointer into `path`. A NULL/empty root, or a path not under
|
|
* `root`, yields just the leading-slash strip. `path`/`root` must stay alive. */
|
|
const char* utils_strip_transfer_root(const char* path, const char* root);
|
|
/* True when `path` contains a ".." component. This is a purely lexical
|
|
* dot-dot check: an absolute path is NOT rejected here, because default
|
|
* (non-relative) transfers legitimately put the sender's absolute source path
|
|
* on the wire and the receiver re-roots it under the destination with
|
|
* path_cat(). Callers that accept a strictly relative path (e.g. batch paths)
|
|
* must reject a leading '/' themselves (see utils_valid_batch_path). */
|
|
bool has_path_traversal(const char* path);
|
|
bool utils_valid_batch_path(const char* path);
|
|
bool format_human_bytes(unsigned long long bytes, char* buffer, size_t buffer_size);
|
|
/* --append / --append-verify tail-resume math (pure). A resume is eligible only
|
|
when an existing destination file is SHORTER than the source; the tail length
|
|
is then the difference. append_resume_eligible answers whether the shorter
|
|
file makes a resume possible; append_tail_length additionally returns that
|
|
tail length, refusing (false) the degenerate old_size >= check_size case. */
|
|
bool append_resume_eligible(unsigned long long old_size, unsigned long long check_size);
|
|
bool append_tail_length(unsigned long long old_size, unsigned long long check_size,
|
|
unsigned long long* tail_out);
|
|
/* Loopback / local-transport classification for the daemon auth gate and the
|
|
client credential rule. utils_sockaddr_is_loopback accepts 127.0.0.0/8,
|
|
IPv6 ::1 and IPv4-mapped ::ffff:127.x.x.x; utils_host_is_loopback additionally
|
|
accepts the literal "localhost". utils_fd_peer_is_local is fail-closed: it is
|
|
true only when getpeername SUCCEEDS and reports a loopback peer -- a non-socket
|
|
descriptor (pipe/socketpair) or any getpeername error yields false. See
|
|
utils.c for the exact accepted forms. */
|
|
bool utils_sockaddr_is_loopback(const struct sockaddr* addr);
|
|
bool utils_fd_peer_is_local(int fd);
|
|
bool utils_host_is_loopback(const char* host);
|
|
/* Numeric peer address of a connected fd (INET6_ADDRSTRLEN is always enough).
|
|
* Returns false and leaves buf empty when the fd is not a connected INET socket
|
|
* or getpeername/inet_ntop fails. Used by the daemon host-access gate; a false
|
|
* return is "cannot tell" and must be treated as fail-closed when ACLs apply. */
|
|
bool utils_fd_peer_ip(int fd, char* buf, size_t len);
|
|
/* Format a sockaddr as "ip:port" (IPv4) or "[ip]:port" (IPv6) for logging.
|
|
* Returns false (buf emptied) for a non-INET family or a formatting failure. */
|
|
bool utils_sockaddr_to_string(const struct sockaddr* addr, char* buf, size_t len);
|
|
|
|
#endif
|