From 1a83e284c448bcb95ad2c8a4b08f7dc1f4b10528 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 13 Sep 2026 04:16:28 +0200 Subject: [PATCH] perf(utils,file-list): index delete keep-set and --files-from lookups --- src/shared/file_list.c | 77 ++++++++++--- src/shared/file_list.h | 10 +- src/shared/utils.c | 229 +++++++++++++++++++++++++++++++++----- src/shared/utils.h | 35 ++++++ tests/runner.c | 2 + tests/test_file_list.c | 123 ++++++++++++++++++++ tests/test_file_list.h | 6 + tests/test_shared_utils.c | 38 +++++++ 8 files changed, 470 insertions(+), 50 deletions(-) create mode 100644 tests/test_file_list.c create mode 100644 tests/test_file_list.h diff --git a/src/shared/file_list.c b/src/shared/file_list.c index 528c8d4..b01a114 100644 --- a/src/shared/file_list.c +++ b/src/shared/file_list.c @@ -104,8 +104,37 @@ static int normalize_entry(const char* raw, size_t len, bool strip_line_endings, return result; } +/* Build the membership index: each non-empty entry plus every ancestor + directory prefix of it. The entry flag lets file_list_affects tell an exact + listed path from an ancestor of a listed path. An empty entry (the source + root) short-circuits every query, so it is recorded as whole_tree. */ +static bool file_list_index_build(FileListSet* set, char* err, size_t err_size) { + if (!str_hash_set_init(&set->node_index, (size_t)set->count * 2 + 1)) { + snprintf(err, err_size, "memory allocation failed"); + return false; + } + for (int i = 0; i < set->count; i++) { + const char* entry = set->entries[i]; + if (entry[0] == '\0') { + set->whole_tree = true; + continue; + } + if (!str_hash_set_insert_ref(&set->node_index, entry, true)) { + snprintf(err, err_size, "memory allocation failed"); + return false; + } + for (const char* slash = entry; (slash = strchr(slash, '/')) != NULL; slash++) { + if (!str_hash_set_insert_copy_n(&set->node_index, entry, (size_t)(slash - entry), false)) { + snprintf(err, err_size, "memory allocation failed"); + return false; + } + } + } + return true; +} + static FileListSet* string_list_to_set(StringList* raw, char* err, size_t err_size) { - FileListSet* set = malloc(sizeof(FileListSet)); + FileListSet* set = calloc(1, sizeof(FileListSet)); if (!set) { snprintf(err, err_size, "memory allocation failed"); return NULL; @@ -114,6 +143,10 @@ static FileListSet* string_list_to_set(StringList* raw, char* err, size_t err_si set->entries = raw->items; raw->items = NULL; raw->count = 0; + if (!file_list_index_build(set, err, err_size)) { + file_list_destroy(set); + return NULL; + } return set; } @@ -163,31 +196,39 @@ void file_list_destroy(FileListSet* set) { for (int i = 0; i < set->count; i++) free(set->entries[i]); free(set->entries); + str_hash_set_free(&set->node_index); free(set); } -static bool path_has_prefix(const char* path, const char* prefix) { - size_t plen = strlen(prefix); - if (strncmp(path, prefix, plen) != 0) - return false; - return path[plen] == '/' || path[plen] == '\0'; -} - bool file_list_affects(const FileListSet* set, const char* rel) { if (!set) return true; if (!rel) return false; - for (int i = 0; i < set->count; i++) { - const char* entry = set->entries[i]; - if (entry[0] == '\0') - return true; /* whole tree listed */ - if (strcmp(rel, entry) == 0) - return true; /* the entry itself is listed */ - if (path_has_prefix(rel, entry)) - return true; /* rel lives under a listed directory */ - if (path_has_prefix(entry, rel)) - return true; /* rel is an ancestor directory of a listed entry */ + if (set->whole_tree) + return true; /* whole tree listed */ + /* A node hit means `rel` is a listed entry, or an ancestor directory of one + (rel lives on the path to some listed entry). */ + if (str_hash_set_lookup(&set->node_index, rel, NULL)) + return true; + /* Otherwise `rel` is affected only when a listed entry is an ancestor of it; + walk rel's directory prefixes (which preserve path-boundary semantics) and + test each for an exact entry. */ + size_t len = strlen(rel); + while (len > 0) { + const char* slash = NULL; + for (size_t i = len; i-- > 0;) { + if (rel[i] == '/') { + slash = rel + i; + break; + } + } + if (!slash) + break; + len = (size_t)(slash - rel); + bool is_entry = false; + if (str_hash_set_lookup_n(&set->node_index, rel, len, &is_entry) && is_entry) + return true; } return false; } diff --git a/src/shared/file_list.h b/src/shared/file_list.h index 0ced332..53c1a79 100644 --- a/src/shared/file_list.h +++ b/src/shared/file_list.h @@ -1,6 +1,7 @@ #ifndef FILE_LIST_H #define FILE_LIST_H +#include "utils.h" #include #include @@ -12,11 +13,16 @@ * of "." means the whole tree, absolute entries and ".." traversal are * rejected at parse time. The set is immutable and shared read-only across * scanner worker threads. - */ - + * + * Membership is answered from `node_index`, built once at load time: it holds + * every entry plus every ancestor directory prefix of an entry, with the entry + * flag distinguishing an exact listed path from a mere ancestor. A lookup is + * O(path length) instead of O(entry count). */ typedef struct { char** entries; /* normalized rel paths; "" means the whole tree */ int count; + StrHashSet node_index; + bool whole_tree; /* an entry of "" lists the source root */ } FileListSet; /* Load and validate a --files-from file. When `null_separated` (-0/--from0) diff --git a/src/shared/utils.c b/src/shared/utils.c index c34d4a6..e15d55b 100644 --- a/src/shared/utils.c +++ b/src/shared/utils.c @@ -13,6 +13,7 @@ #include #include #include +#include static int authorized_root_fd = -1; static char* authorized_root_path; @@ -96,6 +97,153 @@ char* str_dup(const char* string) { return new_string; } +#define STR_HASH_SET_MIN_CAPACITY 16 + +static size_t str_hash_set_hash(const char* key, size_t len) { + return (size_t)XXH64(key, len, 0); +} + +/* Store an already-allocated key. Returns 1 when a new slot was filled and 0 + * for a duplicate (the caller keeps ownership of `key` when owned is true). */ +static int str_hash_set_put(StrHashSet* set, const char* key, size_t len, bool owned, + bool is_entry) { + size_t mask = set->capacity - 1; + size_t index = str_hash_set_hash(key, len) & mask; + while (true) { + StrHashSetSlot* slot = &set->slots[index]; + if (!slot->key) { + slot->key = key; + slot->owned = owned; + slot->is_entry = is_entry; + set->size++; + return 1; + } + if (strlen(slot->key) == len && memcmp(slot->key, key, len) == 0) { + if (is_entry) + slot->is_entry = true; + return 0; + } + index = (index + 1) & mask; + } +} + +static bool str_hash_set_resize(StrHashSet* set, size_t new_capacity) { + StrHashSetSlot* old_slots = set->slots; + size_t old_capacity = set->capacity; + StrHashSetSlot* slots = calloc(new_capacity, sizeof(StrHashSetSlot)); + if (!slots) + return false; + set->slots = slots; + set->capacity = new_capacity; + set->size = 0; + for (size_t i = 0; i < old_capacity; i++) { + if (old_slots[i].key) + (void)str_hash_set_put(set, old_slots[i].key, strlen(old_slots[i].key), old_slots[i].owned, + old_slots[i].is_entry); + } + free(old_slots); + return true; +} + +static bool str_hash_set_grow(StrHashSet* set) { + if (set->capacity != 0 && (set->size + 1) * 4 <= set->capacity * 3) + return true; + size_t new_capacity = set->capacity ? set->capacity * 2 : STR_HASH_SET_MIN_CAPACITY; + return str_hash_set_resize(set, new_capacity); +} + +bool str_hash_set_init(StrHashSet* set, size_t hint) { + if (!set) + return false; + set->slots = NULL; + set->capacity = 0; + set->size = 0; + size_t capacity = STR_HASH_SET_MIN_CAPACITY; + while (capacity < (hint + 1) * 2) + capacity *= 2; + set->slots = calloc(capacity, sizeof(StrHashSetSlot)); + if (!set->slots) + return false; + set->capacity = capacity; + return true; +} + +void str_hash_set_free(StrHashSet* set) { + if (!set) + return; + for (size_t i = 0; i < set->capacity; i++) { + if (set->slots[i].key && set->slots[i].owned) + free((void*)set->slots[i].key); + } + free(set->slots); + set->slots = NULL; + set->capacity = 0; + set->size = 0; +} + +bool str_hash_set_insert_ref(StrHashSet* set, const char* key, bool is_entry) { + if (!set || !key) + return false; + if (!str_hash_set_grow(set)) + return false; + return str_hash_set_put(set, key, strlen(key), false, is_entry) >= 0; +} + +bool str_hash_set_insert_copy_n(StrHashSet* set, const char* key, size_t len, bool is_entry) { + if (!set || !key) + return false; + bool present = false; + if (str_hash_set_lookup_n(set, key, len, &present)) { + if (is_entry) + (void)str_hash_set_put(set, key, len, false, true); /* upgrade in place */ + return true; + } + if (!str_hash_set_grow(set)) + return false; + char* copy = malloc(len + 1); + if (!copy) + return false; + memcpy(copy, key, len); + copy[len] = '\0'; + int result = str_hash_set_put(set, copy, len, true, is_entry); + if (result <= 0) { + free(copy); + return result == 0; + } + return true; +} + +static const StrHashSetSlot* str_hash_set_find_n(const StrHashSet* set, const char* key, + size_t len) { + if (!set || set->capacity == 0 || !key) + return NULL; + size_t mask = set->capacity - 1; + size_t index = str_hash_set_hash(key, len) & mask; + while (true) { + const StrHashSetSlot* slot = &set->slots[index]; + if (!slot->key) + return NULL; + if (strlen(slot->key) == len && memcmp(slot->key, key, len) == 0) + return slot; + index = (index + 1) & mask; + } +} + +bool str_hash_set_lookup_n(const StrHashSet* set, const char* key, size_t len, bool* is_entry) { + const StrHashSetSlot* slot = str_hash_set_find_n(set, key, len); + if (!slot) + return false; + if (is_entry) + *is_entry = slot->is_entry; + return true; +} + +bool str_hash_set_lookup(const StrHashSet* set, const char* key, bool* is_entry) { + if (!key) + return false; + return str_hash_set_lookup_n(set, key, strlen(key), is_entry); +} + char* output_escape(const char* string, bool eight_bit_output) { if (!string) return NULL; @@ -196,17 +344,40 @@ bool format_human_bytes(unsigned long long bytes, char* buffer, size_t buffer_si return written >= 0 && (size_t)written < buffer_size; } -static bool is_dir_in_manifest(const char* rel_path, ArrayList* manifest) { - size_t len = strlen(rel_path); +/* Build the keep-set index: every manifest entry is inserted as an exact entry + and every ancestor directory prefix of it as a non-entry node. A lookup of + `rel` therefore succeeds iff `rel` is a kept file, a kept directory, or an + ancestor directory of kept content (the old is_dir_in_manifest predicate); + the entry flag distinguishes an exact kept file from a mere prefix. */ +static bool build_keep_index(ArrayList* manifest, StrHashSet* index) { + if (!str_hash_set_init(index, manifest && manifest->size > 0 ? (size_t)manifest->size : 1)) + return false; + if (!manifest) + return true; for (int i = 0; i < manifest->size; i++) { const char* entry = (const char*)manifest->items[i]; - // Check if entry starts with rel_path + '/' or matches exactly - if (strncmp(entry, rel_path, len) == 0 && (entry[len] == '/' || entry[len] == '\0')) - return true; + if (!str_hash_set_insert_ref(index, entry, true)) + goto fail; + for (const char* slash = entry; (slash = strchr(slash, '/')) != NULL; slash++) { + if (!str_hash_set_insert_copy_n(index, entry, (size_t)(slash - entry), false)) + goto fail; + } } + return true; +fail: + str_hash_set_free(index); return false; } +static bool keep_is_dir(const StrHashSet* index, const char* rel_path) { + return str_hash_set_lookup(index, rel_path, NULL); +} + +static bool keep_is_file(const StrHashSet* index, const char* rel_path) { + bool is_entry = false; + return str_hash_set_lookup(index, rel_path, &is_entry) && is_entry; +} + /* True when child_rel is, or lies below, a protected entry. A prefix "a" therefore protects "a" and "a/b/c" but not "ab". Entries with top_level_only set only protect DIRECT children of the receive root (at_root); nested @@ -234,7 +405,7 @@ bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSki prefixes) mark the enclosing directory as surviving, exactly as they would make a real rmdir fail with ENOTEMPTY. Stops early once *count reaches the cap (sets *exceeds). Returns false on a traversal error. */ -static bool count_extras_fd(int dirfd, const char* rel_path, ArrayList* manifest, size_t cap, +static bool count_extras_fd(int dirfd, const char* rel_path, const StrHashSet* keep, size_t cap, size_t* count, bool* exceeds, const DeleteSkipEntry* skips, int skip_count, bool* survives) { /* openat(dirfd, ".") opens an independent file description: a dup() would @@ -284,15 +455,15 @@ static bool count_extras_fd(int dirfd, const char* rel_path, ArrayList* manifest bool child_ok = true; bool child_survives = true; if (childfd >= 0) { - child_ok = count_extras_fd(childfd, child_rel, manifest, cap, count, exceeds, skips, - skip_count, &child_survives); + child_ok = count_extras_fd(childfd, child_rel, keep, cap, count, exceeds, skips, skip_count, + &child_survives); close(childfd); } else if (errno != ENOENT) { operation_ok = false; } if (!child_ok) operation_ok = false; - if (is_dir_in_manifest(child_rel, manifest)) { + if (keep_is_dir(keep, child_rel)) { /* A directory with kept content below it is never removed. */ local_survives = true; } else if (child_survives) { @@ -308,13 +479,7 @@ static bool count_extras_fd(int dirfd, const char* rel_path, ArrayList* manifest } } } else { - bool found = false; - for (int i = 0; i < manifest->size; i++) { - if (strcmp((char*)manifest->items[i], child_rel) == 0) { - found = true; - break; - } - } + bool found = keep_is_file(keep, child_rel); if (!found) { if (*count >= cap) { *exceeds = true; @@ -330,7 +495,7 @@ static bool count_extras_fd(int dirfd, const char* rel_path, ArrayList* manifest return operation_ok; } -static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifest, +static bool delete_extras_fd(int dirfd, const char* rel_path, const StrHashSet* keep, size_t max_delete, size_t* deleted_count, const DeleteSkipEntry* skips, int skip_count) { /* Independent file description (see count_extras_fd). */ @@ -379,15 +544,15 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifes int childfd = openat(dirfd, entry->d_name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); bool child_removed = false; if (childfd >= 0) { - child_removed = delete_extras_fd(childfd, child_rel, manifest, max_delete, deleted_count, - skips, skip_count); + child_removed = delete_extras_fd(childfd, child_rel, keep, max_delete, deleted_count, skips, + skip_count); if (!child_removed) operation_ok = false; close(childfd); } else if (errno != ENOENT) { operation_ok = false; } - if (child_removed && !is_dir_in_manifest(child_rel, manifest)) { + if (child_removed && !keep_is_dir(keep, child_rel)) { if (*deleted_count >= max_delete) { operation_ok = false; } else { @@ -406,13 +571,7 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifes } } else { // Check if relative path is in manifest - bool found = false; - for (int i = 0; i < manifest->size; i++) { - if (strcmp((char*)manifest->items[i], child_rel) == 0) { - found = true; - break; - } - } + bool found = keep_is_file(keep, child_rel); if (!found) { if (*deleted_count >= max_delete) { operation_ok = false; @@ -443,6 +602,11 @@ DeleteWalkResult delete_extras_limited(const char* dest_root, ArrayList* manifes *deleted_out = 0; if (!manifest) return DELETE_WALK_ERROR; + /* Index the keep-set once so both passes answer membership in O(path length) + instead of scanning every manifest entry for every destination entry. */ + StrHashSet keep; + if (!build_keep_index(manifest, &keep)) + return DELETE_WALK_ERROR; int rootfd; if (authorized_root_fd >= 0) { if (authorized_root_path) @@ -454,29 +618,34 @@ DeleteWalkResult delete_extras_limited(const char* dest_root, ArrayList* manifes } else { rootfd = open(dest_root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); } - if (rootfd < 0) + if (rootfd < 0) { + str_hash_set_free(&keep); return DELETE_WALK_ERROR; + } if (max_delete != SIZE_MAX) { /* Rehearse the deletion first so a run that would exceed the cap removes nothing (rsync's all-or-nothing --max-delete contract). */ size_t count = 0; bool exceeds = false; bool survives = false; - bool counted_ok = count_extras_fd(rootfd, "", manifest, max_delete, &count, &exceeds, skips, + bool counted_ok = count_extras_fd(rootfd, "", &keep, max_delete, &count, &exceeds, skips, skip_count, &survives); if (!counted_ok) { close(rootfd); + str_hash_set_free(&keep); return DELETE_WALK_ERROR; } if (exceeds) { close(rootfd); + str_hash_set_free(&keep); return DELETE_WALK_LIMIT_EXCEEDED; } } size_t deleted_count = 0; - bool ok = delete_extras_fd(rootfd, "", manifest, max_delete, &deleted_count, skips, skip_count); + bool ok = delete_extras_fd(rootfd, "", &keep, max_delete, &deleted_count, skips, skip_count); if (close(rootfd) != 0) ok = false; + str_hash_set_free(&keep); if (deleted_out) *deleted_out = deleted_count; return ok ? DELETE_WALK_OK : DELETE_WALK_ERROR; diff --git a/src/shared/utils.h b/src/shared/utils.h index 4c9144a..f27c0c0 100644 --- a/src/shared/utils.h +++ b/src/shared/utils.h @@ -6,6 +6,41 @@ #include #include +/* Small open-addressing string hash set used to turn quadratic membership + * scans into O(path length) lookups (the --delete keep-set and the + * --files-from allow-set). Keys are hashed with xxHash64 (seed 0); collisions + * are resolved by linear probing over a power-of-two table that grows at 75% + * load. Slots may either borrow a caller-owned key (insert_ref) or own an + * internal copy (insert_copy_n); owned copies are released by + * str_hash_set_free. The set is not thread-safe for mutation, but a fully + * built set supports concurrent read-only lookups. */ +typedef struct { + const char* key; /* NULL marks an empty slot */ + bool owned; /* key is an internal copy that free() must release */ + bool is_entry; /* key was inserted as an exact entry, not just a prefix */ +} StrHashSetSlot; + +typedef struct { + StrHashSetSlot* slots; + size_t capacity; /* power of two, zero before init */ + size_t size; +} StrHashSet; + +/* Initialize an empty set sized for roughly `hint` entries. Returns false on + * allocation failure. */ +bool str_hash_set_init(StrHashSet* set, size_t hint); +void str_hash_set_free(StrHashSet* set); +/* Insert a borrowed key (must outlive the set). A duplicate only upgrades + * is_entry. Returns false on allocation failure. */ +bool str_hash_set_insert_ref(StrHashSet* set, const char* key, bool is_entry); +/* Insert a copy of the first `len` bytes of `key` (which need not be + * NUL-terminated). Returns false on allocation failure. */ +bool str_hash_set_insert_copy_n(StrHashSet* set, const char* key, size_t len, bool is_entry); +/* Look up a NUL-terminated key / a key of `len` bytes. On a hit, optionally + * reports whether the stored key was inserted as an exact entry. */ +bool str_hash_set_lookup(const StrHashSet* set, const char* key, bool* is_entry); +bool str_hash_set_lookup_n(const StrHashSet* set, const char* key, size_t len, bool* is_entry); + char* str_dup(const char* string); char* output_escape(const char* string, bool eight_bit_output); char* path_cat(const char* path1, const char* path2); diff --git a/tests/runner.c b/tests/runner.c index ec58de9..51938e1 100644 --- a/tests/runner.c +++ b/tests/runner.c @@ -12,6 +12,7 @@ #include "test_delay_updates.h" #include "test_delta.h" #include "test_file.h" +#include "test_file_list.h" #include "test_file_sendfile.h" #include "test_fuzz_smoke.h" #include "test_glob.h" @@ -67,6 +68,7 @@ int main() { RUN_TEST(test_glob); RUN_TEST(test_iconv); RUN_TEST(test_file); + RUN_TEST(test_file_list); RUN_TEST(test_trust_sender); RUN_TEST(test_delay_updates); RUN_TEST(test_file_sendfile); diff --git a/tests/test_file_list.c b/tests/test_file_list.c new file mode 100644 index 0000000..bf31db5 --- /dev/null +++ b/tests/test_file_list.c @@ -0,0 +1,123 @@ +#include "test_file_list.h" +#include "file_list.h" +#include "test_utils.h" +#include +#include +#include + +/* Reference implementation of the ORIGINAL file_list_affects linear scan. The + indexed implementation must agree with it on every query; this pins the + subtle semantics: empty entry == whole tree, exact match, rel under a listed + directory, and rel an ancestor directory of a listed entry. */ +static bool reference_affects(const FileListSet* set, const char* rel) { + if (!set) + return true; + if (!rel) + return false; + for (int i = 0; i < set->count; i++) { + const char* entry = set->entries[i]; + if (entry[0] == '\0') + return true; + if (strcmp(rel, entry) == 0) + return true; + size_t entry_len = strlen(entry); + if (strncmp(rel, entry, entry_len) == 0 && (rel[entry_len] == '/' || rel[entry_len] == '\0')) + return true; + size_t rel_len = strlen(rel); + if (strncmp(entry, rel, rel_len) == 0 && (entry[rel_len] == '/' || entry[rel_len] == '\0')) + return true; + } + return false; +} + +static void write_list(const char* path, const char* bytes) { + FILE* fp = fopen(path, "wb"); + EXPECT_NOT_NULL(fp); + size_t len = strlen(bytes); + EXPECT_EQ_INT((int)fwrite(bytes, 1, len, fp), (int)len); + fclose(fp); +} + +static void check_queries(const FileListSet* set, const char* const* queries, int query_count) { + for (int i = 0; i < query_count; i++) { + bool expected = reference_affects(set, queries[i]); + bool actual = file_list_affects(set, queries[i]); + if (expected != actual) { + printf(" [FAIL] affects(\"%s\"): expected %d, got %d\n", queries[i], expected, actual); + current_test_failed = true; + return; + } + } +} + +static void test_membership_matches_reference() { + const char* path = "test_file_list_case.txt"; + char err[160]; + + /* Nested directories, an ancestor of a listed entry, an exact file, a + non-matching neighbor with the same prefix, and a literal '*'. */ + write_list(path, "a\na/b\na/b/c\nab\nc.txt\nsub/b.bin\n*\n"); + FileListSet* set = file_list_load(path, false, err, sizeof(err)); + EXPECT_NOT_NULL(set); + const char* queries[] = { + "a", "a/b", "a/b/c", "a/b/c/d", "a/bx", "a/x", "ab", + "abc", "c.txt", "c.txt/x", "c", "sub", "sub/b.bin", "sub/b.bin/z", + "sub2", "*", "x", "", "a/b/cd", "/", "a/b/", + }; + check_queries(set, queries, (int)(sizeof(queries) / sizeof(queries[0]))); + EXPECT_TRUE(file_list_affects(set, "a/b/c/d")); + EXPECT_TRUE(file_list_affects(set, "a/bx")); /* under listed directory "a" */ + EXPECT_TRUE(file_list_affects(set, "a/b/c")); + EXPECT_TRUE(file_list_affects(set, "a/x")); /* under listed directory "a" */ + EXPECT_FALSE(file_list_affects(set, "abc")); /* component boundary: not "a" */ + EXPECT_TRUE(file_list_affects(set, "sub")); + EXPECT_FALSE(file_list_affects(set, "sub2")); + file_list_destroy(set); + remove(path); + + /* A single "." entry means the whole tree: every non-NULL query is true. */ + write_list(path, ".\n"); + set = file_list_load(path, false, err, sizeof(err)); + EXPECT_NOT_NULL(set); + const char* root_queries[] = {"", "a", "a/b/c", "unrelated", "*", "/"}; + for (int i = 0; i < (int)(sizeof(root_queries) / sizeof(root_queries[0])); i++) + EXPECT_TRUE(file_list_affects(set, root_queries[i])); + EXPECT_FALSE(file_list_affects(set, NULL)); + check_queries(set, root_queries, (int)(sizeof(root_queries) / sizeof(root_queries[0]))); + file_list_destroy(set); + remove(path); + + /* Trailing slashes and "./" prefixes normalize away, so the query matches the + clean path (and not the raw spelling). */ + write_list(path, "./dir/\ndir2/./x\n"); + set = file_list_load(path, false, err, sizeof(err)); + EXPECT_NOT_NULL(set); + EXPECT_TRUE(file_list_affects(set, "dir")); + EXPECT_TRUE(file_list_affects(set, "dir/x")); + EXPECT_TRUE(file_list_affects(set, "dir2/x")); + EXPECT_TRUE(file_list_affects(set, "dir2")); + EXPECT_TRUE(file_list_affects(set, "dir/")); /* boundary prefix of listed "dir" */ + check_queries(set, (const char*[]){"dir", "dir/", "dir/x", "dir2", "dir2/x", "dir3"}, 6); + file_list_destroy(set); + remove(path); + + /* An empty file yields an empty set: nothing is affected, and NULL set still + means "everything". */ + write_list(path, ""); + set = file_list_load(path, false, err, sizeof(err)); + EXPECT_NOT_NULL(set); + EXPECT_EQ_INT(set->count, 0); + EXPECT_FALSE(file_list_affects(set, "a")); + EXPECT_FALSE(file_list_affects(set, "")); + check_queries(set, (const char*[]){"a", "a/b", ""}, 3); + file_list_destroy(set); + remove(path); + + /* NULL set is the unrestricted case. */ + EXPECT_TRUE(file_list_affects(NULL, "anything")); + EXPECT_TRUE(file_list_affects(NULL, NULL)); +} + +void test_file_list() { + test_membership_matches_reference(); +} diff --git a/tests/test_file_list.h b/tests/test_file_list.h new file mode 100644 index 0000000..4a8d6d8 --- /dev/null +++ b/tests/test_file_list.h @@ -0,0 +1,6 @@ +#ifndef TEST_FILE_LIST_H +#define TEST_FILE_LIST_H + +void test_file_list(); + +#endif diff --git a/tests/test_shared_utils.c b/tests/test_shared_utils.c index 86ce24b..6663695 100644 --- a/tests/test_shared_utils.c +++ b/tests/test_shared_utils.c @@ -150,6 +150,43 @@ static void test_walker_removes_extras_keeps_manifest_and_protected() { free(root); } +static void test_walker_keeps_nested_manifest_dirs() { + /* The keep-set index must preserve deep content: a directory is protected + when its own name is a keep entry OR when kept content lives below it, and + an exact kept file survives while its siblings are removed. */ + char* root = make_walk_root("nestedkeep"); + EXPECT_NOT_NULL(root); + EXPECT_TRUE(write_file_at(root, "extra.txt", "extra")); + EXPECT_EQ_INT(make_subdir(root, "keepdir"), 0); + EXPECT_EQ_INT(make_subdir(root, "keepdir/deep"), 0); + EXPECT_TRUE(write_file_at(root, "keepdir/deep/keep.txt", "kept")); + EXPECT_TRUE(write_file_at(root, "keepdir/extra2.txt", "extra")); + EXPECT_EQ_INT(make_subdir(root, "dropdir"), 0); + EXPECT_EQ_INT(make_subdir(root, "keep2"), 0); + EXPECT_TRUE(write_file_at(root, "keep2/inner.txt", "kept")); + EXPECT_EQ_INT(make_subdir(root, "keep3"), 0); + + const char* keeps[] = {"keepdir/deep/keep.txt", "keep2/inner.txt", "keep3"}; + ArrayList* manifest = make_manifest_strings(keeps, 3); + EXPECT_NOT_NULL(manifest); + size_t deleted = 0; + DeleteWalkResult result = delete_extras_limited(root, manifest, 100000, NULL, 0, &deleted); + EXPECT_EQ_INT((int)result, (int)DELETE_WALK_OK); + EXPECT_FALSE(file_exists(root, "extra.txt")); + EXPECT_TRUE(file_exists(root, "keepdir/deep/keep.txt")); + EXPECT_FALSE(file_exists(root, "keepdir/extra2.txt")); + EXPECT_TRUE(dir_exists(root, "keepdir")); + EXPECT_TRUE(dir_exists(root, "keepdir/deep")); + EXPECT_FALSE(dir_exists(root, "dropdir")); + EXPECT_TRUE(dir_exists(root, "keep2")); + EXPECT_TRUE(file_exists(root, "keep2/inner.txt")); + EXPECT_TRUE(dir_exists(root, "keep3")); /* an exact directory keep entry survives */ + EXPECT_EQ_INT((int)deleted, 3); + array_list_delete(manifest); + remove_walk_tree(root); + free(root); +} + static void test_walker_max_delete_exceeded_deletes_nothing() { char* root = make_walk_root("maxdel"); EXPECT_NOT_NULL(root); @@ -421,6 +458,7 @@ static void test_fd_peer_ip() { void test_shared_utils() { test_walker_removes_extras_keeps_manifest_and_protected(); + test_walker_keeps_nested_manifest_dirs(); test_walker_max_delete_exceeded_deletes_nothing(); test_walker_max_delete_exact_bound_deletes(); test_walker_unlimited_deletes_all();