From 4638030288d85a45fde186ad3e3b75e5cfa807b9 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 13:34:55 +0200 Subject: [PATCH 01/10] refactor(client): split reporting/scan/manifest out of client_send Move the stats/progress reporting, scanner-preparation/scan helpers and manifest/list/dry-run senders out of the ~3.9k-line client_send.c into client_report.c, client_scan.c and client_manifest.c, sharing declarations through the new internal client_send_internal.h. client_send.c keeps the transfer orchestration and is now ~2.1k lines. Decompose the monolithic send_files into static phase helpers (send_files_prepare/_prepare_delete/_run/_finalize/_cleanup) driven by a single SendFilesState; ownership, ordering and exit codes are unchanged. No behavior change. --- CMakeLists.txt | 3 + src/client/client_manifest.c | 682 +++++++++ src/client/client_report.c | 704 +++++++++ src/client/client_scan.c | 454 ++++++ src/client/client_send.c | 2313 ++++------------------------- src/client/client_send_internal.h | 91 ++ 6 files changed, 2207 insertions(+), 2040 deletions(-) create mode 100644 src/client/client_manifest.c create mode 100644 src/client/client_report.c create mode 100644 src/client/client_scan.c create mode 100644 src/client/client_send_internal.h diff --git a/CMakeLists.txt b/CMakeLists.txt index a55d93e..f567378 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -136,6 +136,9 @@ set(SERVER_MAIN_SRCS src/server/server.c) # Client implementation (no main): everything except the CLI entry point. set(CLIENT_CORE_SRCS src/client/change_list.c + src/client/client_manifest.c + src/client/client_report.c + src/client/client_scan.c src/client/client_send.c src/client/client_validation.c src/client/scanner.c diff --git a/src/client/client_manifest.c b/src/client/client_manifest.c new file mode 100644 index 0000000..d476bb3 --- /dev/null +++ b/src/client/client_manifest.c @@ -0,0 +1,682 @@ +#include "client_send_internal.h" +#include "array_list.h" +#include "change_list.h" +#include "charset.h" +#include "config.h" +#include "data.h" +#include "delta.h" +#include "file.h" +#include "format.h" +#include "log.h" +#include "protocol.h" +#include "scanner.h" +#include "transport_tls.h" +#include "utils.h" +#include +#include +#include +#include +#include + +/* True when --dry-run should contact a receiver rather than running the + * client-side local manifest. Any target a real run would reach over the wire + * selects the server-contacting path: a remote (SSH host:path), a daemon + * (host::module/path), an explicit --server-host, --server-port/--port, TLS, or + * a source-bind --address. A plain local destination (none of these) keeps the + * original client-side behavior, which never dials the default 127.0.0.1:8080. */ +bool dry_run_targets_server(const Config* config) { + if (!config) + return false; + if (config->transport == TRANSPORT_SSH) + return true; + if (config->module && config->module[0] != '\0') + return true; + if (config->server_host_set || config->server_port_set) + return true; + if (config->use_tls) + return true; + if (config->address != NULL) + return true; + return false; +} + +bool add_chunk_to_manifest(ArrayList* manifest, const Chunk* chunk) { + if (!manifest) + return true; + for (int i = 0; i < chunk->element_count; i++) { + const char* path = file_wire_path(chunk->items[i]); + if (*path == '/') + path++; + char* entry = str_dup(path); + if (!entry) { + log_message(LOG_LEVEL_ERROR, "Failed to allocate manifest entry"); + return false; + } + if (!array_list_add(manifest, entry)) { + free(entry); + return false; + } + } + return true; +} + +/* Print dry-run manifest showing files that would be transferred. Returns 0 on success. */ +int send_dry_run_manifest(const Config* config) { + int skipped = 0; + ArrayList* missing_dest = NULL; + if (config->delete_missing_args) { + missing_dest = array_list_create(free); + if (!missing_dest) + return -1; + } + if (!files_from_list_check(config, missing_dest, &skipped)) { + if (missing_dest) + array_list_delete(missing_dest); + return -1; + } + PreparedScanner prepared; + if (!prepare_scanner(config, 0, &prepared)) { + if (missing_dest) + array_list_delete(missing_dest); + return -1; + } + DirectoryScanner* scanner = + directory_scanner_create_with_options(config->send_directory, &prepared.options); + if (!scanner) { + prepared_scanner_destroy(&prepared); + if (missing_dest) + array_list_delete(missing_dest); + return -1; + } + Chunk* chunk; + int file_count = 0; + unsigned long long total_bytes = 0; + char size_buffer[32]; + if (!config->quiet) + printf("Dry run: files to be transferred\n"); + while ((chunk = directory_scanner_next(scanner)) != NULL) { + for (int i = 0; i < chunk->element_count; i++) { + if (!config->quiet) { + char* escaped_path = + output_escape(file_wire_path(chunk->items[i]), config->eight_bit_output); + if (!escaped_path) { + chunk_destroy(chunk); + directory_scanner_destroy(scanner); + prepared_scanner_destroy(&prepared); + if (missing_dest) + array_list_delete(missing_dest); + return -1; + } + if (config->human_readable) + printf( + " %s (%s)\n", escaped_path, + display_bytes(chunk->items[i]->data->size, true, size_buffer, sizeof(size_buffer))); + else + printf(" %s (%zu bytes)\n", escaped_path, chunk->items[i]->data->size); + free(escaped_path); + } + total_bytes += chunk->items[i]->data->size; + file_count++; + } + chunk_destroy(chunk); + } + directory_scanner_destroy(scanner); + prepared_scanner_destroy(&prepared); + /* --delete-missing-args: the missing entries' destination mirrors render as + would-be deletions (rsync's dry-run also lists its *deleting lines). */ + if (missing_dest && !config->quiet) { + for (int i = 0; i < missing_dest->size; i++) { + char* escaped = output_escape((char*)missing_dest->items[i], config->eight_bit_output); + printf(" %s (missing; would be deleted)\n", escaped ? escaped : ""); + free(escaped); + } + } + if (missing_dest) + array_list_delete(missing_dest); + if (!config->quiet) { + if (config->human_readable) + printf("Total: %d files, %s\n", file_count, + display_bytes(total_bytes, true, size_buffer, sizeof(size_buffer))); + else + printf("Total: %d files, %.1f MB\n", file_count, (double)total_bytes / (double)BYTES_PER_MIB); + } + return 0; +} + +typedef struct { + char* name; /* transfer-relative name ("" == the source root) */ + mode_t mode; + unsigned long long size; + time_t mtime; + long mtime_nsec; + bool is_dir; + bool is_symlink; + char* link_target; +} ListEntry; + +static void list_entries_destroy(ListEntry* entries, size_t count) { + if (entries == NULL) + return; + for (size_t i = 0; i < count; i++) { + free(entries[i].name); + free(entries[i].link_target); + } + free(entries); +} + +static int compare_list_entries(const void* left, const void* right) { + const ListEntry* a = (const ListEntry*)left; + const ListEntry* b = (const ListEntry*)right; + return strcmp(a->name, b->name); +} + +/* Relative path of an entry below `root` ("" for the root itself). Mirrors + * change_list's relative_name for list-only rendering. */ +static char* list_relative_name(const char* root, const char* full) { + if (root == NULL || full == NULL) + return str_dup(full != NULL ? full : ""); + size_t root_len = strlen(root); + while (root_len > 1 && root[root_len - 1] == '/') + root_len--; + if (strncmp(root, full, root_len) == 0) { + if (full[root_len] == '\0') + return str_dup(""); + if (full[root_len] == '/') + return str_dup(full + root_len + 1); + } + return str_dup(full); +} + +/* --list-only: print an ls-style listing of the entries that WOULD be + * transferred and exit without contacting the server or writing anything. + * Names are transfer-relative (rsync prints `a.txt`, `sub/b.txt`, `.`) and + * directory entries are included. Returns 0 on success, 1 on error. */ +int send_list_only(const Config* config) { + int skipped = 0; + if (!files_from_list_check(config, NULL, &skipped)) + return 1; + PreparedScanner prepared; + if (!prepare_scanner(config, 0, &prepared)) + return 1; + prepared.options.use_metadata = true; /* capture mode + mtime for the listing */ + prepared.options.list_dirs = true; + DirectoryScanner* scanner = + directory_scanner_create_with_options(config->send_directory, &prepared.options); + if (!scanner) { + prepared_scanner_destroy(&prepared); + return 1; + } + ListEntry* entries = NULL; + size_t count = 0; + size_t capacity = 0; + bool oom = false; + + /* rsync lists the source root itself (as "."). Only when the source is a + * directory and no --files-from subset is in effect. */ + if (config->files_from_set == NULL && config->send_directory != NULL) { + struct stat st; + if (stat(config->send_directory, &st) == 0 && S_ISDIR(st.st_mode)) { + capacity = 64; + entries = calloc(capacity, sizeof(ListEntry)); + if (entries == NULL) { + oom = true; + } else if ((entries[0].name = str_dup("")) == NULL) { + /* A NULL name would be dereferenced by qsort/render: fail the listing. */ + oom = true; + } else { + entries[0].mode = st.st_mode; + entries[0].mtime = st.st_mtime; + entries[0].mtime_nsec = st.st_mtim.tv_nsec; + entries[0].size = (unsigned long long)st.st_size; + entries[0].is_dir = true; + count = 1; + } + } + } + + Chunk* chunk; + while (!oom && (chunk = directory_scanner_next(scanner)) != NULL) { + for (int i = 0; i < chunk->element_count; i++) { + File* f = chunk->items[i]; + if (f == NULL) + continue; + if (count == capacity) { + size_t new_capacity = capacity > 0 ? capacity * 2 : 64; + if (new_capacity <= capacity) { + oom = true; + break; + } + ListEntry* grown = realloc(entries, new_capacity * sizeof(ListEntry)); + if (!grown) { + oom = true; + break; + } + entries = grown; + memset(entries + capacity, 0, (new_capacity - capacity) * sizeof(ListEntry)); + capacity = new_capacity; + } + char* name = list_relative_name(config->send_directory, file_wire_path(f)); + if (!name) { + oom = true; + break; + } + mode_t mode = 0; + time_t mtime = 0; + long mtime_nsec = 0; + if (f->metadata != NULL) { + mode = f->metadata->mode; + mtime = f->metadata->mtime_sec; + mtime_nsec = f->metadata->mtime_nsec; + } else { + struct stat st; + if (lstat(f->path, &st) == 0) { + mode = st.st_mode; + mtime = st.st_mtime; + mtime_nsec = st.st_mtim.tv_nsec; + } + } + entries[count].name = name; + entries[count].mode = mode; + entries[count].mtime = mtime; + entries[count].mtime_nsec = mtime_nsec; + if (f->is_symlink) + entries[count].size = f->symlink_target != NULL ? strlen(f->symlink_target) : 0; + else if (f->is_dir) { + struct stat dir_st; + entries[count].size = stat(f->path, &dir_st) == 0 ? (unsigned long long)dir_st.st_size : 0; + } else + entries[count].size = f->data != NULL ? f->data->size : 0; + entries[count].is_dir = f->is_dir; + entries[count].is_symlink = f->is_symlink; + entries[count].link_target = + f->is_symlink && f->symlink_target ? str_dup(f->symlink_target) : NULL; + count++; + } + chunk_destroy(chunk); + } + bool failed = oom || directory_scanner_failed(scanner) || directory_scanner_had_io_error(scanner); + directory_scanner_destroy(scanner); + prepared_scanner_destroy(&prepared); + if (failed) { + list_entries_destroy(entries, count); + if (oom) + log_message(LOG_LEVEL_ERROR, "memory allocation failed while listing"); + return 1; + } + if (count > 1) + qsort(entries, count, sizeof(ListEntry), compare_list_entries); + for (size_t i = 0; i < count; i++) { + ChangeEvent event; + memset(&event, 0, sizeof(event)); + event.name = entries[i].name; + event.path = entries[i].name; + event.mode = entries[i].mode; + event.size = entries[i].size; + event.mtime_sec = entries[i].mtime; + event.mtime_nsec = entries[i].mtime_nsec; + event.is_directory = entries[i].is_dir; + event.is_symlink = entries[i].is_symlink; + event.symlink_target = entries[i].link_target; + char* line = change_render_list_line(config, &event); + if (line != NULL) { + char* escaped = output_escape(line, config->eight_bit_output); + printf("%s\n", escaped != NULL ? escaped : line); + free(escaped); + free(line); + } + } + list_entries_destroy(entries, count); + return 0; +} + +/* Send the delete manifest to the server. Returns 0 on success, -1 on + failure. It carries FOUR sections: the keep-set paths, the protected + excluded prefixes, the --delete-missing-args exact-delete paths, and the + destination-relative directories the sender synchronized this run. + When --delete-excluded is given `protected` is empty: excluded destination + mirrors are then ordinary extras and are removed. When + --delete-missing-args is active `missing_args` holds the destination mirrors + of missing --files-from entries: each is an explicit receiver-side deletion + request, independent of the extras walk. `synced_dirs` confines the extras + walk to entries directly inside a synchronized directory. A NULL + keep-set / protected / missing / dirs list transmits an empty section. All + four sections are unbounded on the sender; the receiver enforces + MAX_MANIFEST_ENTRIES per section and a single MAX_MANIFEST_BYTES budget + shared across the sections, rejecting (with STATUS_ERROR) an over-budget + frame. A heavily filtered source whose exclusion list is large therefore + fails the run cleanly on the receiver rather than being truncated. */ +int send_delete_manifest(int fd, ArrayList* manifest, ArrayList* protected_prefixes, + ArrayList* size_skipped, ArrayList* missing_args, ArrayList* synced_dirs) { + if (!send_status(fd, STATUS_MANIFEST)) + return -1; + int keep_count = manifest ? manifest->size : 0; + if (!send_int(fd, keep_count)) + return -1; + for (int i = 0; i < keep_count; i++) { + if (!send_wire_str(fd, (char*)manifest->items[i])) + return -1; + } + /* The receiver has ONE protected-prefix section; filter-excluded prefixes + (dropped under --delete-excluded) and size-pruned prefixes (always + protected) are concatenated into it. */ + int protected_count = + (protected_prefixes ? protected_prefixes->size : 0) + (size_skipped ? size_skipped->size : 0); + if (!send_int(fd, protected_count)) + return -1; + if (protected_prefixes) { + for (int i = 0; i < protected_prefixes->size; i++) { + if (!send_wire_str(fd, (char*)protected_prefixes->items[i])) + return -1; + } + } + if (size_skipped) { + for (int i = 0; i < size_skipped->size; i++) { + if (!send_wire_str(fd, (char*)size_skipped->items[i])) + return -1; + } + } + int missing_count = missing_args ? missing_args->size : 0; + if (!send_int(fd, missing_count)) + return -1; + for (int i = 0; i < missing_count; i++) { + if (!send_wire_str(fd, (char*)missing_args->items[i])) + return -1; + } + int dirs_count = synced_dirs ? synced_dirs->size : 0; + if (!send_int(fd, dirs_count)) + return -1; + for (int i = 0; i < dirs_count; i++) { + if (!send_wire_str(fd, (char*)synced_dirs->items[i])) + return -1; + } + return 0; +} + +/* Transmit the keep-set manifest and wait for the receiver's verdict. Used by + --delete-before/--delete-during, where the extras are removed on the receiver + BEFORE the first byte of file data is sent: the receiver acknowledges with + STATUS_OK once the bounded delete committed, or STATUS_ERROR if it could not + (in which case the sender aborts without streaming any data). The ACK may + take much longer than an ordinary per-message round trip because the receiver + performs the whole bounded deletion walk (up to MAX_SERVER_DELETE_COUNT + unlinks) before replying, so the wait uses a generous explicit deadline + instead of the default 60 s receive window. */ +#define DELETE_ACK_TIMEOUT_SEC 3600 +/* While waiting for the (potentially slow) receiver-side deletion, send a + * STATUS_KEEPALIVE at most this often so the connection is demonstrably alive + * and neither side's per-message timeout trips. */ +#define DELETE_ACK_KEEPALIVE_SEC 10 + +bool send_delete_manifest_early(Client* client, ArrayList* manifest, ArrayList* protected_prefixes, + ArrayList* size_skipped, ArrayList* missing_args, + ArrayList* synced_dirs) { + if (!client || !manifest) + return false; + if (send_delete_manifest(client->file_descriptor, manifest, protected_prefixes, size_skipped, + missing_args, synced_dirs) != 0) + return false; + Status ack; + /* The wait is long (up to an hour) and runs inline on this thread: a helper + * thread would race the non-thread-safe protocol send path, so keepalives are + * emitted from this wait loop itself. A Ctrl-C/SIGTERM abort flag also ends + * the wait; the caller then best-effort sends STATUS_ABORT. */ + if (!receive_status_keepalive(client->file_descriptor, &ack, DELETE_ACK_TIMEOUT_SEC, + DELETE_ACK_KEEPALIVE_SEC, client_abort_pending)) { + /* A Ctrl-C/SIGTERM abort ends the wait above; tell the receiver before the + caller tears the connection down (best-effort). */ + if (client_abort_pending()) { + log_info_message(LOG_INFO_MISC, + "Abort requested while awaiting delete ack; sending STATUS_ABORT"); + send_status(client->file_descriptor, STATUS_ABORT); + } + return false; + } + if (ack != STATUS_OK) { + log_server_rejection("Server failed to delete files before the transfer"); + return false; + } + return true; +} + +/* Server-contacting --dry-run. Connects to the configured remote/daemon and + * runs the normal per-file incremental decision WITHOUT transmitting any file + * data: the receiver (which also sees dry_run=true on the wire) answers + * STATUS_OK for an up-to-date file and STATUS_DRY_RUN_TRANSFER for a file it + * would otherwise write, mutating nothing on either side. The would-transfer + * set and the same trailer as the local dry-run are printed. A + * --compare-dest exact basis hit with no destination copy is reported as a + * skip by the receiver. + * + * Only regular files take the receiver-consulted check; directory / symlink / + * special / hard-link-sibling entries have no per-file content check, so they + * are reported conservatively as would-transfer and their frames are never + * sent (which is what keeps the receiver mutation-free). --delete* is + * deliberately NOT transmitted in dry-run, so no deletion can occur; the + * would-delete manifest report is a documented follow-up. + * + * Returns 0 on success, 1 on error. */ +int send_dry_run_remote(Config* config) { + int from_skipped = 0; + ArrayList* missing_args = NULL; + if (config->delete_missing_args) { + missing_args = array_list_create(free); + if (!missing_args) + return 1; + } + if (!files_from_list_check(config, missing_args, &from_skipped)) { + if (missing_args) + array_list_delete(missing_args); + return 1; + } + if (missing_args) + array_list_delete(missing_args); + /* A live session may follow, so arm graceful abort handling. */ + client_set_abort_armed(true); + Client* client = connect_transfer_client(config); + if (!client) { + if (config->transport == TRANSPORT_TCP) + log_message(LOG_LEVEL_ERROR, "could not connect to server%s", + config->use_tls ? " via TLS" : ""); + client_set_abort_armed(false); + return 1; + } + ProtocolSession session; + protocol_session_init(&session, client->file_descriptor, client->file_descriptor); + protocol_session_set_io_timeout(&session, config->timeout); + protocol_session_set_ssl(&session, (SSL*)client->ssl); + protocol_session_bind(&session); + + int ret = 1; + time_t dry_start = time(NULL); + ReceiverStats dry_stats; + memset(&dry_stats, 0, sizeof(dry_stats)); + PreparedScanner prepared; + memset(&prepared, 0, sizeof(prepared)); + DirectoryScanner* scanner = NULL; + ArrayList* dry_manifest = NULL; + ArrayList* dry_dirs = NULL; + ArrayList* dry_excluded = NULL; + ArrayList* dry_size_skipped = NULL; + if (!config_send(client->file_descriptor, config)) + goto dry_fail; + receive_daemon_motd(client, config); + if (!prepare_scanner(config, 0, &prepared)) + goto dry_fail; + /* -n --delete: build the same keep-set manifest, protected prefixes, and + synchronized-directory scope a real run would send, so the receiver's + read-only extras walk enumerates exactly the deletions a real run makes. */ + if (config->use_delete) { + dry_manifest = array_list_create(free); + dry_dirs = array_list_create(free); + dry_size_skipped = array_list_create(free); + if (!dry_manifest || !dry_dirs || !dry_size_skipped) + goto dry_fail; + if (!config->delete_excluded) { + dry_excluded = array_list_create(free); + if (!dry_excluded) + goto dry_fail; + prepared.options.excluded_paths = dry_excluded; + } + prepared.options.size_skipped_paths = dry_size_skipped; + /* A --files-from subset confines the extras walk to the directories the + scan synchronized; a full recursive transfer marks the root itself. */ + if (config->files_from_set == NULL) { + char* root_marker = delete_scope_root_marker(config); + if (!root_marker || !array_list_add(dry_dirs, root_marker)) { + free(root_marker); + goto dry_fail; + } + } else { + prepared.options.synced_dirs = dry_dirs; + } + } + scanner = directory_scanner_create_with_options(config->send_directory, &prepared.options); + if (!scanner) + goto dry_fail; + + int file_count = 0; + unsigned long long total_bytes = 0; + char size_buffer[32]; + if (!config->quiet) + printf("Dry run: files to be transferred\n"); + Chunk* chunk; + while ((chunk = directory_scanner_next(scanner)) != NULL) { + if (dry_manifest && !add_chunk_to_manifest(dry_manifest, chunk)) { + chunk_destroy(chunk); + goto dry_fail; + } + for (int i = 0; i < chunk->element_count; i++) { + File* f = chunk->items[i]; + if (!f) + continue; + unsigned long long fsize = f->data ? f->data->size : 0; + bool would; + if (f->is_dir || f->is_symlink || f->is_special || + (f->link_group != 0 && !f->link_first && f->hardlink_target != NULL)) { + /* No receiver-side content check exists for these frame types; a real + run would (re)create them, so report would-transfer and send no + frame (the receiver must stay mutation-free). */ + would = true; + } else if (fsize > MAX_RECEIVE_WHOLE_FILE_SIZE && !config->use_incremental && + !config_has_basis(config)) { + /* A non-incremental run streams a >whole-file-limit source without the + STATUS_CHECK handshake, so no read-only receiver decision is possible + (and none is needed: a real run would transfer it). */ + would = true; + } else { + DeltaSignature* sig = NULL; + unsigned long long resume_offset = 0; + int rc = incremental_check(client, f, config, &sig, &resume_offset); + delta_signature_destroy(sig); + if (rc < 0) { + chunk_destroy(chunk); + goto dry_fail; + } + if (rc == 1) + continue; /* up to date; nothing to report */ + if (rc != 4) { + log_message(LOG_LEVEL_ERROR, "Unexpected receiver reply during dry-run"); + chunk_destroy(chunk); + goto dry_fail; + } + would = true; + } + if (would) { + if (!config->quiet) { + char* escaped_path = output_escape(file_wire_path(f), config->eight_bit_output); + if (!escaped_path) { + chunk_destroy(chunk); + goto dry_fail; + } + if (config->human_readable) + printf(" %s (%s)\n", escaped_path, + display_bytes(fsize, true, size_buffer, sizeof(size_buffer))); + else + printf(" %s (%llu bytes)\n", escaped_path, fsize); + free(escaped_path); + } + total_bytes += fsize; + file_count++; + } + } + chunk_destroy(chunk); + } + bool io_error = directory_scanner_had_io_error(scanner); + if (directory_scanner_failed(scanner)) + goto dry_fail; + if (io_error) + log_message(LOG_LEVEL_WARNING, "source scan hit an unreadable directory"); + /* Send the keep-set manifest (no data frames) so the receiver can enumerate + the destination extras; an early-timing delete ACKs before it will accept + the terminal FINISHED. */ + bool early_delete = config->use_delete && config_delete_timing_early(config); + if (dry_manifest) { + if (send_delete_manifest(client->file_descriptor, dry_manifest, dry_excluded, dry_size_skipped, + NULL, dry_dirs) != 0) + goto dry_fail; + if (early_delete) { + Status ack; + if (!receive_status_keepalive(client->file_descriptor, &ack, DELETE_ACK_TIMEOUT_SEC, + DELETE_ACK_KEEPALIVE_SEC, client_abort_pending) || + ack != STATUS_OK) + goto dry_fail; + } + } + /* Terminate the stream so the receiver emits its success frame; no data frame + is ever sent in dry-run. */ + if (!send_status(client->file_descriptor, STATUS_FINISHED)) + goto dry_fail; + Status status; + if (!receive_status(client->file_descriptor, &status)) + goto dry_fail; + if (status == STATUS_STATS) { + ArrayList* would_delete = array_list_create(free); + if (!would_delete) + goto dry_fail; + if (!receive_stats_record(client->file_descriptor, &dry_stats, would_delete)) { + array_list_delete(would_delete); + goto dry_fail; + } + print_delete_reports(config, would_delete); + array_list_delete(would_delete); + if (!receive_status(client->file_descriptor, &status)) + goto dry_fail; + } + if (status != STATUS_OK) + goto dry_fail; + if (!config->quiet) { + if (config->human_readable) + printf("Total: %d files, %s\n", file_count, + display_bytes(total_bytes, true, size_buffer, sizeof(size_buffer))); + else + printf("Total: %d files, %.1f MB\n", file_count, (double)total_bytes / (double)BYTES_PER_MIB); + } + { + TransferStats dry_transfer; + memset(&dry_transfer, 0, sizeof(dry_transfer)); + dry_transfer.flist_reg = (unsigned long long)file_count; + dry_transfer.total_file_size = total_bytes; + dry_transfer.transferred_regular = (unsigned long long)file_count; + dry_transfer.transferred_file_size = total_bytes; + dry_transfer.literal_data = total_bytes; + report_transfer_stats(config, &dry_transfer, dry_start, &dry_stats); + } + ret = io_error ? 1 : 0; + +dry_fail: + if (dry_manifest) + array_list_delete(dry_manifest); + if (dry_dirs) + array_list_delete(dry_dirs); + if (dry_excluded) + array_list_delete(dry_excluded); + if (dry_size_skipped) + array_list_delete(dry_size_skipped); + if (scanner) + directory_scanner_destroy(scanner); + prepared_scanner_destroy(&prepared); + disconnect_transfer_client(client); + protocol_session_unbind(); + client_set_abort_armed(false); + return ret; +} diff --git a/src/client/client_report.c b/src/client/client_report.c new file mode 100644 index 0000000..08050df --- /dev/null +++ b/src/client/client_report.c @@ -0,0 +1,704 @@ +#include "client_send_internal.h" +#include "array_list.h" +#include "change_list.h" +#include "charset.h" +#include "config.h" +#include "file.h" +#include "format.h" +#include "log.h" +#include "protocol.h" +#include "utils.h" +#include +#include +#include +#include +#include + +/* Surface a server rejection to the user. When the last status exchange + carried a STATUS_ERROR_DETAIL reason (protocol 2.21.0) it is appended to the + client-side context; a bare STATUS_ERROR still logs the context alone. */ +void log_server_rejection(const char* context) { + const char* detail = protocol_last_error(); + if (detail && detail[0] != '\0') { + /* The detail is peer-controlled: escape it so terminal/log-format + * metacharacters cannot be injected into the client's output. */ + char* escaped = output_escape(detail, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "%s: %s", context, escaped ? escaped : ""); + free(escaped); + } else { + log_message(LOG_LEVEL_ERROR, "%s", context); + } +} + +const char* display_bytes(unsigned long long bytes, bool human_readable, char* buffer, + size_t buffer_size) { + if (human_readable && format_human_size_decimal(bytes, buffer, buffer_size)) + return buffer; + snprintf(buffer, buffer_size, "%.1f MB", (double)bytes / (double)BYTES_PER_MIB); + return buffer; +} + +/* rsync byte count: human-readable decimal when -h was given, otherwise a + * comma-grouped integer (rsync's big_num in the C locale). */ +static const char* stats_bytes(const Config* config, unsigned long long bytes, char* buffer, + size_t buffer_size) { + if (!format_big_num(bytes, config->human_readable, buffer, buffer_size)) + snprintf(buffer, buffer_size, "%llu", bytes); + return buffer; +} + +/* Build rsync's per-type parenthetical: each non-zero category, in + reg/dir/link/special order. Empty when every count is zero. */ +static void type_breakdown(unsigned long long reg, unsigned long long dir, unsigned long long link, + unsigned long long special, char* out, size_t out_size) { + if (reg + dir + link + special == 0) { + out[0] = '\0'; + return; + } + out[0] = '\0'; + size_t used = 0; + const struct { + const char* name; + unsigned long long count; + } parts[4] = {{"reg", reg}, {"dir", dir}, {"link", link}, {"special", special}}; + bool first = true; + for (size_t i = 0; i < 4; i++) { + if (parts[i].count == 0) + continue; + int written = snprintf(out + used, out_size - used, "%s%s: %llu", first ? "(" : ", ", + parts[i].name, parts[i].count); + if (written < 0 || (size_t)written >= out_size - used) + break; + used += (size_t)written; + first = false; + } + if (!first && used + 1 < out_size) + out[used++] = ')'; + out[used] = '\0'; +} + +/* Build rsync's `Number of files` parenthetical from the scan's flist counts. */ +static void stats_type_breakdown(const TransferStats* stats, char* out, size_t out_size) { + type_breakdown(stats->flist_reg, stats->flist_dir, stats->flist_link, stats->flist_special, out, + out_size); +} + +/* rsync's `Number of files` counts every directory. A recursive scan that + preserves a directory attribute captures them in `dir_entries`; a `-r` scan + (no -t/-p) captures nothing, so fall back to the scanner's shared counter of + traversed directories that are not already represented by an inline + directory entry. The -d generator counts its explicit directory entries + inline and does not traverse, so it is excluded here. */ +unsigned long long dir_count_for_stats(const Config* config, const ArrayList* dir_entries, + atomic_ullong* counter) { + if (config == NULL || config->dirs || config->list_only) + return 0; + if (dir_metadata_should_capture(config)) + return dir_entries != NULL ? (unsigned long long)dir_entries->size : 0; + return counter != NULL ? (unsigned long long)atomic_load(counter) : 0; +} + +/* Print the rsync `--stats` block on stdout. The source-side flist and + transferred counters come from `stats` (filled while scanning/sending), the + receiver-only counters from the STATUS_STATS frame, and the wire byte totals + from the process-wide protocol counters. The labels, layout and + rate/speedup formulas match rsync 3.4.1. Shared by the single-threaded and + multithreaded send paths. */ +void report_transfer_stats(const Config* config, const TransferStats* stats, time_t start, + const ReceiverStats* recv) { + if (!config->stats || config->quiet) + return; + TransferStats empty = {0}; + if (stats == NULL) + stats = ∅ + ReceiverStats none = {0}; + if (recv == NULL) + recv = &none; + unsigned long long sent = protocol_bytes_written(); + unsigned long long received = protocol_bytes_read(); + /* rsync: bytes_per_sec = (written + read) / (0.5 + (end - start)). */ + double elapsed = difftime(time(NULL), start); + double rate = (double)(sent + received) / (0.5 + elapsed); + char total_buffer[32]; + char transferred_buffer[32]; + char literal_buffer[32]; + char matched_buffer[32]; + char sent_buffer[32]; + char recv_buffer[32]; + char rate_buffer[32] = {0}; + char human_rate[32] = {0}; + const char* total = + stats_bytes(config, stats->total_file_size, total_buffer, sizeof(total_buffer)); + const char* transferred = stats_bytes(config, stats->transferred_file_size, transferred_buffer, + sizeof(transferred_buffer)); + /* Protocol 2.28.0: the receiver reports the bytes it literally stored, which + is exact for a delta transfer (the sender's own literal_data counts each + stored file's whole source size and is only an upper bound). Fall back to + the sender total when the receiver reported no delta/literal accounting + (e.g. a local no-server path). */ + unsigned long long literal_bytes = (recv->literal_bytes != 0 || recv->matched_data != 0) + ? recv->literal_bytes + : stats->literal_data; + const char* literal = stats_bytes(config, literal_bytes, literal_buffer, sizeof(literal_buffer)); + const char* sent_s = stats_bytes(config, sent, sent_buffer, sizeof(sent_buffer)); + const char* recv_s = stats_bytes(config, received, recv_buffer, sizeof(recv_buffer)); + const char* rate_str = rate_buffer; + if (config->human_readable) { + if (!format_human_size_decimal((unsigned long long)rate, human_rate, sizeof(human_rate))) + snprintf(human_rate, sizeof(human_rate), "0"); + rate_str = human_rate; + } else { + snprintf(rate_buffer, sizeof(rate_buffer), "%.2f", rate); + } + double speedup = + (sent + received) > 0 ? (double)stats->total_file_size / (double)(sent + received) : 0.0; + char breakdown[128]; + stats_type_breakdown(stats, breakdown, sizeof(breakdown)); + unsigned long long flist_total = + stats->flist_reg + stats->flist_dir + stats->flist_link + stats->flist_special; + char created_breakdown[128]; + type_breakdown(recv->created_reg, recv->created_dir, recv->created_link, recv->created_special, + created_breakdown, sizeof(created_breakdown)); + unsigned long long created_total = + recv->created_reg + recv->created_dir + recv->created_link + recv->created_special; + printf("\n"); + if (breakdown[0] != '\0') + printf("Number of files: %llu %s\n", flist_total, breakdown); + else + printf("Number of files: %llu\n", flist_total); + /* Protocol 2.28.0: the receiver reports which destination entries it newly + created, split by type, so this line matches rsync exactly. */ + if (created_breakdown[0] != '\0') + printf("Number of created files: %llu %s\n", created_total, created_breakdown); + else + printf("Number of created files: %llu\n", created_total); + printf("Number of deleted files: %llu\n", recv->deleted_files); + printf("Number of regular files transferred: %llu\n", stats->transferred_regular); + printf("Total file size: %s bytes\n", total); + printf("Total transferred file size: %s bytes\n", transferred); + printf("Literal data: %s bytes\n", literal); + const char* matched = + stats_bytes(config, recv->matched_data, matched_buffer, sizeof(matched_buffer)); + printf("Matched data: %s bytes\n", matched); + printf("File list size: 0\n"); + printf("File list generation time: 0.000 seconds\n"); + printf("File list transfer time: 0.000 seconds\n"); + printf("Total bytes sent: %s\n", sent_s); + printf("Total bytes received: %s\n", recv_s); + printf("\n"); + printf("sent %s bytes received %s bytes %s bytes/sec\n", sent_s, recv_s, rate_str); + printf("total size is %s speedup is %.2f%s\n", total, speedup, + config->dry_run ? " (DRY RUN)" : ""); + fflush(stdout); +} + +/* Classify one scanned source entry into the rsync flist counters. Called for + every entry the sender walks, transferred or skipped. Directory entries are + counted here only for the explicit -d/--dirs generator; a recursive scan's + directories are accounted from the scanner's dir_entries list at report time. */ +void transfer_stats_note_entry(TransferStats* stats, const File* file) { + if (stats == NULL || file == NULL) + return; + if (file->is_dir) { + stats->flist_dir++; + return; + } + if (file->is_symlink) { + stats->flist_link++; + stats->total_file_size += file->symlink_target ? strlen(file->symlink_target) : 0; + return; + } + if (file->is_special) { + stats->flist_special++; + return; + } + stats->flist_reg++; + stats->total_file_size += file->data ? file->data->size : 0; +} + +/* Account for a regular file (or a whole-file append) the receiver actually + stored: rsync's transferred-file count and transferred/literal byte totals. + `literal_data` counts the whole source size, which is exact for a whole-file + send but an upper bound for a delta send (the receiver reuses basis blocks + the sender never ships); see TransferStats.literal_data in format.h. */ +void transfer_stats_note_transferred(TransferStats* stats, const File* file) { + if (stats == NULL || file == NULL) + return; + if (file->is_dir || file->is_symlink || file->is_special) + return; + if (file->link_group != 0 && !file->link_first) + return; + unsigned long long size = file->data ? file->data->size : 0; + stats->transferred_regular++; + stats->transferred_file_size += size; + stats->literal_data += size; +} + +/* ---- rsync-style per-file --progress ------------------------------------ + * rsync prints, for each transferred regular file, the file name followed by a + * two-frame progress line: the first at the initial 32 KiB read window (always + * 0.00 kB/s / 0:00:00 on a sub-second transfer) and a final 100% frame carrying + * `(xfr#N, to-chk=X/Y)`. Rates are wall-clock dependent, so only the final + * rate is measured here; the layout matches rsync 3.4.1's progress.c. */ +#define RSYNC_PROGRESS_IO_WINDOW (32ULL * 1024ULL) + +/* Paths-only pre-count of the source file list, built once at transfer start + * when progress output is requested. rsync's `to-chk` denominator is the whole + * file list -- every regular file, directory, symlink and special plus the + * transfer root -- while the streaming scan never emits directories. A + * metadata-only walk (no file reads, no hashing) supplies that total and the + * directory names, so the opt-in pass leaves non-progress runs untouched. */ +typedef struct { + unsigned long long total; + ArrayList* dir_paths; /* owned char* in transfer-relative display form */ +} ProgressPrecount; + +static bool g_progress_active; +static unsigned long long g_progress_xferred; +static unsigned long long g_progress_index; +static unsigned long long g_progress_total; +static struct timespec g_progress_file_start; +static ProgressPrecount g_progress_precount; +static PathIndex g_progress_dir_index; +static bool g_progress_dir_index_valid; +static StrHashSet g_progress_emitted; +static bool g_progress_emitted_valid; +static ArrayList* g_progress_emitted_keys; + +bool progress_requested(const Config* config) { + return config != NULL && !config->quiet && + (config->show_progress || (config->info_level & LOG_INFO_PROGRESS) != 0); +} + +static void progress_precount_dispose(ProgressPrecount* p) { + if (p->dir_paths != NULL) { + array_list_delete(p->dir_paths); + p->dir_paths = NULL; + } + p->total = 0; +} + +void client_progress_cleanup(void) { + if (g_progress_dir_index_valid) { + path_index_free(&g_progress_dir_index); + g_progress_dir_index_valid = false; + } + if (g_progress_emitted_valid) { + str_hash_set_free(&g_progress_emitted); + g_progress_emitted_valid = false; + } + if (g_progress_emitted_keys != NULL) { + array_list_delete(g_progress_emitted_keys); + g_progress_emitted_keys = NULL; + } + progress_precount_dispose(&g_progress_precount); + g_progress_active = false; + g_progress_total = 0; + g_progress_index = 0; + g_progress_xferred = 0; +} + +static void progress_first_frame(unsigned long long size, char* out, size_t out_size) { + char ofs_buf[32]; + unsigned long long ofs = size < RSYNC_PROGRESS_IO_WINDOW ? size : RSYNC_PROGRESS_IO_WINDOW; + if (!format_big_num(ofs, false, ofs_buf, sizeof(ofs_buf))) + snprintf(ofs_buf, sizeof(ofs_buf), "%llu", ofs); + int pct = size == 0 ? 100 : (ofs == size ? 100 : (int)(100.0 * (double)ofs / (double)size)); + snprintf(out, out_size, "\r%15s %3d%% %7.2f%s %s%s", ofs_buf, pct, 0.0, "kB/s", " 0:00:00", + " "); +} + +static void progress_final_frame(unsigned long long size, char* out, size_t out_size) { + char ofs_buf[32]; + char rembuf[32]; + unsigned long long last_ofs = size < RSYNC_PROGRESS_IO_WINDOW ? size : RSYNC_PROGRESS_IO_WINDOW; + if (!format_big_num(size, false, ofs_buf, sizeof(ofs_buf))) + snprintf(ofs_buf, sizeof(ofs_buf), "%llu", size); + struct timespec now; + clock_gettime(CLOCK_MONOTONIC, &now); + long long diff_ms = (long long)(now.tv_sec - g_progress_file_start.tv_sec) * 1000 + + (now.tv_nsec - g_progress_file_start.tv_nsec) / 1000000; + if (diff_ms <= 0) + diff_ms = 1; + double rate = + size > last_ofs ? (double)(size - last_ofs) * 1000.0 / (double)diff_ms / 1024.0 : 0.0; + const char* units = "kB/s"; + if (rate > 1024.0 * 1024.0) { + rate /= 1024.0 * 1024.0; + units = "GB/s"; + } else if (rate > 1024.0) { + rate /= 1024.0; + units = "MB/s"; + } + unsigned long long remain = (unsigned long long)(diff_ms / 1000); + snprintf(rembuf, sizeof(rembuf), "%4u:%02u:%02u", (unsigned)(remain / 3600), + (unsigned)((remain / 60) % 60), (unsigned)(remain % 60)); + /* rsync's `to-chk` denominator is the whole file list (the pre-count); the + numerator falls as each entry is processed, root first. Without a + pre-count (the paths-only walk failed) fall back to the transferred-file + count so the single-file layout stays intact. */ + unsigned long long total = g_progress_total > 0 ? g_progress_total : g_progress_xferred + 1; + unsigned long long to_chk = total > g_progress_index ? total - g_progress_index - 1 : 0; + snprintf(out, out_size, "\r%15s %3d%% %7.2f%s %s (xfr#%llu, to-chk=%llu/%llu)\n", ofs_buf, 100, + rate, units, rembuf, g_progress_xferred, to_chk, total); +} + +bool info_flag_enabled(const Config* config, LogInfoFlag flag) { + return config != NULL && (config->info_level & flag) != 0; +} + +/* Print rsync's deletion lines for a received list of destination-relative + * paths: `*deleting PATH` when itemizing, the --out-format expansion when a + * format is set, else `deleting PATH` for --info=del. Used by both the dry-run + * would-delete report and the real --info=del report. */ +void print_delete_reports(const Config* config, const ArrayList* paths) { + if (!config || !paths || config->quiet) + return; + /* --debug=del is independent of the --info=del/itemize/out-format display: + emit the debug trace even when no deletion line would be printed. */ + if (log_debug_enabled(LOG_DEBUG_DEL)) { + for (int i = 0; i < paths->size; i++) { + const char* raw = (const char*)paths->items[i]; + const char* path = delete_display_path(config, raw); + log_debug_message(LOG_DEBUG_DEL, "del: %s", path ? path : raw); + } + } + if (!(config->itemize_changes || config->out_format != NULL || + info_flag_enabled(config, LOG_INFO_DEL))) + return; + for (int i = 0; i < paths->size; i++) { + const char* raw = (const char*)paths->items[i]; + const char* path = delete_display_path(config, raw); + if (config->out_format != NULL) { + ChangeEvent event; + memset(&event, 0, sizeof(event)); + event.decision = CHANGE_SENT; + event.deleted = true; + event.name = path; + event.path = path; + char* line = change_render_format(config->out_format, config, &event); + if (line) { + char* escaped = output_escape(line, config->eight_bit_output); + printf("%s\n", escaped ? escaped : line); + free(escaped); + free(line); + } + } else { + char* escaped = output_escape(path, config->eight_bit_output); + if (config->itemize_changes) + printf("*deleting %s\n", escaped ? escaped : path); + else + printf("deleting %s\n", escaped ? escaped : path); + free(escaped); + } + } + fflush(stdout); +} + +static void client_progress_emit_ancestors(const Config* config, const char* rel) { + if (!g_progress_dir_index_valid || !g_progress_emitted_valid || g_progress_emitted_keys == NULL || + rel == NULL) + return; + size_t rel_len = strlen(rel); + for (size_t i = 0; i < rel_len; i++) { + if (rel[i] != '/') + continue; + char* prefix = malloc(i + 1); + if (prefix == NULL) + return; + memcpy(prefix, rel, i); + prefix[i] = '\0'; + if (path_index_contains(&g_progress_dir_index, prefix) && + !str_hash_set_lookup(&g_progress_emitted, prefix)) { + char* key = str_dup(prefix); + if (key != NULL && array_list_add(g_progress_emitted_keys, key)) { + str_hash_set_insert_ref(&g_progress_emitted, key); + char* escaped = output_escape(prefix, config->eight_bit_output); + printf("%s/\n", escaped ? escaped : prefix); + free(escaped); + g_progress_index++; + } else { + free(key); + } + } + free(prefix); + } +} + +/* rsync's --info=name/progress line for one entry: transfer-relative name (a + * trailing slash for directories) plus the ` -> target` symlink suffix. */ +static char* progress_entry_line(const File* file, const char* rel) { + const char* arrow = NULL; + const char* target = NULL; + if (file->is_symlink && file->symlink_target != NULL) { + arrow = " -> "; + target = file->symlink_target; + } else if (file->link_group != 0 && !file->link_first && file->hardlink_target != NULL) { + arrow = " => "; + target = file->hardlink_target; + } + size_t rel_len = strlen(rel); + bool dir_slash = file->is_dir && (rel_len == 0 || rel[rel_len - 1] != '/'); + size_t extra = (dir_slash ? 1u : 0u) + (target != NULL ? 4u + strlen(target) : 0u); + char* line = malloc(rel_len + extra + 1); + if (line == NULL) + return NULL; + memcpy(line, rel, rel_len); + size_t off = rel_len; + if (dir_slash) + line[off++] = '/'; + if (target != NULL) { + memcpy(line + off, arrow, 4); + off += 4; + memcpy(line + off, target, strlen(target)); + off += strlen(target); + } + line[off] = '\0'; + return line; +} + +void client_progress_begin(const Config* config) { + change_reset_name_root(); + g_progress_active = progress_requested(config); + g_progress_xferred = 0; + g_progress_index = 1; /* the transfer root is file-list entry #0 */ + if (!g_progress_active) { + /* `--info=flist` prints rsync's file-list header even without progress. */ + if (!config->quiet && info_flag_enabled(config, LOG_INFO_FLIST)) { + printf("sending incremental file list\n"); + fflush(stdout); + } + return; + } + printf("sending incremental file list\n"); + /* rsync prints the transfer-root directory's name before the first file when + that directory is created; FastSync mirrors the source root below the + receive root and creates it on a fresh destination, so emit it here. */ + printf("./\n"); + fflush(stdout); +} + +/* Emit the name (unless itemize/out-format already did) and the two progress + * frames for one transferred regular file. */ +void client_progress_file(const Config* config, const File* file) { + if (!g_progress_active || file == NULL || !file->data) + return; + g_progress_xferred++; + unsigned long long size = file->data->size; + if (!config->itemize_changes && config->out_format == NULL) { + const char* rel = delete_display_path(config, file_wire_path(file)); + client_progress_emit_ancestors(config, rel); + char* escaped = output_escape(rel, config->eight_bit_output); + printf("%s\n", escaped ? escaped : (rel ? rel : "")); + free(escaped); + } + clock_gettime(CLOCK_MONOTONIC, &g_progress_file_start); + char frame[160]; + progress_first_frame(size, frame, sizeof(frame)); + fputs(frame, stdout); + progress_final_frame(size, frame, sizeof(frame)); + fputs(frame, stdout); + g_progress_index++; + fflush(stdout); +} + +/* Emit the name line for a transferred non-regular entry (directory, symlink, + * special or hard-link sibling): rsync prints these in the file list but has no + * progress frame for them. */ +void client_progress_name(const Config* config, const File* file) { + if (!g_progress_active || file == NULL) + return; + const char* rel = delete_display_path(config, file_wire_path(file)); + if (!config->itemize_changes && config->out_format == NULL) { + client_progress_emit_ancestors(config, rel); + char* line = progress_entry_line(file, rel ? rel : ""); + if (line != NULL) { + char* escaped = output_escape(line, config->eight_bit_output); + printf("%s\n", escaped ? escaped : line); + free(escaped); + free(line); + fflush(stdout); + } + } + g_progress_index++; +} + +/* An entry the receiver already had prints no name under --progress but still + * occupies a file-list slot in the `to-chk` numerator. */ +void client_progress_uptodate(const Config* config, const File* file) { + (void)config; + (void)file; + if (!g_progress_active) + return; + g_progress_index++; +} + +static bool progress_precount_add_dir(ProgressPrecount* p, const char* path) { + if (path == NULL || path[0] == '\0') + return true; + char* dup = str_dup(path); + if (dup == NULL) + return false; + if (array_list_add(p->dir_paths, dup)) + return true; + free(dup); + return false; +} + +/* Metadata-only walk collecting the full file-list total and every directory + * name. It uses its own scanner (fresh filter compilation and hard-link table) + * so the data pass's link-group state is never perturbed. */ +static bool progress_precount_scan(const Config* config, ProgressPrecount* out) { + out->dir_paths = array_list_create(free); + if (out->dir_paths == NULL) + return false; + out->total = 0; + PreparedScanner prepared; + memset(&prepared, 0, sizeof(prepared)); + if (!prepare_scanner(config, 0, &prepared)) { + progress_precount_dispose(out); + return false; + } + ScannerOptions local = prepared.options; + local.list_dirs = true; + local.note_nonreg = false; + local.note_mount = false; + local.dir_count = NULL; + local.use_metadata = false; + local.preserve_xattrs = false; + local.preserve_acls = false; + local.checksum = false; + local.capture_dir_times = false; + local.excluded_paths = NULL; + local.size_skipped_paths = NULL; + local.synced_dirs = NULL; + local.plan_dirs = NULL; + local.dir_entries = NULL; + local.dir_entries_mutex = NULL; + local.hardlinks = NULL; + DirectoryScanner* scanner = directory_scanner_create_with_options(config->send_directory, &local); + bool ok = scanner != NULL; + if (scanner != NULL) { + Chunk* chunk; + while (ok && (chunk = directory_scanner_next(scanner)) != NULL) { + out->total += (unsigned long long)chunk->element_count; + for (int i = 0; i < chunk->element_count && ok; i++) { + const File* f = chunk->items[i]; + if (f != NULL && f->is_dir) + ok = progress_precount_add_dir(out, delete_display_path(config, file_wire_path(f))); + } + chunk_destroy(chunk); + } + if (ok && directory_scanner_failed(scanner)) + ok = false; + directory_scanner_destroy(scanner); + } + prepared_scanner_destroy(&prepared); + if (!ok) { + progress_precount_dispose(out); + return false; + } + out->total += 1; /* the transfer root "." */ + return true; +} + +/* Reuse the --delete-during/--delete-delay keep-set pre-scan: its traversed + * directory list already holds every directory and `non_dir_count` the entries + * counted during that same pass, so progress costs no second walk. */ +static bool progress_precount_from_plan_dirs(const Config* config, const ArrayList* plan_dirs, + unsigned long long non_dir_count, + ProgressPrecount* out) { + out->dir_paths = array_list_create(free); + if (out->dir_paths == NULL) + return false; + out->total = non_dir_count + 1; + for (int i = 0; i < plan_dirs->size; i++) { + const char* path = (const char*)plan_dirs->items[i]; + const char* rel = config->send_directory != NULL + ? utils_strip_transfer_root(path, config->send_directory) + : path; + if (!progress_precount_add_dir(out, rel)) { + progress_precount_dispose(out); + return false; + } + } + out->total += (unsigned long long)out->dir_paths->size; + return true; +} + +/* Build the optional progress pre-count. A failed pre-count is non-fatal: the + * transfer proceeds and the progress denominator falls back to the transferred + * file count. */ +void client_progress_prepare(const Config* config, const ArrayList* plan_dirs, + unsigned long long plan_non_dir_count) { + client_progress_cleanup(); + g_progress_active = progress_requested(config); + if (!g_progress_active) + return; + bool ok = plan_dirs != NULL ? progress_precount_from_plan_dirs( + config, plan_dirs, plan_non_dir_count, &g_progress_precount) + : progress_precount_scan(config, &g_progress_precount); + if (!ok) { + g_progress_total = 0; + return; + } + g_progress_total = g_progress_precount.total; + if (g_progress_precount.dir_paths != NULL && g_progress_precount.dir_paths->size > 0 && + path_index_build(&g_progress_dir_index, + (const char* const*)g_progress_precount.dir_paths->items, + (size_t)g_progress_precount.dir_paths->size)) + g_progress_dir_index_valid = true; + if (str_hash_set_init(&g_progress_emitted, (size_t)(g_progress_precount.dir_paths != NULL + ? g_progress_precount.dir_paths->size + 1 + : 1))) + g_progress_emitted_valid = true; + g_progress_emitted_keys = array_list_create(free); +} + +/* Read the optional STATUS_STATS record (protocol 2.25.0) that the receiver + * sends just before its terminal status when report_stats was negotiated. + * Consumes the would-delete path list into `would_delete` (optional). */ +bool receive_stats_record(int fd, ReceiverStats* stats, ArrayList* would_delete) { + if (!format_stats_receive(fd, stats)) + return false; + int count = 0; + if (!receive_int(fd, &count) || count < 0 || count > MAX_MANIFEST_ENTRIES) + return false; + /* Mirror the delete-plan parser: every retained path must be a valid + destination-relative path, and the whole list shares one MAX_MANIFEST_BYTES + budget so a hostile peer cannot make the client retain unbounded memory. */ + size_t bytes = 0; + for (int i = 0; i < count; i++) { + char* path = receive_wire_str(fd); + if (!path) + return false; + if (path[0] == '\0' || path[0] == '/' || has_path_traversal(path)) { + free(path); + return false; + } + if (would_delete) { + size_t entry_size = strlen(path) + sizeof(char*) + 16; + if (entry_size > MAX_MANIFEST_BYTES - bytes) { + free(path); + return false; + } + bytes += entry_size; + if (!array_list_add(would_delete, path)) { + free(path); + return false; + } + } else { + free(path); + } + } + return true; +} + +/* Strip the transfer-root prefix from a receiver-reported destination-relative + * delete path so a `*deleting` line matches rsync's transfer-relative name + * (FastSync's destination mirror includes the source's absolute path). */ +const char* delete_display_path(const Config* config, const char* path) { + if (!config || !path || !config->send_directory) + return path; + return utils_strip_transfer_root(path, config->send_directory); +} diff --git a/src/client/client_scan.c b/src/client/client_scan.c new file mode 100644 index 0000000..0a01530 --- /dev/null +++ b/src/client/client_scan.c @@ -0,0 +1,454 @@ +#include "client_send_internal.h" +#include "array_list.h" +#include "charset.h" +#include "config.h" +#include "delete_plan.h" +#include "file.h" +#include "file_list.h" +#include "filter.h" +#include "hardlink.h" +#include "log.h" +#include "scanner.h" +#include "utils.h" +#include +#include +#include +#include + +/* Build the scanner options for one scan. Returns false and logs on failure. */ +bool prepare_scanner(const Config* config, int num_threads, PreparedScanner* out) { + if (!out) + return false; + out->base_filters = NULL; + out->hardlinks = NULL; + out->relative_prefix = NULL; + memset(&out->options, 0, sizeof(out->options)); + + int rule_count = config->filters ? config->filters->size : 0; + const char** texts = NULL; + if (rule_count > 0) { + texts = malloc((size_t)rule_count * sizeof(char*)); + if (!texts) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed for filter rules"); + return false; + } + for (int i = 0; i < rule_count; i++) + texts[i] = (const char*)config->filters->items[i]; + } + if (rule_count > 0 || config->cvs_exclude) { + char err[160]; + out->base_filters = filter_base_build(texts, rule_count, config->cvs_exclude, + config->delete_excluded, err, sizeof(err)); + free(texts); + if (!out->base_filters) { + log_message(LOG_LEVEL_ERROR, "invalid filter rule: %s", err); + return false; + } + } else { + free(texts); + } + + ScannerOptions* options = &out->options; + options->use_metadata = config->use_metadata; + options->preserve_atimes = config->preserve_atimes; + options->preserve_crtimes = config->preserve_crtimes; + options->preserve_xattrs = config->preserve_xattrs; + options->preserve_acls = config->preserve_acls; + options->chunk_size = config->chunk_size; + /* --exclude/--include are compiled, in command-line order, into the SAME + * ordered filter rule list as --filter/-f (see config_add_selection_rule), so + * the legacy per-kind arrays are deliberately NOT passed to the scanner: + * doing so would re-apply them with the old "excludes first, then includes as + * a mandatory whitelist" precedence and defeat rsync's first-match-wins + * ordering. The arrays remain populated purely for the Config API surface. */ + options->exclude_patterns = NULL; + options->exclude_count = 0; + options->include_patterns = NULL; + options->include_count = 0; + options->max_size = config->max_size; + options->min_size = config->min_size; + options->max_depth = config->max_depth; + options->num_threads = num_threads; + options->follow_symlinks = config->follow_symlinks; + options->copy_links = config->copy_links; + options->safe_links = config->safe_links; + options->copy_unsafe_links = config->copy_unsafe_links; + options->copy_dirlinks = config->copy_dirlinks; + options->munge_links = config->munge_links; + options->checksum = config->checksum; + options->one_file_system = config->one_file_system; + options->preserve_devices = config->preserve_devices; + options->preserve_specials = config->preserve_specials; + options->copy_devices = config->copy_devices; + options->file_list = (const FileListSet*)config->files_from_set; + options->base_filters = out->base_filters; + options->per_dir_filters = config->per_dir_filter; + options->delete_excluded = config->delete_excluded; + options->exclude_per_dir_filter_files = config->per_dir_filter_count >= 2; + options->dirs = config->dirs; + options->relative = config->relative; + /* A real recursive transfer recreates empty source directories (rsync + parity); low-level scanner users leave this off. */ + options->emit_empty_dirs = true; + /* --no-implied-dirs only has meaning with -R (rsync): without it the option + is a documented no-op, so the scanner must not suppress directory + metadata. */ + options->no_implied_dirs = config->no_implied_dirs && config->relative; + /* -R/--relative outside --files-from reconstructs every destination path from + * the source spec (rsync's '/./' cut point). With --files-from the listed + * entry already supplies the bare relative path, so no prefix is built. */ + if (config->relative && config->files_from_set == NULL && config->send_directory) { + out->relative_prefix = scanner_relative_prefix(config->send_directory); + if (!out->relative_prefix) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed building --relative path prefix"); + filter_rule_list_free(out->base_filters); + out->base_filters = NULL; + return false; + } + options->relative_prefix = out->relative_prefix; + } + options->prune_empty_dirs = config->prune_empty_dirs; + options->ignore_io_errors = config->ignore_errors; + options->ignore_missing_args = config->ignore_missing_args || config->delete_missing_args; + options->note_nonreg = (config->info_level & LOG_INFO_NONREG) != 0 && !config->quiet; + options->note_mount = (config->info_level & LOG_INFO_MOUNT) != 0 && !config->quiet; + options->send_directory = config->send_directory; + options->eight_bit_output = config->eight_bit_output; + options->excluded_paths = NULL; + options->excluded_mutex = NULL; + options->size_skipped_paths = NULL; + options->synced_dirs = NULL; + options->hardlinks = NULL; + /* Set by the real send paths; NULL for the metadata-only scans (progress + pre-count, batch) that must not perturb the sender's --stats counter. */ + options->dir_count = NULL; + /* P7 Wave D: capture source directory metadata when a directory attribute is + requested (-p for modes, -t for times unless -O omits them). Whether they + are APPLIED is decided receiver-side. */ + options->capture_dir_times = dir_metadata_should_capture(config); + options->dir_entries = NULL; + options->dir_entries_mutex = NULL; + if (config->preserve_hard_links) { + out->hardlinks = hardlink_table_create(); + if (!out->hardlinks) { + filter_rule_list_free(out->base_filters); + out->base_filters = NULL; + return false; + } + options->hardlinks = out->hardlinks; + } + return true; +} + +void prepared_scanner_destroy(PreparedScanner* prepared) { + if (!prepared) + return; + filter_rule_list_free(prepared->base_filters); + prepared->base_filters = NULL; + hardlink_table_destroy(prepared->hardlinks); + prepared->hardlinks = NULL; + free(prepared->relative_prefix); + prepared->relative_prefix = NULL; +} + +/* -R/--relative implied directories: rsync transmits the metadata of the + * parent directories implied by the source path (every prefix component above + * the source root) so the receiver applies their attributes to the created + * parents. FastSync's scan only covers the source root and below, so append + * one metadata-only directory entry per implied ancestor. --no-implied-dirs + * suppresses this exactly like rsync. A missing ancestor is never fatal. */ +bool append_implied_dir_times(const Config* config, ArrayList* dir_entries) { + if (!dir_entries || !config->relative || config->files_from_set != NULL || + config->no_implied_dirs || !config->send_directory) + return true; + char* prefix = scanner_relative_prefix(config->send_directory); + if (!prefix) + return true; + int ncomp = 0; + for (const char* s = prefix; *s;) { + while (*s == '/') + s++; + if (!*s) + break; + while (*s && *s != '/') + s++; + ncomp++; + } + if (ncomp <= 1) { + free(prefix); + return true; + } + char* fs = str_dup(config->send_directory); + if (!fs) { + free(prefix); + return true; + } + size_t flen = strlen(fs); + while (flen > 1 && fs[flen - 1] == '/') + fs[--flen] = '\0'; + bool ok = true; + /* Walk the source path upwards one component at a time (fs is truncated in + place, so each step targets the next implied ancestor). */ + for (int depth = ncomp - 2; depth >= 0 && ok; depth--) { + char* slash = strrchr(fs, '/'); + if (!slash || slash == fs) + break; + *slash = '\0'; + char* p = prefix; + int c = 0; + while (c <= depth) { + while (*p == '/') + p++; + while (*p && *p != '/') + p++; + c++; + } + char saved = *p; + *p = '\0'; + struct stat st; + if (stat(fs, &st) == 0 && S_ISDIR(st.st_mode)) { + File* file = file_create(fs); + if (!file) { + ok = false; + } else { + file->is_dir = true; + file->metadata = + file_metadata_create(fs, &st, config->preserve_atimes, config->preserve_crtimes); + file->send_path = str_dup(prefix); + if (!file->metadata || !file->send_path || !array_list_add(dir_entries, file)) { + file_destroy(file); + ok = false; + } + } + } + *p = saved; + } + free(fs); + free(prefix); + return ok; +} + +/* The delete-walk root scope for a full (non---files-from) transfer: rsync + * confines --delete to the directories it actually transferred. A plain + * recursive run mirrors the source under the receive root, so "." (the whole + * tree) is correct; an -R run transfers only the reconstructed prefix subtree, + * so the walk is scoped to that prefix instead. Returns a malloc'd wire path + * (or "."), or NULL on allocation failure. */ +char* delete_scope_root_marker(const Config* config) { + if (config->relative && config->files_from_set == NULL && config->send_directory) { + char* prefix = scanner_relative_prefix(config->send_directory); + if (!prefix) + return NULL; + if (prefix[0] != '\0') + return prefix; + free(prefix); + } + return str_dup("."); +} + +/* The -R destination prefix that confines a per-directory delete walk, or NULL + * when the whole receive root is in scope. The marker was installed into + * `synced_dirs` by delete_scope_root_marker(); for a plain recursive transfer + * it is "." (whole root) and for --files-from the list is not a single prefix. */ +const char* delete_plan_walk_root(const Config* config, const ArrayList* synced_dirs) { + if (!config || config->files_from_set != NULL || !config->relative || !config->send_directory) + return NULL; + if (!synced_dirs || synced_dirs->size != 1) + return NULL; + const char* marker = (const char*)synced_dirs->items[0]; + if (marker[0] == '\0' || strcmp(marker, ".") == 0) + return NULL; + return marker; +} + +/* The destination-relative mirror path for a missing --files-from entry: where + a PRESENT entry with the same name would have been written. With -R that is + the entry's bare relative path (the bare wire path the receiver uses); + otherwise it is the full source mirror below the destination root + (`send_directory` joined to the entry, leading '/' stripped), exactly the + path the manifest records for a present sibling. Returns an owned string, or + NULL on allocation failure. */ +static char* files_from_missing_dest_path(const Config* config, const char* entry) { + if (config->relative) + return str_dup(entry); + char* joined = path_cat(config->send_directory, entry); + if (!joined) + return NULL; + const char* rel = *joined == '/' ? joined + 1 : joined; + char* dup = str_dup(rel); + free(joined); + return dup; +} + +/* --files-from semantics: every listed entry must resolve under the source + * root, otherwise rsync reports a hard error instead of silently transferring + * nothing. An entry of "." (the whole tree) and listed-but-empty directories + * are valid. An empty list is valid too: rsync transfers nothing and exits 0. + * With --ignore-missing-args + * (implied by --delete-missing-args) a listed-but-missing entry is instead + * skipped: nothing is transferred for it, it never enters the keep-set and the + * run succeeds for the rest (an all-missing non-empty list succeeds + * transferring nothing, matching rsync). With --delete-missing-args + * `missing_dest` (when non-NULL) collects the entry's destination-relative + * mirror for the receiver's exact-deletion request. Runs before any + * transfer so the failure/skip is surfaced uniformly in the single-threaded, + * -m, dry-run and --list-only paths. */ +bool files_from_list_check(const Config* config, ArrayList* missing_dest, int* skipped_out) { + *skipped_out = 0; + const FileListSet* set = (const FileListSet*)config->files_from_set; + if (!set) + return true; + if (!config->send_directory) { + log_message(LOG_LEVEL_ERROR, "--files-from requires a source directory"); + return false; + } + if (set->count == 0) { + /* rsync treats an empty --files-from list as "nothing to transfer" and + exits 0 (the source directory is still a valid source arg), so this is + not an error. Nothing passes the (empty) allow-set, so no file is sent + and no keep-set entry is produced. */ + return true; + } + bool ignore = config->ignore_missing_args || config->delete_missing_args; + for (int i = 0; i < set->count; i++) { + const char* entry = set->entries[i]; + if (entry[0] == '\0') + continue; /* "." == list the whole tree */ + char* full = path_cat(config->send_directory, entry); + if (!full) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed while validating --files-from"); + return false; + } + struct stat st; + if (lstat(full, &st) != 0) { + free(full); + if (ignore) { + (*skipped_out)++; + char* escaped_entry = output_escape(entry, log_get_8_bit_output()); + log_info_message(LOG_INFO_MISC, "skipping missing --files-from entry '%s'", + escaped_entry ? escaped_entry : ""); + free(escaped_entry); + if (config->delete_missing_args && missing_dest) { + char* mirror = files_from_missing_dest_path(config, entry); + if (!mirror || !array_list_add(missing_dest, mirror)) { + free(mirror); + log_message(LOG_LEVEL_ERROR, "memory allocation failed while validating --files-from"); + return false; + } + } + continue; + } + char* escaped_entry = output_escape(entry, log_get_8_bit_output()); + char* escaped_src = output_escape(config->send_directory, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "--files-from entry '%s' not found in source '%s'", + escaped_entry ? escaped_entry : "", + escaped_src ? escaped_src : ""); + free(escaped_entry); + free(escaped_src); + return false; + } + free(full); + } + if (*skipped_out > 0) { + if (config->delete_missing_args) { + /* --list-only never deletes and a --dry-run only shows intent, so the + summary must not claim a real deletion happened in those modes. */ + if (config->list_only) + log_message(LOG_LEVEL_WARNING, + "--delete-missing-args: %d missing --files-from entr%s skipped (--list-only " + "never deletes)", + *skipped_out, *skipped_out == 1 ? "y" : "ies"); + else if (config->dry_run) + log_message(LOG_LEVEL_WARNING, + "--delete-missing-args: %d missing --files-from entr%s would be deleted from " + "the destination (dry run)", + *skipped_out, *skipped_out == 1 ? "y" : "ies"); + else + log_message( + LOG_LEVEL_WARNING, + "--delete-missing-args: %d missing --files-from entr%s will be deleted from the " + "destination", + *skipped_out, *skipped_out == 1 ? "y" : "ies"); + } else if (config->ignore_missing_args) + log_message(LOG_LEVEL_WARNING, + "--ignore-missing-args: ignored %d missing --files-from entr%s", *skipped_out, + *skipped_out == 1 ? "y" : "ies"); + } + return true; +} + +/* Walk the whole source tree once collecting only destination-relative wire + paths, loading and sending nothing. --delete-before/--delete-during need the + complete keep-set manifest before the first data byte, so it is built by a + dedicated pre-scan pass and transmitted early; the data pass then re-scans + with a fresh scanner. A source I/O error is fatal unless the options carry + --ignore-errors, in which case the scan continues past the unreadable + directory and *io_error_out reports it (the caller still performs the + deletion but reports the run as errored). */ +bool scan_paths_only(const Config* config, const ScannerOptions* options, ArrayList* manifest, + DeletePlanSender* plans, bool* io_error_out, + unsigned long long* non_dir_count_out) { + if (io_error_out) + *io_error_out = false; + if (non_dir_count_out) + *non_dir_count_out = 0; + ScannerOptions local = *options; + /* The pre-scan is a paths-only pass with no client output; it must not emit + --info=nonreg lines (the data pass does that once). */ + local.note_nonreg = false; + DirectoryScanner* scanner = directory_scanner_create_with_options(config->send_directory, &local); + if (!scanner) + return false; + bool ok = true; + Chunk* chunk; + while ((chunk = directory_scanner_next(scanner)) != NULL) { + if (non_dir_count_out) { + for (int i = 0; i < chunk->element_count; i++) { + const File* f = chunk->items[i]; + if (f && !f->is_dir) + (*non_dir_count_out)++; + } + } + if (manifest && !add_chunk_to_manifest(manifest, chunk)) { + ok = false; + chunk_destroy(chunk); + break; + } + if (plans) { + for (int i = 0; i < chunk->element_count; i++) { + File* f = chunk->items[i]; + if (!f) + continue; + const char* path = file_wire_path(f); + if (!delete_plan_sender_add(plans, path, f->is_dir)) { + ok = false; + break; + } + } + if (!ok) { + chunk_destroy(chunk); + break; + } + } + chunk_destroy(chunk); + } + if (ok) { + /* Keep every traversed source directory, including empty ones, so a plan + no longer removes the destination directory itself. Their own plans are + emitted after the data stream (no file frame triggers them). */ + if (plans && options->plan_dirs) { + for (int i = 0; i < options->plan_dirs->size; i++) { + if (!delete_plan_sender_add(plans, (const char*)options->plan_dirs->items[i], true)) { + ok = false; + break; + } + } + } + } + if (ok && directory_scanner_failed(scanner)) + ok = false; + if (io_error_out) + *io_error_out = directory_scanner_had_io_error(scanner); + directory_scanner_destroy(scanner); + return ok; +} diff --git a/src/client/client_send.c b/src/client/client_send.c index e5c71f4..3b77503 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -1,4 +1,5 @@ #include "client_send.h" +#include "client_send_internal.h" #include "array_list.h" #include "batch.h" #include "change_list.h" @@ -44,26 +45,6 @@ receiver's RECEIVER_QUEUE_MAX_BYTES). */ #define SENDER_QUEUE_MAX_BYTES (MAX_CONNECTION_MEMORY - 2 * MAX_CHUNK_SIZE) -/* One mebibyte in bytes; the unit used by the --stats/--progress lines. - Always cast to double when dividing so the output stays fractional. */ -#define BYTES_PER_MIB (1024ULL * 1024ULL) - -/* Surface a server rejection to the user. When the last status exchange - carried a STATUS_ERROR_DETAIL reason (protocol 2.21.0) it is appended to the - client-side context; a bare STATUS_ERROR still logs the context alone. */ -static void log_server_rejection(const char* context) { - const char* detail = protocol_last_error(); - if (detail && detail[0] != '\0') { - /* The detail is peer-controlled: escape it so terminal/log-format - * metacharacters cannot be injected into the client's output. */ - char* escaped = output_escape(detail, log_get_8_bit_output()); - log_message(LOG_LEVEL_ERROR, "%s: %s", context, escaped ? escaped : ""); - free(escaped); - } else { - log_message(LOG_LEVEL_ERROR, "%s", context); - } -} - /* rsync's --ignore-errors semantics: an I/O error during the transfer normally * suppresses deletion entirely ("IO error encountered -- skipping file * deletion"); --ignore-errors lets the deletion run anyway. FastSync always @@ -74,1007 +55,6 @@ bool ignore_errors_allows_delete(const Config* config, bool had_io_error) { return !had_io_error || (config && config->ignore_errors); } -static const char* display_bytes(unsigned long long bytes, bool human_readable, char* buffer, - size_t buffer_size) { - if (human_readable && format_human_size_decimal(bytes, buffer, buffer_size)) - return buffer; - snprintf(buffer, buffer_size, "%.1f MB", (double)bytes / (double)BYTES_PER_MIB); - return buffer; -} - -/* rsync byte count: human-readable decimal when -h was given, otherwise a - * comma-grouped integer (rsync's big_num in the C locale). */ -static const char* stats_bytes(const Config* config, unsigned long long bytes, char* buffer, - size_t buffer_size) { - if (!format_big_num(bytes, config->human_readable, buffer, buffer_size)) - snprintf(buffer, buffer_size, "%llu", bytes); - return buffer; -} - -/* Build rsync's per-type parenthetical: each non-zero category, in - reg/dir/link/special order. Empty when every count is zero. */ -static void type_breakdown(unsigned long long reg, unsigned long long dir, unsigned long long link, - unsigned long long special, char* out, size_t out_size) { - if (reg + dir + link + special == 0) { - out[0] = '\0'; - return; - } - out[0] = '\0'; - size_t used = 0; - const struct { - const char* name; - unsigned long long count; - } parts[4] = {{"reg", reg}, {"dir", dir}, {"link", link}, {"special", special}}; - bool first = true; - for (size_t i = 0; i < 4; i++) { - if (parts[i].count == 0) - continue; - int written = snprintf(out + used, out_size - used, "%s%s: %llu", first ? "(" : ", ", - parts[i].name, parts[i].count); - if (written < 0 || (size_t)written >= out_size - used) - break; - used += (size_t)written; - first = false; - } - if (!first && used + 1 < out_size) - out[used++] = ')'; - out[used] = '\0'; -} - -/* Build rsync's `Number of files` parenthetical from the scan's flist counts. */ -static void stats_type_breakdown(const TransferStats* stats, char* out, size_t out_size) { - type_breakdown(stats->flist_reg, stats->flist_dir, stats->flist_link, stats->flist_special, out, - out_size); -} - -/* rsync's `Number of files` counts every directory. A recursive scan that - preserves a directory attribute captures them in `dir_entries`; a `-r` scan - (no -t/-p) captures nothing, so fall back to the scanner's shared counter of - traversed directories that are not already represented by an inline - directory entry. The -d generator counts its explicit directory entries - inline and does not traverse, so it is excluded here. */ -static unsigned long long dir_count_for_stats(const Config* config, const ArrayList* dir_entries, - atomic_ullong* counter) { - if (config == NULL || config->dirs || config->list_only) - return 0; - if (dir_metadata_should_capture(config)) - return dir_entries != NULL ? (unsigned long long)dir_entries->size : 0; - return counter != NULL ? (unsigned long long)atomic_load(counter) : 0; -} - -/* Print the rsync `--stats` block on stdout. The source-side flist and - transferred counters come from `stats` (filled while scanning/sending), the - receiver-only counters from the STATUS_STATS frame, and the wire byte totals - from the process-wide protocol counters. The labels, layout and - rate/speedup formulas match rsync 3.4.1. Shared by the single-threaded and - multithreaded send paths. */ -static void report_transfer_stats(const Config* config, const TransferStats* stats, time_t start, - const ReceiverStats* recv) { - if (!config->stats || config->quiet) - return; - TransferStats empty = {0}; - if (stats == NULL) - stats = ∅ - ReceiverStats none = {0}; - if (recv == NULL) - recv = &none; - unsigned long long sent = protocol_bytes_written(); - unsigned long long received = protocol_bytes_read(); - /* rsync: bytes_per_sec = (written + read) / (0.5 + (end - start)). */ - double elapsed = difftime(time(NULL), start); - double rate = (double)(sent + received) / (0.5 + elapsed); - char total_buffer[32]; - char transferred_buffer[32]; - char literal_buffer[32]; - char matched_buffer[32]; - char sent_buffer[32]; - char recv_buffer[32]; - char rate_buffer[32] = {0}; - char human_rate[32] = {0}; - const char* total = - stats_bytes(config, stats->total_file_size, total_buffer, sizeof(total_buffer)); - const char* transferred = stats_bytes(config, stats->transferred_file_size, transferred_buffer, - sizeof(transferred_buffer)); - /* Protocol 2.28.0: the receiver reports the bytes it literally stored, which - is exact for a delta transfer (the sender's own literal_data counts each - stored file's whole source size and is only an upper bound). Fall back to - the sender total when the receiver reported no delta/literal accounting - (e.g. a local no-server path). */ - unsigned long long literal_bytes = (recv->literal_bytes != 0 || recv->matched_data != 0) - ? recv->literal_bytes - : stats->literal_data; - const char* literal = stats_bytes(config, literal_bytes, literal_buffer, sizeof(literal_buffer)); - const char* sent_s = stats_bytes(config, sent, sent_buffer, sizeof(sent_buffer)); - const char* recv_s = stats_bytes(config, received, recv_buffer, sizeof(recv_buffer)); - const char* rate_str = rate_buffer; - if (config->human_readable) { - if (!format_human_size_decimal((unsigned long long)rate, human_rate, sizeof(human_rate))) - snprintf(human_rate, sizeof(human_rate), "0"); - rate_str = human_rate; - } else { - snprintf(rate_buffer, sizeof(rate_buffer), "%.2f", rate); - } - double speedup = - (sent + received) > 0 ? (double)stats->total_file_size / (double)(sent + received) : 0.0; - char breakdown[128]; - stats_type_breakdown(stats, breakdown, sizeof(breakdown)); - unsigned long long flist_total = - stats->flist_reg + stats->flist_dir + stats->flist_link + stats->flist_special; - char created_breakdown[128]; - type_breakdown(recv->created_reg, recv->created_dir, recv->created_link, recv->created_special, - created_breakdown, sizeof(created_breakdown)); - unsigned long long created_total = - recv->created_reg + recv->created_dir + recv->created_link + recv->created_special; - printf("\n"); - if (breakdown[0] != '\0') - printf("Number of files: %llu %s\n", flist_total, breakdown); - else - printf("Number of files: %llu\n", flist_total); - /* Protocol 2.28.0: the receiver reports which destination entries it newly - created, split by type, so this line matches rsync exactly. */ - if (created_breakdown[0] != '\0') - printf("Number of created files: %llu %s\n", created_total, created_breakdown); - else - printf("Number of created files: %llu\n", created_total); - printf("Number of deleted files: %llu\n", recv->deleted_files); - printf("Number of regular files transferred: %llu\n", stats->transferred_regular); - printf("Total file size: %s bytes\n", total); - printf("Total transferred file size: %s bytes\n", transferred); - printf("Literal data: %s bytes\n", literal); - const char* matched = - stats_bytes(config, recv->matched_data, matched_buffer, sizeof(matched_buffer)); - printf("Matched data: %s bytes\n", matched); - printf("File list size: 0\n"); - printf("File list generation time: 0.000 seconds\n"); - printf("File list transfer time: 0.000 seconds\n"); - printf("Total bytes sent: %s\n", sent_s); - printf("Total bytes received: %s\n", recv_s); - printf("\n"); - printf("sent %s bytes received %s bytes %s bytes/sec\n", sent_s, recv_s, rate_str); - printf("total size is %s speedup is %.2f%s\n", total, speedup, - config->dry_run ? " (DRY RUN)" : ""); - fflush(stdout); -} - -/* Classify one scanned source entry into the rsync flist counters. Called for - every entry the sender walks, transferred or skipped. Directory entries are - counted here only for the explicit -d/--dirs generator; a recursive scan's - directories are accounted from the scanner's dir_entries list at report time. */ -static void transfer_stats_note_entry(TransferStats* stats, const File* file) { - if (stats == NULL || file == NULL) - return; - if (file->is_dir) { - stats->flist_dir++; - return; - } - if (file->is_symlink) { - stats->flist_link++; - stats->total_file_size += file->symlink_target ? strlen(file->symlink_target) : 0; - return; - } - if (file->is_special) { - stats->flist_special++; - return; - } - stats->flist_reg++; - stats->total_file_size += file->data ? file->data->size : 0; -} - -/* Account for a regular file (or a whole-file append) the receiver actually - stored: rsync's transferred-file count and transferred/literal byte totals. - `literal_data` counts the whole source size, which is exact for a whole-file - send but an upper bound for a delta send (the receiver reuses basis blocks - the sender never ships); see TransferStats.literal_data in format.h. */ -static void transfer_stats_note_transferred(TransferStats* stats, const File* file) { - if (stats == NULL || file == NULL) - return; - if (file->is_dir || file->is_symlink || file->is_special) - return; - if (file->link_group != 0 && !file->link_first) - return; - unsigned long long size = file->data ? file->data->size : 0; - stats->transferred_regular++; - stats->transferred_file_size += size; - stats->literal_data += size; -} - -/* ---- rsync-style per-file --progress ------------------------------------ - * rsync prints, for each transferred regular file, the file name followed by a - * two-frame progress line: the first at the initial 32 KiB read window (always - * 0.00 kB/s / 0:00:00 on a sub-second transfer) and a final 100% frame carrying - * `(xfr#N, to-chk=X/Y)`. Rates are wall-clock dependent, so only the final - * rate is measured here; the layout matches rsync 3.4.1's progress.c. */ -#define RSYNC_PROGRESS_IO_WINDOW (32ULL * 1024ULL) - -static const char* delete_display_path(const Config* config, const char* path); - -/* Paths-only pre-count of the source file list, built once at transfer start - * when progress output is requested. rsync's `to-chk` denominator is the whole - * file list -- every regular file, directory, symlink and special plus the - * transfer root -- while the streaming scan never emits directories. A - * metadata-only walk (no file reads, no hashing) supplies that total and the - * directory names, so the opt-in pass leaves non-progress runs untouched. */ -typedef struct { - unsigned long long total; - ArrayList* dir_paths; /* owned char* in transfer-relative display form */ -} ProgressPrecount; - -static bool g_progress_active; -static unsigned long long g_progress_xferred; -static unsigned long long g_progress_index; -static unsigned long long g_progress_total; -static struct timespec g_progress_file_start; -static ProgressPrecount g_progress_precount; -static PathIndex g_progress_dir_index; -static bool g_progress_dir_index_valid; -static StrHashSet g_progress_emitted; -static bool g_progress_emitted_valid; -static ArrayList* g_progress_emitted_keys; - -static bool progress_requested(const Config* config) { - return config != NULL && !config->quiet && - (config->show_progress || (config->info_level & LOG_INFO_PROGRESS) != 0); -} - -static void progress_precount_dispose(ProgressPrecount* p) { - if (p->dir_paths != NULL) { - array_list_delete(p->dir_paths); - p->dir_paths = NULL; - } - p->total = 0; -} - -static void client_progress_cleanup(void) { - if (g_progress_dir_index_valid) { - path_index_free(&g_progress_dir_index); - g_progress_dir_index_valid = false; - } - if (g_progress_emitted_valid) { - str_hash_set_free(&g_progress_emitted); - g_progress_emitted_valid = false; - } - if (g_progress_emitted_keys != NULL) { - array_list_delete(g_progress_emitted_keys); - g_progress_emitted_keys = NULL; - } - progress_precount_dispose(&g_progress_precount); - g_progress_active = false; - g_progress_total = 0; - g_progress_index = 0; - g_progress_xferred = 0; -} - -static void progress_first_frame(unsigned long long size, char* out, size_t out_size) { - char ofs_buf[32]; - unsigned long long ofs = size < RSYNC_PROGRESS_IO_WINDOW ? size : RSYNC_PROGRESS_IO_WINDOW; - if (!format_big_num(ofs, false, ofs_buf, sizeof(ofs_buf))) - snprintf(ofs_buf, sizeof(ofs_buf), "%llu", ofs); - int pct = size == 0 ? 100 : (ofs == size ? 100 : (int)(100.0 * (double)ofs / (double)size)); - snprintf(out, out_size, "\r%15s %3d%% %7.2f%s %s%s", ofs_buf, pct, 0.0, "kB/s", " 0:00:00", - " "); -} - -static void progress_final_frame(unsigned long long size, char* out, size_t out_size) { - char ofs_buf[32]; - char rembuf[32]; - unsigned long long last_ofs = size < RSYNC_PROGRESS_IO_WINDOW ? size : RSYNC_PROGRESS_IO_WINDOW; - if (!format_big_num(size, false, ofs_buf, sizeof(ofs_buf))) - snprintf(ofs_buf, sizeof(ofs_buf), "%llu", size); - struct timespec now; - clock_gettime(CLOCK_MONOTONIC, &now); - long long diff_ms = (long long)(now.tv_sec - g_progress_file_start.tv_sec) * 1000 + - (now.tv_nsec - g_progress_file_start.tv_nsec) / 1000000; - if (diff_ms <= 0) - diff_ms = 1; - double rate = - size > last_ofs ? (double)(size - last_ofs) * 1000.0 / (double)diff_ms / 1024.0 : 0.0; - const char* units = "kB/s"; - if (rate > 1024.0 * 1024.0) { - rate /= 1024.0 * 1024.0; - units = "GB/s"; - } else if (rate > 1024.0) { - rate /= 1024.0; - units = "MB/s"; - } - unsigned long long remain = (unsigned long long)(diff_ms / 1000); - snprintf(rembuf, sizeof(rembuf), "%4u:%02u:%02u", (unsigned)(remain / 3600), - (unsigned)((remain / 60) % 60), (unsigned)(remain % 60)); - /* rsync's `to-chk` denominator is the whole file list (the pre-count); the - numerator falls as each entry is processed, root first. Without a - pre-count (the paths-only walk failed) fall back to the transferred-file - count so the single-file layout stays intact. */ - unsigned long long total = g_progress_total > 0 ? g_progress_total : g_progress_xferred + 1; - unsigned long long to_chk = total > g_progress_index ? total - g_progress_index - 1 : 0; - snprintf(out, out_size, "\r%15s %3d%% %7.2f%s %s (xfr#%llu, to-chk=%llu/%llu)\n", ofs_buf, 100, - rate, units, rembuf, g_progress_xferred, to_chk, total); -} - -static bool info_flag_enabled(const Config* config, LogInfoFlag flag) { - return config != NULL && (config->info_level & flag) != 0; -} - -/* Print rsync's deletion lines for a received list of destination-relative - * paths: `*deleting PATH` when itemizing, the --out-format expansion when a - * format is set, else `deleting PATH` for --info=del. Used by both the dry-run - * would-delete report and the real --info=del report. */ -static void print_delete_reports(const Config* config, const ArrayList* paths) { - if (!config || !paths || config->quiet) - return; - /* --debug=del is independent of the --info=del/itemize/out-format display: - emit the debug trace even when no deletion line would be printed. */ - if (log_debug_enabled(LOG_DEBUG_DEL)) { - for (int i = 0; i < paths->size; i++) { - const char* raw = (const char*)paths->items[i]; - const char* path = delete_display_path(config, raw); - log_debug_message(LOG_DEBUG_DEL, "del: %s", path ? path : raw); - } - } - if (!(config->itemize_changes || config->out_format != NULL || - info_flag_enabled(config, LOG_INFO_DEL))) - return; - for (int i = 0; i < paths->size; i++) { - const char* raw = (const char*)paths->items[i]; - const char* path = delete_display_path(config, raw); - if (config->out_format != NULL) { - ChangeEvent event; - memset(&event, 0, sizeof(event)); - event.decision = CHANGE_SENT; - event.deleted = true; - event.name = path; - event.path = path; - char* line = change_render_format(config->out_format, config, &event); - if (line) { - char* escaped = output_escape(line, config->eight_bit_output); - printf("%s\n", escaped ? escaped : line); - free(escaped); - free(line); - } - } else { - char* escaped = output_escape(path, config->eight_bit_output); - if (config->itemize_changes) - printf("*deleting %s\n", escaped ? escaped : path); - else - printf("deleting %s\n", escaped ? escaped : path); - free(escaped); - } - } - fflush(stdout); -} - -static void client_progress_emit_ancestors(const Config* config, const char* rel) { - if (!g_progress_dir_index_valid || !g_progress_emitted_valid || g_progress_emitted_keys == NULL || - rel == NULL) - return; - size_t rel_len = strlen(rel); - for (size_t i = 0; i < rel_len; i++) { - if (rel[i] != '/') - continue; - char* prefix = malloc(i + 1); - if (prefix == NULL) - return; - memcpy(prefix, rel, i); - prefix[i] = '\0'; - if (path_index_contains(&g_progress_dir_index, prefix) && - !str_hash_set_lookup(&g_progress_emitted, prefix)) { - char* key = str_dup(prefix); - if (key != NULL && array_list_add(g_progress_emitted_keys, key)) { - str_hash_set_insert_ref(&g_progress_emitted, key); - char* escaped = output_escape(prefix, config->eight_bit_output); - printf("%s/\n", escaped ? escaped : prefix); - free(escaped); - g_progress_index++; - } else { - free(key); - } - } - free(prefix); - } -} - -/* rsync's --info=name/progress line for one entry: transfer-relative name (a - * trailing slash for directories) plus the ` -> target` symlink suffix. */ -static char* progress_entry_line(const File* file, const char* rel) { - const char* arrow = NULL; - const char* target = NULL; - if (file->is_symlink && file->symlink_target != NULL) { - arrow = " -> "; - target = file->symlink_target; - } else if (file->link_group != 0 && !file->link_first && file->hardlink_target != NULL) { - arrow = " => "; - target = file->hardlink_target; - } - size_t rel_len = strlen(rel); - bool dir_slash = file->is_dir && (rel_len == 0 || rel[rel_len - 1] != '/'); - size_t extra = (dir_slash ? 1u : 0u) + (target != NULL ? 4u + strlen(target) : 0u); - char* line = malloc(rel_len + extra + 1); - if (line == NULL) - return NULL; - memcpy(line, rel, rel_len); - size_t off = rel_len; - if (dir_slash) - line[off++] = '/'; - if (target != NULL) { - memcpy(line + off, arrow, 4); - off += 4; - memcpy(line + off, target, strlen(target)); - off += strlen(target); - } - line[off] = '\0'; - return line; -} - -static void client_progress_begin(const Config* config) { - change_reset_name_root(); - g_progress_active = progress_requested(config); - g_progress_xferred = 0; - g_progress_index = 1; /* the transfer root is file-list entry #0 */ - if (!g_progress_active) { - /* `--info=flist` prints rsync's file-list header even without progress. */ - if (!config->quiet && info_flag_enabled(config, LOG_INFO_FLIST)) { - printf("sending incremental file list\n"); - fflush(stdout); - } - return; - } - printf("sending incremental file list\n"); - /* rsync prints the transfer-root directory's name before the first file when - that directory is created; FastSync mirrors the source root below the - receive root and creates it on a fresh destination, so emit it here. */ - printf("./\n"); - fflush(stdout); -} - -/* Emit the name (unless itemize/out-format already did) and the two progress - * frames for one transferred regular file. */ -static void client_progress_file(const Config* config, const File* file) { - if (!g_progress_active || file == NULL || !file->data) - return; - g_progress_xferred++; - unsigned long long size = file->data->size; - if (!config->itemize_changes && config->out_format == NULL) { - const char* rel = delete_display_path(config, file_wire_path(file)); - client_progress_emit_ancestors(config, rel); - char* escaped = output_escape(rel, config->eight_bit_output); - printf("%s\n", escaped ? escaped : (rel ? rel : "")); - free(escaped); - } - clock_gettime(CLOCK_MONOTONIC, &g_progress_file_start); - char frame[160]; - progress_first_frame(size, frame, sizeof(frame)); - fputs(frame, stdout); - progress_final_frame(size, frame, sizeof(frame)); - fputs(frame, stdout); - g_progress_index++; - fflush(stdout); -} - -/* Emit the name line for a transferred non-regular entry (directory, symlink, - * special or hard-link sibling): rsync prints these in the file list but has no - * progress frame for them. */ -static void client_progress_name(const Config* config, const File* file) { - if (!g_progress_active || file == NULL) - return; - const char* rel = delete_display_path(config, file_wire_path(file)); - if (!config->itemize_changes && config->out_format == NULL) { - client_progress_emit_ancestors(config, rel); - char* line = progress_entry_line(file, rel ? rel : ""); - if (line != NULL) { - char* escaped = output_escape(line, config->eight_bit_output); - printf("%s\n", escaped ? escaped : line); - free(escaped); - free(line); - fflush(stdout); - } - } - g_progress_index++; -} - -/* An entry the receiver already had prints no name under --progress but still - * occupies a file-list slot in the `to-chk` numerator. */ -static void client_progress_uptodate(const Config* config, const File* file) { - (void)config; - (void)file; - if (!g_progress_active) - return; - g_progress_index++; -} - -/* Compiled scanner inputs that are shared read-only across scanner instances - * and, in -m mode, across worker threads. `base_filters` owns the compiled - * command-line + -C rules; the FileListSet allow-set lives in the Config. - * `hardlinks` owns the --hard-links/-H link-group detection table (NULL when - * off) and is shared (mutex-guarded) across every scanner/worker of one scan. */ -typedef struct { - ScannerOptions options; - FilterRuleList* base_filters; /* owned; may be NULL */ - HardLinkTable* hardlinks; /* owned; may be NULL */ - char* relative_prefix; /* owned -R prefix; may be NULL */ -} PreparedScanner; - -/* Build the scanner options for one scan. Returns false and logs on failure. */ -static bool prepare_scanner(const Config* config, int num_threads, PreparedScanner* out) { - if (!out) - return false; - out->base_filters = NULL; - out->hardlinks = NULL; - out->relative_prefix = NULL; - memset(&out->options, 0, sizeof(out->options)); - - int rule_count = config->filters ? config->filters->size : 0; - const char** texts = NULL; - if (rule_count > 0) { - texts = malloc((size_t)rule_count * sizeof(char*)); - if (!texts) { - log_message(LOG_LEVEL_ERROR, "memory allocation failed for filter rules"); - return false; - } - for (int i = 0; i < rule_count; i++) - texts[i] = (const char*)config->filters->items[i]; - } - if (rule_count > 0 || config->cvs_exclude) { - char err[160]; - out->base_filters = filter_base_build(texts, rule_count, config->cvs_exclude, - config->delete_excluded, err, sizeof(err)); - free(texts); - if (!out->base_filters) { - log_message(LOG_LEVEL_ERROR, "invalid filter rule: %s", err); - return false; - } - } else { - free(texts); - } - - ScannerOptions* options = &out->options; - options->use_metadata = config->use_metadata; - options->preserve_atimes = config->preserve_atimes; - options->preserve_crtimes = config->preserve_crtimes; - options->preserve_xattrs = config->preserve_xattrs; - options->preserve_acls = config->preserve_acls; - options->chunk_size = config->chunk_size; - /* --exclude/--include are compiled, in command-line order, into the SAME - * ordered filter rule list as --filter/-f (see config_add_selection_rule), so - * the legacy per-kind arrays are deliberately NOT passed to the scanner: - * doing so would re-apply them with the old "excludes first, then includes as - * a mandatory whitelist" precedence and defeat rsync's first-match-wins - * ordering. The arrays remain populated purely for the Config API surface. */ - options->exclude_patterns = NULL; - options->exclude_count = 0; - options->include_patterns = NULL; - options->include_count = 0; - options->max_size = config->max_size; - options->min_size = config->min_size; - options->max_depth = config->max_depth; - options->num_threads = num_threads; - options->follow_symlinks = config->follow_symlinks; - options->copy_links = config->copy_links; - options->safe_links = config->safe_links; - options->copy_unsafe_links = config->copy_unsafe_links; - options->copy_dirlinks = config->copy_dirlinks; - options->munge_links = config->munge_links; - options->checksum = config->checksum; - options->one_file_system = config->one_file_system; - options->preserve_devices = config->preserve_devices; - options->preserve_specials = config->preserve_specials; - options->copy_devices = config->copy_devices; - options->file_list = (const FileListSet*)config->files_from_set; - options->base_filters = out->base_filters; - options->per_dir_filters = config->per_dir_filter; - options->delete_excluded = config->delete_excluded; - options->exclude_per_dir_filter_files = config->per_dir_filter_count >= 2; - options->dirs = config->dirs; - options->relative = config->relative; - /* A real recursive transfer recreates empty source directories (rsync - parity); low-level scanner users leave this off. */ - options->emit_empty_dirs = true; - /* --no-implied-dirs only has meaning with -R (rsync): without it the option - is a documented no-op, so the scanner must not suppress directory - metadata. */ - options->no_implied_dirs = config->no_implied_dirs && config->relative; - /* -R/--relative outside --files-from reconstructs every destination path from - * the source spec (rsync's '/./' cut point). With --files-from the listed - * entry already supplies the bare relative path, so no prefix is built. */ - if (config->relative && config->files_from_set == NULL && config->send_directory) { - out->relative_prefix = scanner_relative_prefix(config->send_directory); - if (!out->relative_prefix) { - log_message(LOG_LEVEL_ERROR, "memory allocation failed building --relative path prefix"); - filter_rule_list_free(out->base_filters); - out->base_filters = NULL; - return false; - } - options->relative_prefix = out->relative_prefix; - } - options->prune_empty_dirs = config->prune_empty_dirs; - options->ignore_io_errors = config->ignore_errors; - options->ignore_missing_args = config->ignore_missing_args || config->delete_missing_args; - options->note_nonreg = (config->info_level & LOG_INFO_NONREG) != 0 && !config->quiet; - options->note_mount = (config->info_level & LOG_INFO_MOUNT) != 0 && !config->quiet; - options->send_directory = config->send_directory; - options->eight_bit_output = config->eight_bit_output; - options->excluded_paths = NULL; - options->excluded_mutex = NULL; - options->size_skipped_paths = NULL; - options->synced_dirs = NULL; - options->hardlinks = NULL; - /* Set by the real send paths; NULL for the metadata-only scans (progress - pre-count, batch) that must not perturb the sender's --stats counter. */ - options->dir_count = NULL; - /* P7 Wave D: capture source directory metadata when a directory attribute is - requested (-p for modes, -t for times unless -O omits them). Whether they - are APPLIED is decided receiver-side. */ - options->capture_dir_times = dir_metadata_should_capture(config); - options->dir_entries = NULL; - options->dir_entries_mutex = NULL; - if (config->preserve_hard_links) { - out->hardlinks = hardlink_table_create(); - if (!out->hardlinks) { - filter_rule_list_free(out->base_filters); - out->base_filters = NULL; - return false; - } - options->hardlinks = out->hardlinks; - } - return true; -} - -static void prepared_scanner_destroy(PreparedScanner* prepared) { - if (!prepared) - return; - filter_rule_list_free(prepared->base_filters); - prepared->base_filters = NULL; - hardlink_table_destroy(prepared->hardlinks); - prepared->hardlinks = NULL; - free(prepared->relative_prefix); - prepared->relative_prefix = NULL; -} - -static bool progress_precount_add_dir(ProgressPrecount* p, const char* path) { - if (path == NULL || path[0] == '\0') - return true; - char* dup = str_dup(path); - if (dup == NULL) - return false; - if (array_list_add(p->dir_paths, dup)) - return true; - free(dup); - return false; -} - -/* Metadata-only walk collecting the full file-list total and every directory - * name. It uses its own scanner (fresh filter compilation and hard-link table) - * so the data pass's link-group state is never perturbed. */ -static bool progress_precount_scan(const Config* config, ProgressPrecount* out) { - out->dir_paths = array_list_create(free); - if (out->dir_paths == NULL) - return false; - out->total = 0; - PreparedScanner prepared; - memset(&prepared, 0, sizeof(prepared)); - if (!prepare_scanner(config, 0, &prepared)) { - progress_precount_dispose(out); - return false; - } - ScannerOptions local = prepared.options; - local.list_dirs = true; - local.note_nonreg = false; - local.note_mount = false; - local.dir_count = NULL; - local.use_metadata = false; - local.preserve_xattrs = false; - local.preserve_acls = false; - local.checksum = false; - local.capture_dir_times = false; - local.excluded_paths = NULL; - local.size_skipped_paths = NULL; - local.synced_dirs = NULL; - local.plan_dirs = NULL; - local.dir_entries = NULL; - local.dir_entries_mutex = NULL; - local.hardlinks = NULL; - DirectoryScanner* scanner = directory_scanner_create_with_options(config->send_directory, &local); - bool ok = scanner != NULL; - if (scanner != NULL) { - Chunk* chunk; - while (ok && (chunk = directory_scanner_next(scanner)) != NULL) { - out->total += (unsigned long long)chunk->element_count; - for (int i = 0; i < chunk->element_count && ok; i++) { - const File* f = chunk->items[i]; - if (f != NULL && f->is_dir) - ok = progress_precount_add_dir(out, delete_display_path(config, file_wire_path(f))); - } - chunk_destroy(chunk); - } - if (ok && directory_scanner_failed(scanner)) - ok = false; - directory_scanner_destroy(scanner); - } - prepared_scanner_destroy(&prepared); - if (!ok) { - progress_precount_dispose(out); - return false; - } - out->total += 1; /* the transfer root "." */ - return true; -} - -/* Reuse the --delete-during/--delete-delay keep-set pre-scan: its traversed - * directory list already holds every directory and `non_dir_count` the entries - * counted during that same pass, so progress costs no second walk. */ -static bool progress_precount_from_plan_dirs(const Config* config, const ArrayList* plan_dirs, - unsigned long long non_dir_count, - ProgressPrecount* out) { - out->dir_paths = array_list_create(free); - if (out->dir_paths == NULL) - return false; - out->total = non_dir_count + 1; - for (int i = 0; i < plan_dirs->size; i++) { - const char* path = (const char*)plan_dirs->items[i]; - const char* rel = config->send_directory != NULL - ? utils_strip_transfer_root(path, config->send_directory) - : path; - if (!progress_precount_add_dir(out, rel)) { - progress_precount_dispose(out); - return false; - } - } - out->total += (unsigned long long)out->dir_paths->size; - return true; -} - -/* Build the optional progress pre-count. A failed pre-count is non-fatal: the - * transfer proceeds and the progress denominator falls back to the transferred - * file count. */ -static void client_progress_prepare(const Config* config, const ArrayList* plan_dirs, - unsigned long long plan_non_dir_count) { - client_progress_cleanup(); - g_progress_active = progress_requested(config); - if (!g_progress_active) - return; - bool ok = plan_dirs != NULL ? progress_precount_from_plan_dirs( - config, plan_dirs, plan_non_dir_count, &g_progress_precount) - : progress_precount_scan(config, &g_progress_precount); - if (!ok) { - g_progress_total = 0; - return; - } - g_progress_total = g_progress_precount.total; - if (g_progress_precount.dir_paths != NULL && g_progress_precount.dir_paths->size > 0 && - path_index_build(&g_progress_dir_index, - (const char* const*)g_progress_precount.dir_paths->items, - (size_t)g_progress_precount.dir_paths->size)) - g_progress_dir_index_valid = true; - if (str_hash_set_init(&g_progress_emitted, (size_t)(g_progress_precount.dir_paths != NULL - ? g_progress_precount.dir_paths->size + 1 - : 1))) - g_progress_emitted_valid = true; - g_progress_emitted_keys = array_list_create(free); -} - -/* -R/--relative implied directories: rsync transmits the metadata of the - * parent directories implied by the source path (every prefix component above - * the source root) so the receiver applies their attributes to the created - * parents. FastSync's scan only covers the source root and below, so append - * one metadata-only directory entry per implied ancestor. --no-implied-dirs - * suppresses this exactly like rsync. A missing ancestor is never fatal. */ -static bool append_implied_dir_times(const Config* config, ArrayList* dir_entries) { - if (!dir_entries || !config->relative || config->files_from_set != NULL || - config->no_implied_dirs || !config->send_directory) - return true; - char* prefix = scanner_relative_prefix(config->send_directory); - if (!prefix) - return true; - int ncomp = 0; - for (const char* s = prefix; *s;) { - while (*s == '/') - s++; - if (!*s) - break; - while (*s && *s != '/') - s++; - ncomp++; - } - if (ncomp <= 1) { - free(prefix); - return true; - } - char* fs = str_dup(config->send_directory); - if (!fs) { - free(prefix); - return true; - } - size_t flen = strlen(fs); - while (flen > 1 && fs[flen - 1] == '/') - fs[--flen] = '\0'; - bool ok = true; - /* Walk the source path upwards one component at a time (fs is truncated in - place, so each step targets the next implied ancestor). */ - for (int depth = ncomp - 2; depth >= 0 && ok; depth--) { - char* slash = strrchr(fs, '/'); - if (!slash || slash == fs) - break; - *slash = '\0'; - char* p = prefix; - int c = 0; - while (c <= depth) { - while (*p == '/') - p++; - while (*p && *p != '/') - p++; - c++; - } - char saved = *p; - *p = '\0'; - struct stat st; - if (stat(fs, &st) == 0 && S_ISDIR(st.st_mode)) { - File* file = file_create(fs); - if (!file) { - ok = false; - } else { - file->is_dir = true; - file->metadata = - file_metadata_create(fs, &st, config->preserve_atimes, config->preserve_crtimes); - file->send_path = str_dup(prefix); - if (!file->metadata || !file->send_path || !array_list_add(dir_entries, file)) { - file_destroy(file); - ok = false; - } - } - } - *p = saved; - } - free(fs); - free(prefix); - return ok; -} - -/* The delete-walk root scope for a full (non---files-from) transfer: rsync - * confines --delete to the directories it actually transferred. A plain - * recursive run mirrors the source under the receive root, so "." (the whole - * tree) is correct; an -R run transfers only the reconstructed prefix subtree, - * so the walk is scoped to that prefix instead. Returns a malloc'd wire path - * (or "."), or NULL on allocation failure. */ -static char* delete_scope_root_marker(const Config* config) { - if (config->relative && config->files_from_set == NULL && config->send_directory) { - char* prefix = scanner_relative_prefix(config->send_directory); - if (!prefix) - return NULL; - if (prefix[0] != '\0') - return prefix; - free(prefix); - } - return str_dup("."); -} - -/* The -R destination prefix that confines a per-directory delete walk, or NULL - * when the whole receive root is in scope. The marker was installed into - * `synced_dirs` by delete_scope_root_marker(); for a plain recursive transfer - * it is "." (whole root) and for --files-from the list is not a single prefix. */ -static const char* delete_plan_walk_root(const Config* config, const ArrayList* synced_dirs) { - if (!config || config->files_from_set != NULL || !config->relative || !config->send_directory) - return NULL; - if (!synced_dirs || synced_dirs->size != 1) - return NULL; - const char* marker = (const char*)synced_dirs->items[0]; - if (marker[0] == '\0' || strcmp(marker, ".") == 0) - return NULL; - return marker; -} - -/* The destination-relative mirror path for a missing --files-from entry: where - a PRESENT entry with the same name would have been written. With -R that is - the entry's bare relative path (the bare wire path the receiver uses); - otherwise it is the full source mirror below the destination root - (`send_directory` joined to the entry, leading '/' stripped), exactly the - path the manifest records for a present sibling. Returns an owned string, or - NULL on allocation failure. */ -static char* files_from_missing_dest_path(const Config* config, const char* entry) { - if (config->relative) - return str_dup(entry); - char* joined = path_cat(config->send_directory, entry); - if (!joined) - return NULL; - const char* rel = *joined == '/' ? joined + 1 : joined; - char* dup = str_dup(rel); - free(joined); - return dup; -} - -/* --files-from semantics: every listed entry must resolve under the source - * root, otherwise rsync reports a hard error instead of silently transferring - * nothing. An entry of "." (the whole tree) and listed-but-empty directories - * are valid. An empty list is valid too: rsync transfers nothing and exits 0. - * With --ignore-missing-args - * (implied by --delete-missing-args) a listed-but-missing entry is instead - * skipped: nothing is transferred for it, it never enters the keep-set and the - * run succeeds for the rest (an all-missing non-empty list succeeds - * transferring nothing, matching rsync). With --delete-missing-args - * `missing_dest` (when non-NULL) collects the entry's destination-relative - * mirror for the receiver's exact-deletion request. Runs before any - * transfer so the failure/skip is surfaced uniformly in the single-threaded, - * -m, dry-run and --list-only paths. */ -static bool files_from_list_check(const Config* config, ArrayList* missing_dest, int* skipped_out) { - *skipped_out = 0; - const FileListSet* set = (const FileListSet*)config->files_from_set; - if (!set) - return true; - if (!config->send_directory) { - log_message(LOG_LEVEL_ERROR, "--files-from requires a source directory"); - return false; - } - if (set->count == 0) { - /* rsync treats an empty --files-from list as "nothing to transfer" and - exits 0 (the source directory is still a valid source arg), so this is - not an error. Nothing passes the (empty) allow-set, so no file is sent - and no keep-set entry is produced. */ - return true; - } - bool ignore = config->ignore_missing_args || config->delete_missing_args; - for (int i = 0; i < set->count; i++) { - const char* entry = set->entries[i]; - if (entry[0] == '\0') - continue; /* "." == list the whole tree */ - char* full = path_cat(config->send_directory, entry); - if (!full) { - log_message(LOG_LEVEL_ERROR, "memory allocation failed while validating --files-from"); - return false; - } - struct stat st; - if (lstat(full, &st) != 0) { - free(full); - if (ignore) { - (*skipped_out)++; - char* escaped_entry = output_escape(entry, log_get_8_bit_output()); - log_info_message(LOG_INFO_MISC, "skipping missing --files-from entry '%s'", - escaped_entry ? escaped_entry : ""); - free(escaped_entry); - if (config->delete_missing_args && missing_dest) { - char* mirror = files_from_missing_dest_path(config, entry); - if (!mirror || !array_list_add(missing_dest, mirror)) { - free(mirror); - log_message(LOG_LEVEL_ERROR, "memory allocation failed while validating --files-from"); - return false; - } - } - continue; - } - char* escaped_entry = output_escape(entry, log_get_8_bit_output()); - char* escaped_src = output_escape(config->send_directory, log_get_8_bit_output()); - log_message(LOG_LEVEL_ERROR, "--files-from entry '%s' not found in source '%s'", - escaped_entry ? escaped_entry : "", - escaped_src ? escaped_src : ""); - free(escaped_entry); - free(escaped_src); - return false; - } - free(full); - } - if (*skipped_out > 0) { - if (config->delete_missing_args) { - /* --list-only never deletes and a --dry-run only shows intent, so the - summary must not claim a real deletion happened in those modes. */ - if (config->list_only) - log_message(LOG_LEVEL_WARNING, - "--delete-missing-args: %d missing --files-from entr%s skipped (--list-only " - "never deletes)", - *skipped_out, *skipped_out == 1 ? "y" : "ies"); - else if (config->dry_run) - log_message(LOG_LEVEL_WARNING, - "--delete-missing-args: %d missing --files-from entr%s would be deleted from " - "the destination (dry run)", - *skipped_out, *skipped_out == 1 ? "y" : "ies"); - else - log_message( - LOG_LEVEL_WARNING, - "--delete-missing-args: %d missing --files-from entr%s will be deleted from the " - "destination", - *skipped_out, *skipped_out == 1 ? "y" : "ies"); - } else if (config->ignore_missing_args) - log_message(LOG_LEVEL_WARNING, - "--ignore-missing-args: ignored %d missing --files-from entr%s", *skipped_out, - *skipped_out == 1 ? "y" : "ies"); - } - return true; -} - /* Read the daemon's MOTD frame and, unless --no-motd, display it on stdout. * * The daemon sends the MOTD as the first thing after the config-frame STATUS_OK @@ -1085,7 +65,7 @@ static bool files_from_list_check(const Config* config, ArrayList* missing_dest, * has no MOTD frame. The text is rendered through motd_render so a hostile * server cannot inject terminal escape sequences. A read failure is not fatal * here: the transfer that follows surfaces the real connection error. */ -static void receive_daemon_motd(Client* client, const Config* config) { +void receive_daemon_motd(Client* client, const Config* config) { if (!config->module || config->module[0] == '\0') return; char* motd = motd_receive(client->file_descriptor); @@ -1106,7 +86,7 @@ static void receive_daemon_motd(Client* client, const Config* config) { } /* Select the configured transport for both transfer execution paths. */ -static Client* connect_transfer_client(const Config* config) { +Client* connect_transfer_client(const Config* config) { if (config->transport == TRANSPORT_SSH) { if (config->use_sendfile) { log_message(LOG_LEVEL_ERROR, "--sendfile is not supported with SSH transport"); @@ -1144,55 +124,13 @@ static Client* connect_transfer_client(const Config* config) { return client; } -static void disconnect_transfer_client(Client* client) { +void disconnect_transfer_client(Client* client) { if (!client) return; client_disconnect(client); client_delete(client); } -/* True when --dry-run should contact a receiver rather than running the - * client-side local manifest. Any target a real run would reach over the wire - * selects the server-contacting path: a remote (SSH host:path), a daemon - * (host::module/path), an explicit --server-host, --server-port/--port, TLS, or - * a source-bind --address. A plain local destination (none of these) keeps the - * original client-side behavior, which never dials the default 127.0.0.1:8080. */ -static bool dry_run_targets_server(const Config* config) { - if (!config) - return false; - if (config->transport == TRANSPORT_SSH) - return true; - if (config->module && config->module[0] != '\0') - return true; - if (config->server_host_set || config->server_port_set) - return true; - if (config->use_tls) - return true; - if (config->address != NULL) - return true; - return false; -} - -static bool add_chunk_to_manifest(ArrayList* manifest, const Chunk* chunk) { - if (!manifest) - return true; - for (int i = 0; i < chunk->element_count; i++) { - const char* path = file_wire_path(chunk->items[i]); - if (*path == '/') - path++; - char* entry = str_dup(path); - if (!entry) { - log_message(LOG_LEVEL_ERROR, "Failed to allocate manifest entry"); - return false; - } - if (!array_list_add(manifest, entry)) { - free(entry); - return false; - } - } - return true; -} - /* (finalize_transfer is defined after the SourceFile helpers below.) */ typedef struct SourceFile { @@ -1298,54 +236,6 @@ static void mark_sender_done(PipelineContextSender* context) { mtx_unlock(&context->mutex_progress); } -/* Read the optional STATUS_STATS record (protocol 2.25.0) that the receiver - * sends just before its terminal status when report_stats was negotiated. - * Consumes the would-delete path list into `would_delete` (optional). */ -static bool receive_stats_record(int fd, ReceiverStats* stats, ArrayList* would_delete) { - if (!format_stats_receive(fd, stats)) - return false; - int count = 0; - if (!receive_int(fd, &count) || count < 0 || count > MAX_MANIFEST_ENTRIES) - return false; - /* Mirror the delete-plan parser: every retained path must be a valid - destination-relative path, and the whole list shares one MAX_MANIFEST_BYTES - budget so a hostile peer cannot make the client retain unbounded memory. */ - size_t bytes = 0; - for (int i = 0; i < count; i++) { - char* path = receive_wire_str(fd); - if (!path) - return false; - if (path[0] == '\0' || path[0] == '/' || has_path_traversal(path)) { - free(path); - return false; - } - if (would_delete) { - size_t entry_size = strlen(path) + sizeof(char*) + 16; - if (entry_size > MAX_MANIFEST_BYTES - bytes) { - free(path); - return false; - } - bytes += entry_size; - if (!array_list_add(would_delete, path)) { - free(path); - return false; - } - } else { - free(path); - } - } - return true; -} - -/* Strip the transfer-root prefix from a receiver-reported destination-relative - * delete path so a `*deleting` line matches rsync's transfer-relative name - * (FastSync's destination mirror includes the source's absolute path). */ -static const char* delete_display_path(const Config* config, const char* path) { - if (!config || !path || !config->send_directory) - return path; - return utils_strip_transfer_root(path, config->send_directory); -} - /* Send the final STATUS_FINISHED frame and await the receiver's verdict. When --remove-source-files is active the receiver acknowledges each data file it processed, in send order: STATUS_NEXT means the file was written, @@ -1429,463 +319,8 @@ static void pipeline_cancel(PipelineContextSender* context) { mtx_unlock(&context->mutex_scanner); } -/* Print dry-run manifest showing files that would be transferred. Returns 0 on success. */ -static int send_dry_run_manifest(const Config* config) { - int skipped = 0; - ArrayList* missing_dest = NULL; - if (config->delete_missing_args) { - missing_dest = array_list_create(free); - if (!missing_dest) - return -1; - } - if (!files_from_list_check(config, missing_dest, &skipped)) { - if (missing_dest) - array_list_delete(missing_dest); - return -1; - } - PreparedScanner prepared; - if (!prepare_scanner(config, 0, &prepared)) { - if (missing_dest) - array_list_delete(missing_dest); - return -1; - } - DirectoryScanner* scanner = - directory_scanner_create_with_options(config->send_directory, &prepared.options); - if (!scanner) { - prepared_scanner_destroy(&prepared); - if (missing_dest) - array_list_delete(missing_dest); - return -1; - } - Chunk* chunk; - int file_count = 0; - unsigned long long total_bytes = 0; - char size_buffer[32]; - if (!config->quiet) - printf("Dry run: files to be transferred\n"); - while ((chunk = directory_scanner_next(scanner)) != NULL) { - for (int i = 0; i < chunk->element_count; i++) { - if (!config->quiet) { - char* escaped_path = - output_escape(file_wire_path(chunk->items[i]), config->eight_bit_output); - if (!escaped_path) { - chunk_destroy(chunk); - directory_scanner_destroy(scanner); - prepared_scanner_destroy(&prepared); - if (missing_dest) - array_list_delete(missing_dest); - return -1; - } - if (config->human_readable) - printf( - " %s (%s)\n", escaped_path, - display_bytes(chunk->items[i]->data->size, true, size_buffer, sizeof(size_buffer))); - else - printf(" %s (%zu bytes)\n", escaped_path, chunk->items[i]->data->size); - free(escaped_path); - } - total_bytes += chunk->items[i]->data->size; - file_count++; - } - chunk_destroy(chunk); - } - directory_scanner_destroy(scanner); - prepared_scanner_destroy(&prepared); - /* --delete-missing-args: the missing entries' destination mirrors render as - would-be deletions (rsync's dry-run also lists its *deleting lines). */ - if (missing_dest && !config->quiet) { - for (int i = 0; i < missing_dest->size; i++) { - char* escaped = output_escape((char*)missing_dest->items[i], config->eight_bit_output); - printf(" %s (missing; would be deleted)\n", escaped ? escaped : ""); - free(escaped); - } - } - if (missing_dest) - array_list_delete(missing_dest); - if (!config->quiet) { - if (config->human_readable) - printf("Total: %d files, %s\n", file_count, - display_bytes(total_bytes, true, size_buffer, sizeof(size_buffer))); - else - printf("Total: %d files, %.1f MB\n", file_count, (double)total_bytes / (double)BYTES_PER_MIB); - } - return 0; -} - -typedef struct { - char* name; /* transfer-relative name ("" == the source root) */ - mode_t mode; - unsigned long long size; - time_t mtime; - long mtime_nsec; - bool is_dir; - bool is_symlink; - char* link_target; -} ListEntry; - -static void list_entries_destroy(ListEntry* entries, size_t count) { - if (entries == NULL) - return; - for (size_t i = 0; i < count; i++) { - free(entries[i].name); - free(entries[i].link_target); - } - free(entries); -} - -static int compare_list_entries(const void* left, const void* right) { - const ListEntry* a = (const ListEntry*)left; - const ListEntry* b = (const ListEntry*)right; - return strcmp(a->name, b->name); -} - -/* Relative path of an entry below `root` ("" for the root itself). Mirrors - * change_list's relative_name for list-only rendering. */ -static char* list_relative_name(const char* root, const char* full) { - if (root == NULL || full == NULL) - return str_dup(full != NULL ? full : ""); - size_t root_len = strlen(root); - while (root_len > 1 && root[root_len - 1] == '/') - root_len--; - if (strncmp(root, full, root_len) == 0) { - if (full[root_len] == '\0') - return str_dup(""); - if (full[root_len] == '/') - return str_dup(full + root_len + 1); - } - return str_dup(full); -} - -/* --list-only: print an ls-style listing of the entries that WOULD be - * transferred and exit without contacting the server or writing anything. - * Names are transfer-relative (rsync prints `a.txt`, `sub/b.txt`, `.`) and - * directory entries are included. Returns 0 on success, 1 on error. */ -static int send_list_only(const Config* config) { - int skipped = 0; - if (!files_from_list_check(config, NULL, &skipped)) - return 1; - PreparedScanner prepared; - if (!prepare_scanner(config, 0, &prepared)) - return 1; - prepared.options.use_metadata = true; /* capture mode + mtime for the listing */ - prepared.options.list_dirs = true; - DirectoryScanner* scanner = - directory_scanner_create_with_options(config->send_directory, &prepared.options); - if (!scanner) { - prepared_scanner_destroy(&prepared); - return 1; - } - ListEntry* entries = NULL; - size_t count = 0; - size_t capacity = 0; - bool oom = false; - - /* rsync lists the source root itself (as "."). Only when the source is a - * directory and no --files-from subset is in effect. */ - if (config->files_from_set == NULL && config->send_directory != NULL) { - struct stat st; - if (stat(config->send_directory, &st) == 0 && S_ISDIR(st.st_mode)) { - capacity = 64; - entries = calloc(capacity, sizeof(ListEntry)); - if (entries == NULL) { - oom = true; - } else if ((entries[0].name = str_dup("")) == NULL) { - /* A NULL name would be dereferenced by qsort/render: fail the listing. */ - oom = true; - } else { - entries[0].mode = st.st_mode; - entries[0].mtime = st.st_mtime; - entries[0].mtime_nsec = st.st_mtim.tv_nsec; - entries[0].size = (unsigned long long)st.st_size; - entries[0].is_dir = true; - count = 1; - } - } - } - - Chunk* chunk; - while (!oom && (chunk = directory_scanner_next(scanner)) != NULL) { - for (int i = 0; i < chunk->element_count; i++) { - File* f = chunk->items[i]; - if (f == NULL) - continue; - if (count == capacity) { - size_t new_capacity = capacity > 0 ? capacity * 2 : 64; - if (new_capacity <= capacity) { - oom = true; - break; - } - ListEntry* grown = realloc(entries, new_capacity * sizeof(ListEntry)); - if (!grown) { - oom = true; - break; - } - entries = grown; - memset(entries + capacity, 0, (new_capacity - capacity) * sizeof(ListEntry)); - capacity = new_capacity; - } - char* name = list_relative_name(config->send_directory, file_wire_path(f)); - if (!name) { - oom = true; - break; - } - mode_t mode = 0; - time_t mtime = 0; - long mtime_nsec = 0; - if (f->metadata != NULL) { - mode = f->metadata->mode; - mtime = f->metadata->mtime_sec; - mtime_nsec = f->metadata->mtime_nsec; - } else { - struct stat st; - if (lstat(f->path, &st) == 0) { - mode = st.st_mode; - mtime = st.st_mtime; - mtime_nsec = st.st_mtim.tv_nsec; - } - } - entries[count].name = name; - entries[count].mode = mode; - entries[count].mtime = mtime; - entries[count].mtime_nsec = mtime_nsec; - if (f->is_symlink) - entries[count].size = f->symlink_target != NULL ? strlen(f->symlink_target) : 0; - else if (f->is_dir) { - struct stat dir_st; - entries[count].size = stat(f->path, &dir_st) == 0 ? (unsigned long long)dir_st.st_size : 0; - } else - entries[count].size = f->data != NULL ? f->data->size : 0; - entries[count].is_dir = f->is_dir; - entries[count].is_symlink = f->is_symlink; - entries[count].link_target = - f->is_symlink && f->symlink_target ? str_dup(f->symlink_target) : NULL; - count++; - } - chunk_destroy(chunk); - } - bool failed = oom || directory_scanner_failed(scanner) || directory_scanner_had_io_error(scanner); - directory_scanner_destroy(scanner); - prepared_scanner_destroy(&prepared); - if (failed) { - list_entries_destroy(entries, count); - if (oom) - log_message(LOG_LEVEL_ERROR, "memory allocation failed while listing"); - return 1; - } - if (count > 1) - qsort(entries, count, sizeof(ListEntry), compare_list_entries); - for (size_t i = 0; i < count; i++) { - ChangeEvent event; - memset(&event, 0, sizeof(event)); - event.name = entries[i].name; - event.path = entries[i].name; - event.mode = entries[i].mode; - event.size = entries[i].size; - event.mtime_sec = entries[i].mtime; - event.mtime_nsec = entries[i].mtime_nsec; - event.is_directory = entries[i].is_dir; - event.is_symlink = entries[i].is_symlink; - event.symlink_target = entries[i].link_target; - char* line = change_render_list_line(config, &event); - if (line != NULL) { - char* escaped = output_escape(line, config->eight_bit_output); - printf("%s\n", escaped != NULL ? escaped : line); - free(escaped); - free(line); - } - } - list_entries_destroy(entries, count); - return 0; -} - -/* Send the delete manifest to the server. Returns 0 on success, -1 on - failure. It carries FOUR sections: the keep-set paths, the protected - excluded prefixes, the --delete-missing-args exact-delete paths, and the - destination-relative directories the sender synchronized this run. - When --delete-excluded is given `protected` is empty: excluded destination - mirrors are then ordinary extras and are removed. When - --delete-missing-args is active `missing_args` holds the destination mirrors - of missing --files-from entries: each is an explicit receiver-side deletion - request, independent of the extras walk. `synced_dirs` confines the extras - walk to entries directly inside a synchronized directory. A NULL - keep-set / protected / missing / dirs list transmits an empty section. All - four sections are unbounded on the sender; the receiver enforces - MAX_MANIFEST_ENTRIES per section and a single MAX_MANIFEST_BYTES budget - shared across the sections, rejecting (with STATUS_ERROR) an over-budget - frame. A heavily filtered source whose exclusion list is large therefore - fails the run cleanly on the receiver rather than being truncated. */ -static int send_delete_manifest(int fd, ArrayList* manifest, ArrayList* protected_prefixes, - ArrayList* size_skipped, ArrayList* missing_args, - ArrayList* synced_dirs) { - if (!send_status(fd, STATUS_MANIFEST)) - return -1; - int keep_count = manifest ? manifest->size : 0; - if (!send_int(fd, keep_count)) - return -1; - for (int i = 0; i < keep_count; i++) { - if (!send_wire_str(fd, (char*)manifest->items[i])) - return -1; - } - /* The receiver has ONE protected-prefix section; filter-excluded prefixes - (dropped under --delete-excluded) and size-pruned prefixes (always - protected) are concatenated into it. */ - int protected_count = - (protected_prefixes ? protected_prefixes->size : 0) + (size_skipped ? size_skipped->size : 0); - if (!send_int(fd, protected_count)) - return -1; - if (protected_prefixes) { - for (int i = 0; i < protected_prefixes->size; i++) { - if (!send_wire_str(fd, (char*)protected_prefixes->items[i])) - return -1; - } - } - if (size_skipped) { - for (int i = 0; i < size_skipped->size; i++) { - if (!send_wire_str(fd, (char*)size_skipped->items[i])) - return -1; - } - } - int missing_count = missing_args ? missing_args->size : 0; - if (!send_int(fd, missing_count)) - return -1; - for (int i = 0; i < missing_count; i++) { - if (!send_wire_str(fd, (char*)missing_args->items[i])) - return -1; - } - int dirs_count = synced_dirs ? synced_dirs->size : 0; - if (!send_int(fd, dirs_count)) - return -1; - for (int i = 0; i < dirs_count; i++) { - if (!send_wire_str(fd, (char*)synced_dirs->items[i])) - return -1; - } - return 0; -} - -/* Transmit the keep-set manifest and wait for the receiver's verdict. Used by - --delete-before/--delete-during, where the extras are removed on the receiver - BEFORE the first byte of file data is sent: the receiver acknowledges with - STATUS_OK once the bounded delete committed, or STATUS_ERROR if it could not - (in which case the sender aborts without streaming any data). The ACK may - take much longer than an ordinary per-message round trip because the receiver - performs the whole bounded deletion walk (up to MAX_SERVER_DELETE_COUNT - unlinks) before replying, so the wait uses a generous explicit deadline - instead of the default 60 s receive window. */ -#define DELETE_ACK_TIMEOUT_SEC 3600 -/* While waiting for the (potentially slow) receiver-side deletion, send a - * STATUS_KEEPALIVE at most this often so the connection is demonstrably alive - * and neither side's per-message timeout trips. */ -#define DELETE_ACK_KEEPALIVE_SEC 10 - -static bool send_delete_manifest_early(Client* client, ArrayList* manifest, - ArrayList* protected_prefixes, ArrayList* size_skipped, - ArrayList* missing_args, ArrayList* synced_dirs) { - if (!client || !manifest) - return false; - if (send_delete_manifest(client->file_descriptor, manifest, protected_prefixes, size_skipped, - missing_args, synced_dirs) != 0) - return false; - Status ack; - /* The wait is long (up to an hour) and runs inline on this thread: a helper - * thread would race the non-thread-safe protocol send path, so keepalives are - * emitted from this wait loop itself. A Ctrl-C/SIGTERM abort flag also ends - * the wait; the caller then best-effort sends STATUS_ABORT. */ - if (!receive_status_keepalive(client->file_descriptor, &ack, DELETE_ACK_TIMEOUT_SEC, - DELETE_ACK_KEEPALIVE_SEC, client_abort_pending)) { - /* A Ctrl-C/SIGTERM abort ends the wait above; tell the receiver before the - caller tears the connection down (best-effort). */ - if (client_abort_pending()) { - log_info_message(LOG_INFO_MISC, - "Abort requested while awaiting delete ack; sending STATUS_ABORT"); - send_status(client->file_descriptor, STATUS_ABORT); - } - return false; - } - if (ack != STATUS_OK) { - log_server_rejection("Server failed to delete files before the transfer"); - return false; - } - return true; -} - -/* Walk the whole source tree once collecting only destination-relative wire - paths, loading and sending nothing. --delete-before/--delete-during need the - complete keep-set manifest before the first data byte, so it is built by a - dedicated pre-scan pass and transmitted early; the data pass then re-scans - with a fresh scanner. A source I/O error is fatal unless the options carry - --ignore-errors, in which case the scan continues past the unreadable - directory and *io_error_out reports it (the caller still performs the - deletion but reports the run as errored). */ -static bool scan_paths_only(const Config* config, const ScannerOptions* options, - ArrayList* manifest, DeletePlanSender* plans, bool* io_error_out, - unsigned long long* non_dir_count_out) { - if (io_error_out) - *io_error_out = false; - if (non_dir_count_out) - *non_dir_count_out = 0; - ScannerOptions local = *options; - /* The pre-scan is a paths-only pass with no client output; it must not emit - --info=nonreg lines (the data pass does that once). */ - local.note_nonreg = false; - DirectoryScanner* scanner = directory_scanner_create_with_options(config->send_directory, &local); - if (!scanner) - return false; - bool ok = true; - Chunk* chunk; - while ((chunk = directory_scanner_next(scanner)) != NULL) { - if (non_dir_count_out) { - for (int i = 0; i < chunk->element_count; i++) { - const File* f = chunk->items[i]; - if (f && !f->is_dir) - (*non_dir_count_out)++; - } - } - if (manifest && !add_chunk_to_manifest(manifest, chunk)) { - ok = false; - chunk_destroy(chunk); - break; - } - if (plans) { - for (int i = 0; i < chunk->element_count; i++) { - File* f = chunk->items[i]; - if (!f) - continue; - const char* path = file_wire_path(f); - if (!delete_plan_sender_add(plans, path, f->is_dir)) { - ok = false; - break; - } - } - if (!ok) { - chunk_destroy(chunk); - break; - } - } - chunk_destroy(chunk); - } - if (ok) { - /* Keep every traversed source directory, including empty ones, so a plan - no longer removes the destination directory itself. Their own plans are - emitted after the data stream (no file frame triggers them). */ - if (plans && options->plan_dirs) { - for (int i = 0; i < options->plan_dirs->size; i++) { - if (!delete_plan_sender_add(plans, (const char*)options->plan_dirs->items[i], true)) { - ok = false; - break; - } - } - } - } - if (ok && directory_scanner_failed(scanner)) - ok = false; - if (io_error_out) - *io_error_out = directory_scanner_had_io_error(scanner); - directory_scanner_destroy(scanner); - return ok; -} - -static int incremental_check(Client* client, File* file, const Config* config, - DeltaSignature** out_sig, unsigned long long* resume_offset) { +int incremental_check(Client* client, File* file, const Config* config, DeltaSignature** out_sig, + unsigned long long* resume_offset) { *out_sig = NULL; if (resume_offset) *resume_offset = 0; @@ -2141,249 +576,6 @@ static int send_append(const Client* client, File* file, Config* config, return ok ? 0 : -1; } -/* Server-contacting --dry-run. Connects to the configured remote/daemon and - * runs the normal per-file incremental decision WITHOUT transmitting any file - * data: the receiver (which also sees dry_run=true on the wire) answers - * STATUS_OK for an up-to-date file and STATUS_DRY_RUN_TRANSFER for a file it - * would otherwise write, mutating nothing on either side. The would-transfer - * set and the same trailer as the local dry-run are printed. A - * --compare-dest exact basis hit with no destination copy is reported as a - * skip by the receiver. - * - * Only regular files take the receiver-consulted check; directory / symlink / - * special / hard-link-sibling entries have no per-file content check, so they - * are reported conservatively as would-transfer and their frames are never - * sent (which is what keeps the receiver mutation-free). --delete* is - * deliberately NOT transmitted in dry-run, so no deletion can occur; the - * would-delete manifest report is a documented follow-up. - * - * Returns 0 on success, 1 on error. */ -static int send_dry_run_remote(Config* config) { - int from_skipped = 0; - ArrayList* missing_args = NULL; - if (config->delete_missing_args) { - missing_args = array_list_create(free); - if (!missing_args) - return 1; - } - if (!files_from_list_check(config, missing_args, &from_skipped)) { - if (missing_args) - array_list_delete(missing_args); - return 1; - } - if (missing_args) - array_list_delete(missing_args); - /* A live session may follow, so arm graceful abort handling. */ - client_set_abort_armed(true); - Client* client = connect_transfer_client(config); - if (!client) { - if (config->transport == TRANSPORT_TCP) - log_message(LOG_LEVEL_ERROR, "could not connect to server%s", - config->use_tls ? " via TLS" : ""); - client_set_abort_armed(false); - return 1; - } - ProtocolSession session; - protocol_session_init(&session, client->file_descriptor, client->file_descriptor); - protocol_session_set_io_timeout(&session, config->timeout); - protocol_session_set_ssl(&session, (SSL*)client->ssl); - protocol_session_bind(&session); - - int ret = 1; - time_t dry_start = time(NULL); - ReceiverStats dry_stats; - memset(&dry_stats, 0, sizeof(dry_stats)); - PreparedScanner prepared; - memset(&prepared, 0, sizeof(prepared)); - DirectoryScanner* scanner = NULL; - ArrayList* dry_manifest = NULL; - ArrayList* dry_dirs = NULL; - ArrayList* dry_excluded = NULL; - ArrayList* dry_size_skipped = NULL; - if (!config_send(client->file_descriptor, config)) - goto dry_fail; - receive_daemon_motd(client, config); - if (!prepare_scanner(config, 0, &prepared)) - goto dry_fail; - /* -n --delete: build the same keep-set manifest, protected prefixes, and - synchronized-directory scope a real run would send, so the receiver's - read-only extras walk enumerates exactly the deletions a real run makes. */ - if (config->use_delete) { - dry_manifest = array_list_create(free); - dry_dirs = array_list_create(free); - dry_size_skipped = array_list_create(free); - if (!dry_manifest || !dry_dirs || !dry_size_skipped) - goto dry_fail; - if (!config->delete_excluded) { - dry_excluded = array_list_create(free); - if (!dry_excluded) - goto dry_fail; - prepared.options.excluded_paths = dry_excluded; - } - prepared.options.size_skipped_paths = dry_size_skipped; - /* A --files-from subset confines the extras walk to the directories the - scan synchronized; a full recursive transfer marks the root itself. */ - if (config->files_from_set == NULL) { - char* root_marker = delete_scope_root_marker(config); - if (!root_marker || !array_list_add(dry_dirs, root_marker)) { - free(root_marker); - goto dry_fail; - } - } else { - prepared.options.synced_dirs = dry_dirs; - } - } - scanner = directory_scanner_create_with_options(config->send_directory, &prepared.options); - if (!scanner) - goto dry_fail; - - int file_count = 0; - unsigned long long total_bytes = 0; - char size_buffer[32]; - if (!config->quiet) - printf("Dry run: files to be transferred\n"); - Chunk* chunk; - while ((chunk = directory_scanner_next(scanner)) != NULL) { - if (dry_manifest && !add_chunk_to_manifest(dry_manifest, chunk)) { - chunk_destroy(chunk); - goto dry_fail; - } - for (int i = 0; i < chunk->element_count; i++) { - File* f = chunk->items[i]; - if (!f) - continue; - unsigned long long fsize = f->data ? f->data->size : 0; - bool would; - if (f->is_dir || f->is_symlink || f->is_special || - (f->link_group != 0 && !f->link_first && f->hardlink_target != NULL)) { - /* No receiver-side content check exists for these frame types; a real - run would (re)create them, so report would-transfer and send no - frame (the receiver must stay mutation-free). */ - would = true; - } else if (fsize > MAX_RECEIVE_WHOLE_FILE_SIZE && !config->use_incremental && - !config_has_basis(config)) { - /* A non-incremental run streams a >whole-file-limit source without the - STATUS_CHECK handshake, so no read-only receiver decision is possible - (and none is needed: a real run would transfer it). */ - would = true; - } else { - DeltaSignature* sig = NULL; - unsigned long long resume_offset = 0; - int rc = incremental_check(client, f, config, &sig, &resume_offset); - delta_signature_destroy(sig); - if (rc < 0) { - chunk_destroy(chunk); - goto dry_fail; - } - if (rc == 1) - continue; /* up to date; nothing to report */ - if (rc != 4) { - log_message(LOG_LEVEL_ERROR, "Unexpected receiver reply during dry-run"); - chunk_destroy(chunk); - goto dry_fail; - } - would = true; - } - if (would) { - if (!config->quiet) { - char* escaped_path = output_escape(file_wire_path(f), config->eight_bit_output); - if (!escaped_path) { - chunk_destroy(chunk); - goto dry_fail; - } - if (config->human_readable) - printf(" %s (%s)\n", escaped_path, - display_bytes(fsize, true, size_buffer, sizeof(size_buffer))); - else - printf(" %s (%llu bytes)\n", escaped_path, fsize); - free(escaped_path); - } - total_bytes += fsize; - file_count++; - } - } - chunk_destroy(chunk); - } - bool io_error = directory_scanner_had_io_error(scanner); - if (directory_scanner_failed(scanner)) - goto dry_fail; - if (io_error) - log_message(LOG_LEVEL_WARNING, "source scan hit an unreadable directory"); - /* Send the keep-set manifest (no data frames) so the receiver can enumerate - the destination extras; an early-timing delete ACKs before it will accept - the terminal FINISHED. */ - bool early_delete = config->use_delete && config_delete_timing_early(config); - if (dry_manifest) { - if (send_delete_manifest(client->file_descriptor, dry_manifest, dry_excluded, dry_size_skipped, - NULL, dry_dirs) != 0) - goto dry_fail; - if (early_delete) { - Status ack; - if (!receive_status_keepalive(client->file_descriptor, &ack, DELETE_ACK_TIMEOUT_SEC, - DELETE_ACK_KEEPALIVE_SEC, client_abort_pending) || - ack != STATUS_OK) - goto dry_fail; - } - } - /* Terminate the stream so the receiver emits its success frame; no data frame - is ever sent in dry-run. */ - if (!send_status(client->file_descriptor, STATUS_FINISHED)) - goto dry_fail; - Status status; - if (!receive_status(client->file_descriptor, &status)) - goto dry_fail; - if (status == STATUS_STATS) { - ArrayList* would_delete = array_list_create(free); - if (!would_delete) - goto dry_fail; - if (!receive_stats_record(client->file_descriptor, &dry_stats, would_delete)) { - array_list_delete(would_delete); - goto dry_fail; - } - print_delete_reports(config, would_delete); - array_list_delete(would_delete); - if (!receive_status(client->file_descriptor, &status)) - goto dry_fail; - } - if (status != STATUS_OK) - goto dry_fail; - if (!config->quiet) { - if (config->human_readable) - printf("Total: %d files, %s\n", file_count, - display_bytes(total_bytes, true, size_buffer, sizeof(size_buffer))); - else - printf("Total: %d files, %.1f MB\n", file_count, (double)total_bytes / (double)BYTES_PER_MIB); - } - { - TransferStats dry_transfer; - memset(&dry_transfer, 0, sizeof(dry_transfer)); - dry_transfer.flist_reg = (unsigned long long)file_count; - dry_transfer.total_file_size = total_bytes; - dry_transfer.transferred_regular = (unsigned long long)file_count; - dry_transfer.transferred_file_size = total_bytes; - dry_transfer.literal_data = total_bytes; - report_transfer_stats(config, &dry_transfer, dry_start, &dry_stats); - } - ret = io_error ? 1 : 0; - -dry_fail: - if (dry_manifest) - array_list_delete(dry_manifest); - if (dry_dirs) - array_list_delete(dry_dirs); - if (dry_excluded) - array_list_delete(dry_excluded); - if (dry_size_skipped) - array_list_delete(dry_size_skipped); - if (scanner) - directory_scanner_destroy(scanner); - prepared_scanner_destroy(&prepared); - disconnect_transfer_client(client); - protocol_session_unbind(); - client_set_abort_armed(false); - return ret; -} - // Send a single file directly (non-incremental path). static bool send_file_direct(File* file, int fd, bool use_metadata, int compression_level, const Config* config) { @@ -3185,133 +1377,104 @@ int apply_batch_to_dest(const Config* config, const char* batch_path, const char return rc; } -int send_files(Config* config) { - if (config->list_only) - return send_list_only(config); - if (config->dry_run) - return dry_run_targets_server(config) ? send_dry_run_remote(config) - : send_dry_run_manifest(config); - ArrayList* missing_args = NULL; - int skipped = 0; - if (config->delete_missing_args) { - missing_args = array_list_create(free); - if (!missing_args) - return 1; - } - if (!files_from_list_check(config, missing_args, &skipped)) { - if (missing_args) - array_list_delete(missing_args); - return 1; - } - - /* From here on a server session may be live, so Ctrl-C/SIGTERM should set the - abort flag (and be forwarded as STATUS_ABORT) instead of terminating. */ - client_set_abort_armed(true); - Client* client = connect_transfer_client(config); - if (!client) { - if (config->transport == TRANSPORT_TCP) - log_message(LOG_LEVEL_ERROR, "could not connect to server%s", - config->use_tls ? " via TLS" : ""); - if (missing_args) - array_list_delete(missing_args); - return 1; - } - ProtocolSession session; - protocol_session_init(&session, client->file_descriptor, client->file_descriptor); - protocol_session_set_io_timeout(&session, config->timeout); - protocol_session_set_ssl(&session, (SSL*)client->ssl); - protocol_session_bind(&session); - int ret = 1; - DirectoryScanner* scanner = NULL; - ArrayList* manifest = NULL; - DeletePlanSender* plan_sender = NULL; - ArrayList* remove_sources = NULL; - /* P7 Wave D: captured source directory times, transmitted in trailing - STATUS_DIR_TIMES frame(s) (only when metadata rides the wire). */ - ArrayList* dir_entries = NULL; - /* --stats directory accounting for the no-metadata (-r) case. */ +/* Resources and phase state threaded through the single-threaded send path. + * The send_files_* helpers populate it incrementally and send_files_cleanup + * releases every owned field; the members mirror the locals of the original + * monolithic send_files, so ownership and destruction order are unchanged. */ +typedef struct { + Client* client; + DirectoryScanner* scanner; + ArrayList* manifest; + DeletePlanSender* plan_sender; + ArrayList* remove_sources; + ArrayList* dir_entries; atomic_ullong dir_count; - atomic_init(&dir_count, 0); - /* Protected excluded prefixes (delete-excluded default protection). */ - ArrayList* excluded = NULL; - /* Size-pruned prefixes (always protected) and synchronized directories. */ - ArrayList* size_skipped = NULL; - ArrayList* synced_dirs = NULL; - /* Traversed source directories for the per-directory delete keep set. */ - ArrayList* plan_dirs = NULL; - bool delete_early = config->use_delete && config_delete_timing_early(config); - /* --delete-during/--delete-delay use per-directory plans for every transfer - shape. For -d/--dirs the generator records only the directories whose - direct children it actually enumerated, so the plan removes extras directly - inside a listed directory while an untraversed (kept) subdirectory is - shielded -- rsync's `-d DIR/ --delete`. */ - bool delete_per_dir = config->use_delete && config_delete_timing_per_dir(config); - bool send_failed = false; - bool had_scan_io = false; - unsigned long long per_dir_non_dir_count = 0; + ArrayList* excluded; + ArrayList* size_skipped; + ArrayList* synced_dirs; + ArrayList* plan_dirs; + ArrayList* missing_args; PreparedScanner prepared; - memset(&prepared, 0, sizeof(prepared)); + StopCondition stop; + TransferStats transfer_stats; + time_t start; + bool delete_early; + bool delete_per_dir; + bool had_scan_io; + bool scan_stopped_early; + unsigned long long per_dir_non_dir_count; +} SendFilesState; + +/* Post-connect setup: negotiate the protocol, consume the daemon MOTD, build + * the scanner and allocate the delete/keep-set/remove-source list containers. */ +static bool send_files_prepare(Config* config, SendFilesState* state) { + Client* client = state->client; if (!config_send(client->file_descriptor, config)) - goto send_fail; + return false; receive_daemon_motd(client, config); - if (!prepare_scanner(config, 0, &prepared)) - goto send_fail; + if (!prepare_scanner(config, 0, &state->prepared)) + return false; if (dir_metadata_should_capture(config)) { - dir_entries = array_list_create(file_destroy); - if (!dir_entries) - goto send_fail; - if (!append_implied_dir_times(config, dir_entries)) - goto send_fail; + state->dir_entries = array_list_create(file_destroy); + if (!state->dir_entries) + return false; + if (!append_implied_dir_times(config, state->dir_entries)) + return false; } if (config->remove_source_files) - remove_sources = array_list_create(source_file_destroy); - if (config->remove_source_files && !remove_sources) - goto send_fail; + state->remove_sources = array_list_create(source_file_destroy); + if (config->remove_source_files && !state->remove_sources) + return false; /* Unless --delete-excluded opts out, collect the paths the source scan prunes by user-selection rules so the receiver protects their destination mirrors from --delete (rsync's default). Only scans that build the keep-set get the sink attached (prescan for early timing, the streaming data pass otherwise). */ if (config->use_delete) { if (!config->delete_excluded) { - excluded = array_list_create(free); - if (!excluded) - goto send_fail; - prepared.options.excluded_paths = excluded; + state->excluded = array_list_create(free); + if (!state->excluded) + return false; + state->prepared.options.excluded_paths = state->excluded; } - size_skipped = array_list_create(free); - synced_dirs = array_list_create(free); - if (!size_skipped || !synced_dirs) - goto send_fail; - prepared.options.size_skipped_paths = size_skipped; + state->size_skipped = array_list_create(free); + state->synced_dirs = array_list_create(free); + if (!state->size_skipped || !state->synced_dirs) + return false; + state->prepared.options.size_skipped_paths = state->size_skipped; /* Only a --files-from subset confines the extras walk to the directories the scan synchronized; a full recursive transfer deletes throughout the receive root, so mark the root itself (the "." sentinel) and let the scanner record nothing extra. */ if (config->files_from_set == NULL) { char* root_marker = delete_scope_root_marker(config); - if (!root_marker || !array_list_add(synced_dirs, root_marker)) { + if (!root_marker || !array_list_add(state->synced_dirs, root_marker)) { free(root_marker); - goto send_fail; + return false; } } else { - prepared.options.synced_dirs = synced_dirs; + state->prepared.options.synced_dirs = state->synced_dirs; } } - /* The late-timing modes (--delete-after/--delete-commit) build the manifest - while streaming and send it after the last data frame. --delete-before - sends a whole-tree keep-set up front; --delete-during/--delete-delay build - the complete per-directory plan set up front (paths only) and transmit it - all before the first data frame, so a mid-transfer abort has already - applied every planned removal. */ - if (delete_early) { + return true; +} + +/* Delete-timing pre-pass. The late-timing modes (--delete-after/--delete-commit) + * build the manifest while streaming (handled by the run/finalize phases); + * --delete-before sends a whole-tree keep-set up front, and --delete-during/ + * --delete-delay build the complete per-directory plan set up front (paths only) + * and transmit it all before the first data frame, so a mid-transfer abort has + * already applied every planned removal. */ +static bool send_files_prepare_delete(Config* config, SendFilesState* state) { + Client* client = state->client; + if (state->delete_early) { /* Pass 1: collect the complete keep-set (paths only, no data loaded) and transmit it now, before any file data. The receiver removes extras and acks; the transfer aborts here if the deletion could not commit. */ ArrayList* early_manifest = array_list_create(free); if (!early_manifest) - goto send_fail; - bool prescan_ok = - scan_paths_only(config, &prepared.options, early_manifest, NULL, &had_scan_io, NULL); + return false; + bool prescan_ok = scan_paths_only(config, &state->prepared.options, early_manifest, NULL, + &state->had_scan_io, NULL); bool early_ok = false; bool skip_delete = false; if (prescan_ok) { @@ -3320,84 +1483,92 @@ int send_files(Config* config) { and an empty keep-set would delete the whole destination. Refuse to delete; the genuine-empty-source case has no io_error and still sends its (empty) keep-set. */ - if (had_scan_io && early_manifest->size == 0) { + if (state->had_scan_io && early_manifest->size == 0) { log_message(LOG_LEVEL_ERROR, "source scan hit an I/O error before finding any file; refusing to delete " "with an empty keep-set (--delete)"); prescan_ok = false; - } else if (!ignore_errors_allows_delete(config, had_scan_io)) { + } else if (!ignore_errors_allows_delete(config, state->had_scan_io)) { /* rsync default: an I/O error suppresses deletion unless --ignore-errors. Skip the manifest; the transfer still proceeds. */ log_message(LOG_LEVEL_WARNING, "IO error encountered -- skipping file deletion"); skip_delete = true; } else { - early_ok = send_delete_manifest_early(client, early_manifest, excluded, size_skipped, - missing_args, synced_dirs); + early_ok = + send_delete_manifest_early(client, early_manifest, state->excluded, state->size_skipped, + state->missing_args, state->synced_dirs); } } array_list_delete(early_manifest); /* The keep-set (and its protected prefixes and synchronized directories) are already on the wire; the data pass must not append to those lists again. */ - prepared.options.excluded_paths = NULL; - prepared.options.size_skipped_paths = NULL; - prepared.options.synced_dirs = NULL; + state->prepared.options.excluded_paths = NULL; + state->prepared.options.size_skipped_paths = NULL; + state->prepared.options.synced_dirs = NULL; if (!prescan_ok || (!early_ok && !skip_delete)) - goto send_fail; - } else if (delete_per_dir) { + return false; + } else if (state->delete_per_dir) { /* --delete-during/--delete-delay: build one plan per source directory from a path-only pre-scan and transmit the COMPLETE plan set now, before any data, so every planned removal has already been applied when a later transfer phase fails -- exactly like rsync's generator, whose deletion list runs ahead of its throttled sender. A completed run is unaffected. */ - plan_sender = delete_plan_sender_create(); - plan_dirs = array_list_create(free); - if (!plan_sender || !plan_dirs) - goto send_fail; - prepared.options.plan_dirs = plan_dirs; - bool prescan_ok = scan_paths_only(config, &prepared.options, NULL, plan_sender, &had_scan_io, - &per_dir_non_dir_count); + state->plan_sender = delete_plan_sender_create(); + state->plan_dirs = array_list_create(free); + if (!state->plan_sender || !state->plan_dirs) + return false; + state->prepared.options.plan_dirs = state->plan_dirs; + bool prescan_ok = scan_paths_only(config, &state->prepared.options, NULL, state->plan_sender, + &state->had_scan_io, &state->per_dir_non_dir_count); bool plans_ok = false; bool skip_delete = false; if (prescan_ok) { - const char* walk_root = delete_plan_walk_root(config, synced_dirs); + const char* walk_root = delete_plan_walk_root(config, state->synced_dirs); const ArrayList* scope = - config->files_from_set ? synced_dirs : (walk_root ? synced_dirs : NULL); - delete_plan_sender_finalize(plan_sender, scope, walk_root); - delete_plan_sender_set_config(plan_sender, excluded, size_skipped, missing_args); - if (had_scan_io && delete_plan_sender_empty(plan_sender)) { + config->files_from_set ? state->synced_dirs : (walk_root ? state->synced_dirs : NULL); + delete_plan_sender_finalize(state->plan_sender, scope, walk_root); + delete_plan_sender_set_config(state->plan_sender, state->excluded, state->size_skipped, + state->missing_args); + if (state->had_scan_io && delete_plan_sender_empty(state->plan_sender)) { log_message(LOG_LEVEL_ERROR, "source scan hit an I/O error before finding any file; refusing to delete " "with an empty keep-set (--delete)"); prescan_ok = false; - } else if (!ignore_errors_allows_delete(config, had_scan_io)) { + } else if (!ignore_errors_allows_delete(config, state->had_scan_io)) { /* rsync default: an I/O error suppresses deletion unless --ignore-errors. Drop the plans; the transfer still proceeds. */ log_message(LOG_LEVEL_WARNING, "IO error encountered -- skipping file deletion"); - delete_plan_sender_destroy(plan_sender); - plan_sender = NULL; - array_list_delete(plan_dirs); - plan_dirs = NULL; + delete_plan_sender_destroy(state->plan_sender); + state->plan_sender = NULL; + array_list_delete(state->plan_dirs); + state->plan_dirs = NULL; skip_delete = true; } else { - plans_ok = delete_plan_send_all(client->file_descriptor, plan_sender, plan_dirs) == 0; + plans_ok = delete_plan_send_all(client->file_descriptor, state->plan_sender, + state->plan_dirs) == 0; } } - prepared.options.excluded_paths = NULL; - prepared.options.size_skipped_paths = NULL; - prepared.options.synced_dirs = NULL; - prepared.options.plan_dirs = NULL; + state->prepared.options.excluded_paths = NULL; + state->prepared.options.size_skipped_paths = NULL; + state->prepared.options.synced_dirs = NULL; + state->prepared.options.plan_dirs = NULL; if (!prescan_ok || (!plans_ok && !skip_delete)) - goto send_fail; + return false; } else if (config->use_delete) { - manifest = array_list_create(free); - if (!manifest) - goto send_fail; + state->manifest = array_list_create(free); + if (!state->manifest) + return false; } - /* --progress/--info=progress: pre-count the file list for rsync's to-chk - denominator. When a --delete-during/--delete-delay pre-scan already ran, - reuse its traversed directory list instead of walking the tree again. */ + return true; +} + +/* Streaming run phase: pre-count progress, arm the stop deadline, scan the + * source and transmit every chunk. Returns false on a fatal error (the caller + * runs the shared cleanup). */ +static bool send_files_run(Config* config, SendFilesState* state) { + Client* client = state->client; if (progress_requested(config)) - client_progress_prepare(config, plan_dirs, per_dir_non_dir_count); + client_progress_prepare(config, state->plan_dirs, state->per_dir_non_dir_count); /* Phase 6: compute the client-only stop deadline once at transfer start. The early-delete pre-scan above deliberately ignores it so the keep-set (and its committed deletion) is always complete and correct. */ @@ -3406,27 +1577,27 @@ int send_files(Config* config) { now_mono.tv_sec = 0; now_mono.tv_nsec = 0; } - StopCondition stop = stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, - config->stop_at_set, config->stop_at, now_mono); - prepared.options.stop_condition = &stop; + state->stop = stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, + config->stop_at_set, config->stop_at, now_mono); + state->prepared.options.stop_condition = &state->stop; /* The early-delete pre-scan above already ran; only the data pass should feed the directory-time list (otherwise every directory would be captured twice). */ - prepared.options.dir_entries = dir_entries; - prepared.options.dir_count = config->stats ? &dir_count : NULL; - scanner = directory_scanner_create_with_options(config->send_directory, &prepared.options); - if (!scanner) - goto send_fail; + state->prepared.options.dir_entries = state->dir_entries; + state->prepared.options.dir_count = config->stats ? &state->dir_count : NULL; + state->scanner = + directory_scanner_create_with_options(config->send_directory, &state->prepared.options); + if (!state->scanner) + return false; Chunk* current_chunk; - TransferStats transfer_stats; - memset(&transfer_stats, 0, sizeof(transfer_stats)); - time_t start = time(NULL); + memset(&state->transfer_stats, 0, sizeof(state->transfer_stats)); + state->start = time(NULL); client_progress_begin(config); /* True when the stop deadline cut the scan short so the keep-set manifest is only a prefix of the source. */ - bool scan_stopped_early = false; - while ((current_chunk = directory_scanner_next(scanner)) != NULL) { + bool send_failed = false; + while ((current_chunk = directory_scanner_next(state->scanner)) != NULL) { /* Graceful abort (Ctrl-C/SIGTERM): notify the receiver and clean up. The session is active (config_send already succeeded); a send failure here is fine because the client is exiting anyway. */ @@ -3434,21 +1605,21 @@ int send_files(Config* config) { log_info_message(LOG_INFO_MISC, "Abort requested; sending STATUS_ABORT to server"); chunk_destroy(current_chunk); send_status(client->file_descriptor, STATUS_ABORT); - goto send_fail; + return false; } /* Phase 6: stop-elegantly at the next chunk boundary once the deadline has passed. The scanner may also have stopped early itself; either way the completion tail below keeps everything already sent. */ - if (stop_condition_reached(&stop)) { + if (stop_condition_reached(&state->stop)) { chunk_destroy(current_chunk); log_info_message(LOG_INFO_MISC, "Stop deadline reached; stopping transfer at the next chunk boundary"); - scan_stopped_early = true; + state->scan_stopped_early = true; break; } - if (manifest && !add_chunk_to_manifest(manifest, current_chunk)) { + if (state->manifest && !add_chunk_to_manifest(state->manifest, current_chunk)) { chunk_destroy(current_chunk); - goto send_fail; + return false; } if (!config->use_sendfile) { bool load_ok = true; @@ -3464,11 +1635,11 @@ int send_files(Config* config) { } if (!load_ok) { chunk_destroy(current_chunk); - goto send_fail; + return false; } } - if (send_chunk_with_removal(client, current_chunk, config, remove_sources, &transfer_stats) != - 0) { + if (send_chunk_with_removal(client, current_chunk, config, state->remove_sources, + &state->transfer_stats) != 0) { log_message(LOG_LEVEL_ERROR, "Failed to send chunk"); chunk_destroy(current_chunk); send_failed = true; @@ -3476,31 +1647,32 @@ int send_files(Config* config) { } chunk_destroy(current_chunk); } - if (send_failed) { - if (manifest) { - array_list_delete(manifest); - manifest = NULL; - } - goto send_fail; - } - if (directory_scanner_failed(scanner)) - goto send_fail; - if (directory_scanner_had_io_error(scanner)) - had_scan_io = true; + return !send_failed; +} + +/* Completion tail: send the late delete manifest and captured directory times, + * finalize the receiver handshake, remove transferred sources and report stats. + * Returns the rsync-compatible exit code. */ +static int send_files_finalize(Config* config, SendFilesState* state) { + Client* client = state->client; + if (directory_scanner_failed(state->scanner)) + return 1; + if (directory_scanner_had_io_error(state->scanner)) + state->had_scan_io = true; /* An abort that arrived after the last chunk must still stop the completion tail (manifest/finalize) rather than let it run to success. */ if (client_abort_pending()) { log_info_message(LOG_INFO_MISC, "Abort requested; sending STATUS_ABORT to server"); send_status(client->file_descriptor, STATUS_ABORT); - goto send_fail; + return 1; } /* Phase 6: the scanner may have stopped early (returning NULL without a failure) as soon as the deadline passed, so reflect that here too. A deadline that cut the scan short leaves an incomplete keep-set; transmitting it would make the receiver --delete the unscanned source mirrors (data loss), so the late delete manifest is suppressed below. */ - scan_stopped_early = scan_stopped_early || stop_condition_reached(&stop); - if (scan_stopped_early) { + state->scan_stopped_early = state->scan_stopped_early || stop_condition_reached(&state->stop); + if (state->scan_stopped_early) { if (config->use_delete || config->delete_missing_args) log_message(LOG_LEVEL_WARNING, "transfer stopped early (stop deadline); skipping --delete keep-set so " @@ -3508,25 +1680,25 @@ int send_files(Config* config) { else log_message(LOG_LEVEL_WARNING, "transfer stopped early (stop deadline)"); } else { - if (had_scan_io && manifest && manifest->size == 0) { + if (state->had_scan_io && state->manifest && state->manifest->size == 0) { /* A scan that hit an I/O error and produced no keep entries is ambiguous; an empty keep-set would delete the whole destination. Refuse to delete (see the early-timing comment above). */ log_message(LOG_LEVEL_ERROR, "source scan hit an I/O error before finding any file; refusing to delete with " "an empty keep-set (--delete)"); - goto send_fail; + return 1; } /* rsync default: a scan I/O error suppresses deletion unless --ignore-errors, even in the late (commit) modes. Drop the keep-set so the receiver removes nothing; the readable tree still transferred. */ - bool late_delete = - (manifest || config->delete_missing_args) && !delete_early && !delete_per_dir; - if (late_delete && !ignore_errors_allows_delete(config, had_scan_io)) { + bool late_delete = (state->manifest || config->delete_missing_args) && !state->delete_early && + !state->delete_per_dir; + if (late_delete && !ignore_errors_allows_delete(config, state->had_scan_io)) { log_message(LOG_LEVEL_WARNING, "IO error encountered -- skipping file deletion"); - if (manifest) { - array_list_delete(manifest); - manifest = NULL; + if (state->manifest) { + array_list_delete(state->manifest); + state->manifest = NULL; } } else if (late_delete) { /* Late (commit) ordering: all file data is out; transmit the manifest so @@ -3535,83 +1707,144 @@ int send_files(Config* config) { succeeds. In the early modes (--delete-before) and the per-directory modes the deletion already went out with the data, so nothing is re-sent here. */ - if (send_delete_manifest(client->file_descriptor, manifest, excluded, size_skipped, - missing_args, synced_dirs) != 0) { - if (manifest) { - array_list_delete(manifest); - manifest = NULL; + if (send_delete_manifest(client->file_descriptor, state->manifest, state->excluded, + state->size_skipped, state->missing_args, state->synced_dirs) != 0) { + if (state->manifest) { + array_list_delete(state->manifest); + state->manifest = NULL; } - goto send_fail; + return 1; } - if (manifest) { - array_list_delete(manifest); - manifest = NULL; + if (state->manifest) { + array_list_delete(state->manifest); + state->manifest = NULL; } } } /* P7 Wave D: every directory has now been traversed (or the scan stopped early), so transmit the captured directory times last. The receiver defers applying them until after its own deletion/publication phase. */ - if (!send_dir_times(client, config, dir_entries)) - goto send_fail; + if (!send_dir_times(client, config, state->dir_entries)) + return 1; bool delete_limit = false; ReceiverStats recv_stats; memset(&recv_stats, 0, sizeof(recv_stats)); - bool ok = finalize_transfer(client, config, remove_sources, &delete_limit, &recv_stats); + bool ok = finalize_transfer(client, config, state->remove_sources, &delete_limit, &recv_stats); if (!ok && config->use_delete) log_message(LOG_LEVEL_ERROR, "server reported a deletion failure (--delete); see the server log for the reason"); if (ok) - remove_transferred_sources(config, remove_sources); + remove_transferred_sources(config, state->remove_sources); /* A recursive -a scan has no directory entries in its chunks; account them from the scanner's captured directory list (present whenever a directory attribute is preserved, e.g. -a/-t/-p). The -d generator counts its explicit directory entries inline instead. */ - transfer_stats.flist_dir += dir_count_for_stats(config, dir_entries, &dir_count); - report_transfer_stats(config, &transfer_stats, start, &recv_stats); + state->transfer_stats.flist_dir += + dir_count_for_stats(config, state->dir_entries, &state->dir_count); + report_transfer_stats(config, &state->transfer_stats, state->start, &recv_stats); log_info_message(LOG_INFO_STATS, "Transfer summary: %llu files, %.1f MB", - transfer_stats.transferred_regular, - (double)transfer_stats.transferred_file_size / (double)BYTES_PER_MIB); + state->transfer_stats.transferred_regular, + (double)state->transfer_stats.transferred_file_size / (double)BYTES_PER_MIB); /* A skipped source entry (--ignore-errors past an unreadable directory, or a dereferenced symlink with no referent) makes rsync report a partial transfer (exit 23) even though the rest of the run succeeded. A --max-delete-capped commit is a successful transfer that rsync reports with exit code 25. */ if (!ok) - ret = 1; - else if (had_scan_io) - ret = 23; - else - ret = delete_limit ? 25 : 0; + return 1; + if (state->had_scan_io) + return 23; + return delete_limit ? 25 : 0; +} -send_fail: - /* Single cleanup path for all exits. The manifest is intentionally deleted - here even on success without --delete, fixing a pre-existing leak. */ - if (manifest) - array_list_delete(manifest); - if (plan_sender) - delete_plan_sender_destroy(plan_sender); - if (excluded) - array_list_delete(excluded); - if (size_skipped) - array_list_delete(size_skipped); - if (synced_dirs) - array_list_delete(synced_dirs); - if (plan_dirs) - array_list_delete(plan_dirs); - if (missing_args) - array_list_delete(missing_args); - if (remove_sources) - array_list_delete(remove_sources); - if (dir_entries) - array_list_delete(dir_entries); - if (scanner) - directory_scanner_destroy(scanner); - prepared_scanner_destroy(&prepared); +/* Single cleanup path for all exits. The manifest is intentionally deleted + * here even on success without --delete, fixing a pre-existing leak. */ +static void send_files_cleanup(SendFilesState* state) { + if (state->manifest) + array_list_delete(state->manifest); + if (state->plan_sender) + delete_plan_sender_destroy(state->plan_sender); + if (state->excluded) + array_list_delete(state->excluded); + if (state->size_skipped) + array_list_delete(state->size_skipped); + if (state->synced_dirs) + array_list_delete(state->synced_dirs); + if (state->plan_dirs) + array_list_delete(state->plan_dirs); + if (state->missing_args) + array_list_delete(state->missing_args); + if (state->remove_sources) + array_list_delete(state->remove_sources); + if (state->dir_entries) + array_list_delete(state->dir_entries); + if (state->scanner) + directory_scanner_destroy(state->scanner); + prepared_scanner_destroy(&state->prepared); client_progress_cleanup(); - disconnect_transfer_client(client); + disconnect_transfer_client(state->client); protocol_session_unbind(); client_set_abort_armed(false); +} + +int send_files(Config* config) { + if (config->list_only) + return send_list_only(config); + if (config->dry_run) + return dry_run_targets_server(config) ? send_dry_run_remote(config) + : send_dry_run_manifest(config); + SendFilesState state; + memset(&state, 0, sizeof(state)); + atomic_init(&state.dir_count, 0); + state.delete_early = config->use_delete && config_delete_timing_early(config); + /* --delete-during/--delete-delay use per-directory plans for every transfer + shape. For -d/--dirs the generator records only the directories whose + direct children it actually enumerated, so the plan removes extras directly + inside a listed directory while an untraversed (kept) subdirectory is + shielded -- rsync's `-d DIR/ --delete`. */ + state.delete_per_dir = config->use_delete && config_delete_timing_per_dir(config); + int skipped = 0; + if (config->delete_missing_args) { + state.missing_args = array_list_create(free); + if (!state.missing_args) + return 1; + } + if (!files_from_list_check(config, state.missing_args, &skipped)) { + if (state.missing_args) + array_list_delete(state.missing_args); + return 1; + } + + /* From here on a server session may be live, so Ctrl-C/SIGTERM should set the + abort flag (and be forwarded as STATUS_ABORT) instead of terminating. */ + client_set_abort_armed(true); + Client* client = connect_transfer_client(config); + if (!client) { + if (config->transport == TRANSPORT_TCP) + log_message(LOG_LEVEL_ERROR, "could not connect to server%s", + config->use_tls ? " via TLS" : ""); + if (state.missing_args) + array_list_delete(state.missing_args); + return 1; + } + state.client = client; + ProtocolSession session; + protocol_session_init(&session, client->file_descriptor, client->file_descriptor); + protocol_session_set_io_timeout(&session, config->timeout); + protocol_session_set_ssl(&session, (SSL*)client->ssl); + protocol_session_bind(&session); + + int ret = 1; + if (!send_files_prepare(config, &state)) + goto send_fail; + if (!send_files_prepare_delete(config, &state)) + goto send_fail; + if (!send_files_run(config, &state)) + goto send_fail; + ret = send_files_finalize(config, &state); + +send_fail: + send_files_cleanup(&state); return ret; } diff --git a/src/client/client_send_internal.h b/src/client/client_send_internal.h new file mode 100644 index 0000000..6d042b5 --- /dev/null +++ b/src/client/client_send_internal.h @@ -0,0 +1,91 @@ +#ifndef CLIENT_SEND_INTERNAL_H +#define CLIENT_SEND_INTERNAL_H + +/* Declarations shared between the client_send.c transfer orchestration and the + * reporting (client_report.c), scanner-preparation (client_scan.c) and + * manifest/list/dry-run (client_manifest.c) translation units that were split + * out of it. Nothing here is part of the public client_send.h facade. */ + +#include "array_list.h" +#include "client_send.h" +#include "config.h" +#include "delete_plan.h" +#include "delta.h" +#include "format.h" +#include "log.h" +#include "scanner.h" +#include +#include +#include +#include + +/* One mebibyte in bytes; the unit used by the --stats/--progress lines. + Always cast to double when dividing so the output stays fractional. */ +#define BYTES_PER_MIB (1024ULL * 1024ULL) + +/* Compiled scanner inputs that are shared read-only across scanner instances + * and, in -m mode, across worker threads. `base_filters` owns the compiled + * command-line + -C rules; the FileListSet allow-set lives in the Config. + * `hardlinks` owns the --hard-links/-H link-group detection table (NULL when + * off) and is shared (mutex-guarded) across every scanner/worker of one scan. */ +typedef struct { + ScannerOptions options; + FilterRuleList* base_filters; /* owned; may be NULL */ + HardLinkTable* hardlinks; /* owned; may be NULL */ + char* relative_prefix; /* owned -R prefix; may be NULL */ +} PreparedScanner; + +/* client_scan.c */ +bool prepare_scanner(const Config* config, int num_threads, PreparedScanner* out); +void prepared_scanner_destroy(PreparedScanner* prepared); +bool append_implied_dir_times(const Config* config, ArrayList* dir_entries); +char* delete_scope_root_marker(const Config* config); +const char* delete_plan_walk_root(const Config* config, const ArrayList* synced_dirs); +bool files_from_list_check(const Config* config, ArrayList* missing_dest, int* skipped_out); +bool scan_paths_only(const Config* config, const ScannerOptions* options, ArrayList* manifest, + DeletePlanSender* plans, bool* io_error_out, + unsigned long long* non_dir_count_out); + +/* client_report.c */ +void log_server_rejection(const char* context); +const char* display_bytes(unsigned long long bytes, bool human_readable, char* buffer, + size_t buffer_size); +unsigned long long dir_count_for_stats(const Config* config, const ArrayList* dir_entries, + atomic_ullong* counter); +void report_transfer_stats(const Config* config, const TransferStats* stats, time_t start, + const ReceiverStats* recv); +void transfer_stats_note_entry(TransferStats* stats, const File* file); +void transfer_stats_note_transferred(TransferStats* stats, const File* file); +bool info_flag_enabled(const Config* config, LogInfoFlag flag); +void print_delete_reports(const Config* config, const ArrayList* paths); +const char* delete_display_path(const Config* config, const char* path); +bool progress_requested(const Config* config); +void client_progress_cleanup(void); +void client_progress_begin(const Config* config); +void client_progress_file(const Config* config, const File* file); +void client_progress_name(const Config* config, const File* file); +void client_progress_uptodate(const Config* config, const File* file); +void client_progress_prepare(const Config* config, const ArrayList* plan_dirs, + unsigned long long plan_non_dir_count); +bool receive_stats_record(int fd, ReceiverStats* stats, ArrayList* would_delete); + +/* client_send.c */ +void receive_daemon_motd(Client* client, const Config* config); +Client* connect_transfer_client(const Config* config); +void disconnect_transfer_client(Client* client); +int incremental_check(Client* client, File* file, const Config* config, DeltaSignature** out_sig, + unsigned long long* resume_offset); + +/* client_manifest.c */ +bool dry_run_targets_server(const Config* config); +bool add_chunk_to_manifest(ArrayList* manifest, const Chunk* chunk); +int send_dry_run_manifest(const Config* config); +int send_list_only(const Config* config); +int send_dry_run_remote(Config* config); +int send_delete_manifest(int fd, ArrayList* manifest, ArrayList* protected_prefixes, + ArrayList* size_skipped, ArrayList* missing_args, ArrayList* synced_dirs); +bool send_delete_manifest_early(Client* client, ArrayList* manifest, ArrayList* protected_prefixes, + ArrayList* size_skipped, ArrayList* missing_args, + ArrayList* synced_dirs); + +#endif From a6659472feb3cdacd513829dec071e3a2d176385 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 13:38:05 +0200 Subject: [PATCH 02/10] refactor(scanner): split into filter/sequential/parallel TUs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Move the filter rule-tree/context helpers and entry inspection into scanner_filter.c, the parallel scanner into scanner_parallel.c, and keep the sequential scanner in scanner.c. Shared internal declarations live in the new scanner_internal.h; scanner.h stays the public façade. Decompose directory_scanner_next into static helpers (skipped-entry, selection-protection, mount/-x, directory-finish and per-entry handlers) with no semantic change. --- CMakeLists.txt | 2 + src/client/scanner.c | 1929 ++++++--------------------------- src/client/scanner_filter.c | 671 ++++++++++++ src/client/scanner_internal.h | 107 ++ src/client/scanner_parallel.c | 703 ++++++++++++ 5 files changed, 1789 insertions(+), 1623 deletions(-) create mode 100644 src/client/scanner_filter.c create mode 100644 src/client/scanner_internal.h create mode 100644 src/client/scanner_parallel.c diff --git a/CMakeLists.txt b/CMakeLists.txt index a55d93e..dc99f62 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -139,6 +139,8 @@ set(CLIENT_CORE_SRCS src/client/client_send.c src/client/client_validation.c src/client/scanner.c + src/client/scanner_filter.c + src/client/scanner_parallel.c src/client/usage.c ) set(CLIENT_MAIN_SRCS src/client/client_cli.c) diff --git a/src/client/scanner.c b/src/client/scanner.c index 1adbab8..55f38f1 100644 --- a/src/client/scanner.c +++ b/src/client/scanner.c @@ -1,5 +1,6 @@ #include "log.h" #include "scanner.h" +#include "scanner_internal.h" #include "array_list.h" #include "chunk.h" #include "file.h" @@ -17,194 +18,6 @@ #include "xattr.h" -typedef struct { - char* path; - int depth; - FilterNode* context; /* inherited per-directory filter context */ -} DirEntry; - -/* A chain node: `own` holds the .rsync-filter rules of one directory, `parent` - * the context that directory inherited (nearest ancestor with a filter file). - * The chain for a directory's contents runs from that directory's own node up - * to the root; the command-line base rules are evaluated after the whole - * chain. */ -struct FilterNode { - FilterNode* parent; - FilterRuleList* own; -}; - -static void filter_node_destroy(void* item) { - if (item) { - FilterNode* node = (FilterNode*)item; - if (node->own) - filter_rule_list_free(node->own); - free(node); - } -} - -static FilterNode* filter_node_alloc(FilterNode* parent, FilterRuleList* own) { - FilterNode* node = malloc(sizeof(FilterNode)); - if (!node) - return NULL; - node->parent = parent; - node->own = own; - return node; -} - -/* Evaluate a rule chain for one entry. rsync precedence, highest first: the - * innermost (current) directory's .rsync-filter rules, then each ancestor's, - * then the root's, and finally the command-line base rules (--filter/-C). The - * sender-side verdict decides whether the entry is hidden from the transfer; - * the receiver-side verdict decides whether its destination mirror is protected - * from --delete. Each side takes the FIRST matching rule independently. */ -typedef struct { - bool hide; /* sender-side exclude matched */ - bool protect; /* receiver-side exclude matched */ -} FilterOutcome; - -static void chain_rules_outcome(const FilterRuleList* base, const FilterNode* node, const char* rel, - const char* leaf, bool is_dir, FilterOutcome* out) { - memset(out, 0, sizeof(*out)); - bool sender_decided = false; - bool receiver_decided = false; - const FilterNode* n = node; - while (!sender_decided || !receiver_decided) { - const FilterRuleList* list = n ? n->own : base; - if (list) { - if (!sender_decided) { - FilterAction action = filter_rules_apply_side(list, rel, leaf, is_dir, FILTER_SIDE_SENDER); - if (action != FILTER_ACTION_NONE) { - out->hide = action == FILTER_ACTION_EXCLUDE; - sender_decided = true; - } - } - if (!receiver_decided) { - FilterAction action = - filter_rules_apply_side(list, rel, leaf, is_dir, FILTER_SIDE_RECEIVER); - if (action != FILTER_ACTION_NONE) { - out->protect = action == FILTER_ACTION_PROTECT; - receiver_decided = true; - } - } - } - if (!n) - break; - n = n->parent; - } -} - -static bool entry_allowed(const FilterRuleList* base, const FilterNode* node, const char* rel, - const char* leaf, bool is_dir, bool exclude_filter_files, - bool* protect_out) { - /* -FF: per-directory .rsync-filter files are never transferred (single -F - transfers them, matching rsync). */ - if (exclude_filter_files && !is_dir && strcmp(leaf, ".rsync-filter") == 0) { - if (protect_out) - *protect_out = false; - return false; - } - FilterOutcome outcome; - chain_rules_outcome(base, node, rel, leaf, is_dir, &outcome); - if (protect_out) - *protect_out = outcome.protect; - return !outcome.hide; -} - -static void dir_entry_destroy(void* item) { - if (item) { - DirEntry* de = (DirEntry*)item; - free(de->path); - free(de); - } -} - -static DirEntry* dir_entry_create(const char* path, int depth, FilterNode* context) { - DirEntry* de = malloc(sizeof(DirEntry)); - if (!de) - return NULL; - de->path = str_dup(path); - if (!de->path) { - free(de); - return NULL; - } - de->depth = depth; - de->context = context; - return de; -} - -/* How rsync's readlink_stat()/generator resolves one source symlink. */ -typedef enum { - LINK_ACTION_SKIP, /* not transferred (no link option) */ - LINK_ACTION_SKIP_PROTECTED, /* ignored as unsafe by --safe-links; rsync keeps - it in the transfer, so its destination mirror - must be protected from --delete */ - LINK_ACTION_DEREF, /* follow the referent (--copy-links, an unsafe - target under --copy-unsafe-links, or -k dir) */ - LINK_ACTION_CARRY, /* transmit the link itself (-l) */ -} LinkAction; - -/* Apply rsync's symlink-resolution precedence to one S_ISLNK entry: - * --copy-links dereferences every symlink; - * --copy-unsafe-links dereferences only targets unsafe_symlink() flags; - * -k/--copy-dirlinks dereferences only a symlink whose referent is a dir; - * --safe-links (receiver-side in rsync; modelled here) ignores an unsafe - * target that would otherwise be carried; with --munge-links - * every stored target becomes absolute, so --safe-links then - * ignores every symlink, exactly as rsync documents; - * -l/--links carries the link. - * `link_rel` is the symlink's transfer-relative path (incl. name) and is used - * only for the lexical unsafe test. `target` receives the raw link value. */ -static LinkAction scanner_link_action(const ScannerOptions* options, const char* path, - const char* link_rel, char* target, size_t target_size) { - if (!options->follow_symlinks && !options->copy_links && !options->safe_links && - !options->copy_unsafe_links && !options->copy_dirlinks) - return LINK_ACTION_SKIP; - ssize_t length = readlink(path, target, target_size - 1); - if (length < 0) - return LINK_ACTION_SKIP; - target[length] = '\0'; - - bool unsafe = file_symlink_unsafe(target, link_rel); - if (options->copy_links || (options->copy_unsafe_links && unsafe)) - return LINK_ACTION_DEREF; - if (options->copy_dirlinks) { - struct stat ref; - if (stat(path, &ref) == 0 && S_ISDIR(ref.st_mode)) - return LINK_ACTION_DEREF; - } - if (options->safe_links && (unsafe || options->munge_links)) - return LINK_ACTION_SKIP_PROTECTED; - if (!options->follow_symlinks || target[0] == '\0') - return LINK_ACTION_SKIP; - return LINK_ACTION_CARRY; -} - -typedef struct { - char* path; - struct stat stats; - bool is_directory; - /* True when the entry should be carried through as a SYMLINK (is_symlink) - rather than a dereferenced file/directory. When true, `link_target` holds - the owned target string to transmit (sender-munged under --munge-links); - ownership transfers to the File built from this entry. */ - bool is_symlink; - char* link_target; - /* True when the entry was pruned by a user selection rule (--filter/-C/per-dir - rules or the --exclude/--include layer) rather than skipped for another - reason (unreadable, symlink policy, not applicable). */ - bool excluded; - /* True when the entry was skipped specifically by --max-size/--min-size. - Size pruning protects the destination mirror even under --delete-excluded, - so it is recorded into a separate sink from `excluded`. */ - bool size_excluded; - /* True when a symlink selected for dereferencing (-L/--copy-links or an - unsafe target under --copy-unsafe-links) had no usable referent (a broken - link or a stat() failure). rsync still reports this as a partial transfer - (exit 23) even though the entry is skipped, so the scanner records it as a - non-fatal I/O error. */ - bool referent_error; -} ScannerEntry; - /* One inspected directory entry buffered so the sequential scanner can emit the stream in rsync's flist order. `name` is the raw dirent name (owned here); `entry` is the scanner_inspect_entry() result whose path/link_target are owned @@ -216,522 +29,6 @@ typedef struct { int inspection; } SortedEntry; -/* --one-file-system (-x) decision. Only directories can carry a different - * device than their parent (mount points), so this is checked when a child - * directory is about to be descended into. */ -bool scanner_same_filesystem(int one_file_system, dev_t root_device, dev_t entry_device) { - return one_file_system <= 0 || entry_device == root_device; -} - -/* Build a payload-less directory File carrying the captured metadata (when - * requested). Used by -x mount-point emission and --list-only directory - * entries. Returns NULL on allocation failure. */ -static File* scanner_build_dir_file(const char* path, const struct stat* stats, - const ScannerOptions* options) { - File* dir = file_create(path); - if (dir == NULL) - return NULL; - dir->is_dir = true; - if (options->use_metadata) { - dir->metadata = - file_metadata_create(dir->path, stats, options->preserve_atimes, options->preserve_crtimes); - if (!dir->metadata) { - file_destroy(dir); - return NULL; - } - } - return dir; -} - -/* Relative path of an on-disk path below `root`. The transfer root may be - * given with a trailing slash; the returned rel path never has one and is "" - * for the root itself. A root of "/" is handled (its children start at "/"). - * Exposed so tests can exercise the mapping directly. */ -char* scanner_path_relative(const char* root, const char* fs_path) { - size_t root_len = strlen(root); - while (root_len > 1 && root[root_len - 1] == '/') - root_len--; - if (strncmp(root, fs_path, root_len) != 0) - return NULL; - if (root_len == 1 && root[0] == '/') { - if (fs_path[1] == '\0') - return str_dup(""); - return str_dup(fs_path + 1); - } - if (fs_path[root_len] == '\0') - return str_dup(""); - if (fs_path[root_len] != '/') - return NULL; - return str_dup(fs_path + root_len + 1); -} - -/* -R/--relative destination-relative prefix reconstructed from a source spec: - * everything after the first '.' path component (rsync's '/./' cut point), - * with leading/trailing slashes removed; or the whole spec (normalized) when - * there is no cut. Returns "" for the receive root. Exposed for tests. */ -char* scanner_relative_prefix(const char* spec) { - if (!spec || spec[0] == '\0') - return NULL; - const char* after = spec; - if (spec[0] == '.' && spec[1] == '/') { - after = spec + 2; - } else { - const char* cut = strstr(spec, "/./"); - if (cut) - after = cut + 3; - } - size_t cap = strlen(spec) + 1; - char* out = malloc(cap); - if (!out) - return NULL; - size_t len = 0; - for (const char* s = after; *s;) { - while (*s == '/') - s++; - const char* comp = s; - while (*s && *s != '/') - s++; - size_t clen = (size_t)(s - comp); - if (clen == 0 || (clen == 1 && comp[0] == '.')) - continue; - if (len) - out[len++] = '/'; - memcpy(out + len, comp, clen); - len += clen; - } - out[len] = '\0'; - return out; -} - -/* Relative path of a child entry below the current directory. */ -static char* child_rel_path(const char* parent_rel, const char* name) { - if (!parent_rel || parent_rel[0] == '\0') - return str_dup(name); - return path_cat(parent_rel, name); -} - -/* Destination-relative wire path for an entry under an -R prefix. */ -static char* scanner_prefix_send_path(const char* prefix, const char* rel) { - if (prefix[0] == '\0') - return str_dup(rel); - if (rel[0] == '\0') - return str_dup(prefix); - return path_cat(prefix, rel); -} - -/* Apply the --files-from allow-set and the filter layer to one entry. On - * return `*protect_out` is true when a receiver-side rule protects the entry's - * destination mirror from deletion. */ -static bool entry_passes_selection(const FileListSet* file_list, const FilterRuleList* base, - const FilterNode* node, const char* rel, const char* leaf, - bool is_dir, bool per_dir_filters, bool exclude_filter_files, - bool* protect_out) { - if (protect_out) - *protect_out = false; - if (file_list && !file_list_affects(file_list, rel)) - return false; - if (base || per_dir_filters) - return entry_allowed(base, node, rel, leaf, is_dir, exclude_filter_files, protect_out); - return true; -} - -/* Best-effort capture of the file's whitelisted xattrs (-X/-A). A failure to - * read xattrs is non-fatal: the file is transferred without them. */ -static void scanner_capture_xattrs(const DirectoryScanner* scanner, File* file) { - if (!scanner || !file || !(scanner->options.preserve_xattrs || scanner->options.preserve_acls)) - return; - file->xattrs = xattr_capture_path(file->path, scanner->options.preserve_acls); -} - -/* Apply --hard-links (-H) detection to one regular File. On a sibling (a - * later member of an already-seen source inode) the File keeps the group id - * and the first member's wire path but carries NO data payload (size 0); the - * first member is left untouched (data present, link_first). Allocation - * failure is fatal: the scanner is marked failed. */ -static void scanner_assign_hardlink(DirectoryScanner* scanner, HardLinkTable* table, File* file, - const struct stat* stats) { - if (!table || !file || !stats) - return; - int gid; - bool is_first; - char* first_path = NULL; - if (!hardlink_table_assign(table, file_wire_path(file), stats->st_dev, stats->st_ino, &gid, - &is_first, &first_path)) { - if (scanner) - scanner->failed = true; - return; - } - file->link_group = gid; - file->link_first = is_first; - if (!is_first) { - file->hardlink_target = first_path; - file->data->size = 0; - } else { - free(first_path); - } -} - -/* Phase 4 special/devices decision for one non-regular entry, matching rsync: - - a char/block device is RECREATED as a node under -D/--devices, unless - --copy-devices asks for its content to be copied into a regular file; - - a FIFO/socket is RECREATED under --specials; - - when the matching flag is absent the entry is SKIPPED ("skipping - non-regular file"), exactly like rsync's default, instead of being - silently copied as a zero-length regular file; - - anything else (regular/directory) is left to the normal data path. */ -typedef enum { - SCANNER_SPECIAL_REGULAR, /* ordinary file: transfer content */ - SCANNER_SPECIAL_RECREATE, /* is_special node to recreate on the receiver */ - SCANNER_SPECIAL_SKIP, /* non-regular entry not requested: skip */ -} ScannerSpecial; - -static ScannerSpecial scanner_prepare_special(bool preserve_devices, bool preserve_specials, - bool copy_devices, File* file, - const struct stat* stats) { - if (!file || !stats) - return SCANNER_SPECIAL_REGULAR; - bool is_device = S_ISCHR(stats->st_mode) || S_ISBLK(stats->st_mode); - bool is_fifo = S_ISFIFO(stats->st_mode); - bool is_socket = S_ISSOCK(stats->st_mode); - if (!is_device && !is_fifo && !is_socket) - return SCANNER_SPECIAL_REGULAR; - if (is_device && copy_devices) - return SCANNER_SPECIAL_REGULAR; /* copy device content as a regular file */ - bool preserve = is_device ? preserve_devices : preserve_specials; - if (!preserve) - return SCANNER_SPECIAL_SKIP; - file->is_special = true; - file->data->size = 0; - file->data->data = NULL; - if (is_device) { - file->rdev_major = (int32_t)major(stats->st_rdev); - file->rdev_minor = (int32_t)minor(stats->st_rdev); - } - return SCANNER_SPECIAL_RECREATE; -} - -/* Append `rel` to the caller's exclusion sink, taking `mtx` when shared across - parallel worker threads. Returns false on allocation failure (list left - unchanged). */ -static bool excluded_sink_append(ArrayList* list, mtx_t* mtx, const char* rel) { - if (!list) - return true; - char* dup = str_dup(rel); - if (!dup) - return false; - if (mtx) - mtx_lock(mtx); - bool ok = array_list_add(list, dup); - if (mtx) - mtx_unlock(mtx); - if (!ok) - free(dup); - return ok; -} - -/* Record one pruned filesystem path in a delete-protection sink. The stored - form is the entry's wire/destination-relative path (a single leading '/' - removed, exactly how manifest keep entries are stored), so the receiver's - walker prefixes match the destination layout. An allocation failure is a - fatal scan error. */ -static void scanner_record_protected(DirectoryScanner* scanner, const char* fs_path, - ArrayList* sink) { - if (!sink || !fs_path) - return; - const char* rel = *fs_path == '/' ? fs_path + 1 : fs_path; - if (!excluded_sink_append(sink, scanner->options.excluded_mutex, rel)) - scanner->failed = true; -} - -/* rsync's `--info=nonreg` line for a non-regular entry that is not being - * preserved: `skipping non-regular file "NAME"`. The name is the path relative - * to the transfer root, so it matches rsync's displayed name. */ -static void scanner_note_nonreg(const ScannerOptions* options, const char* fs_path) { - if (!options || !options->note_nonreg || !fs_path) - return; - const char* rel = utils_strip_transfer_root(fs_path, options->send_directory); - char* escaped = output_escape(rel, options->eight_bit_output); - printf("skipping non-regular file \"%s\"\n", escaped ? escaped : rel); - free(escaped); - fflush(stdout); -} - -/* rsync 3.4.1's `--info=mount` line, emitted when `-xx` drops a mount-point - * directory: `[sender] skipping mount-point dir NAME` (the client is the - * sender). Plain `-x` keeps the empty directory and prints nothing, matching - * rsync. */ -static void scanner_note_mount(const ScannerOptions* options, const char* fs_path) { - if (!options || !options->note_mount || !fs_path) - return; - const char* rel = utils_strip_transfer_root(fs_path, options->send_directory); - char* escaped = output_escape(rel, options->eight_bit_output); - printf("[sender] skipping mount-point dir %s\n", escaped ? escaped : rel); - free(escaped); - fflush(stdout); -} - -/* --debug=filter: a selection/filter decision dropped an entry. */ -static void scanner_note_filter(const ScannerOptions* options, const char* name) { - if (!options || !log_debug_enabled(LOG_DEBUG_FILTER) || !name) - return; - log_debug_message(LOG_DEBUG_FILTER, "filter: excluded %s", name); -} - -/* Account for a directory that will not be represented by an inline directory - * entry. Paired with scanner_dir_count_uncount for empty directories that are - * emitted inline, so every traversed directory is counted exactly once. */ -static void scanner_dir_count_count(const ScannerOptions* options) { - if (options && options->dir_count) - atomic_fetch_add(options->dir_count, 1); -} - -static void scanner_dir_count_uncount(const ScannerOptions* options) { - if (options && options->dir_count) - atomic_fetch_sub(options->dir_count, 1); -} - -/* A user-selection exclusion (--filter/-C/per-dir or --exclude/--include). */ -static void scanner_record_excluded(DirectoryScanner* scanner, const char* fs_path) { - scanner_record_protected(scanner, fs_path, scanner->options.excluded_paths); -} - -/* A --max-size/--min-size prune (always protected, even under --delete-excluded). */ -static void scanner_record_size_skipped(DirectoryScanner* scanner, const char* fs_path) { - scanner_record_protected(scanner, fs_path, scanner->options.size_skipped_paths); -} - -/* Record a directory the scan synchronized. `fs_path` is its absolute path and - `rel` its path relative to the transfer root ("" for the root); the stored - form matches the wire layout (the bare relative path in -R+--files-from, else - the source path with a leading '/' removed, with "." for the receive root). - Returns false on allocation failure. */ -static bool scanner_record_synced_dir(const ScannerOptions* options, const char* fs_path, - const char* rel, bool relative_mode) { - if (!options->synced_dirs && !options->plan_dirs) - return true; - if (!file_list_dir_in_scope(options->file_list, rel)) - return true; - char* prefixed = NULL; - const char* dest; - if (relative_mode) { - dest = rel; - } else if (options->relative_prefix) { - prefixed = scanner_prefix_send_path(options->relative_prefix, rel); - if (!prefixed) - return false; - dest = prefixed; - } else { - dest = fs_path; - } - if (dest[0] == '/') - dest++; - if (dest[0] == '\0') - dest = "."; - bool ok = true; - if (options->synced_dirs) - ok = excluded_sink_append(options->synced_dirs, options->excluded_mutex, dest); - /* The delete-plan keep set needs an entry for every traversed source - directory, including empty ones, so its destination mirror is kept rather - than deleted as an extra; the receive root (".") is implicit. */ - if (ok && options->plan_dirs && strcmp(dest, ".") != 0) - ok = excluded_sink_append(options->plan_dirs, options->excluded_mutex, dest); - free(prefixed); - return ok; -} - -/* Read every per-directory filter file that applies to `dir_path` (its - * .rsync-filter when -F is active, plus each registered "dir-merge NAME") into a - * fresh list. Returns NULL on allocation/parse failure (message in `err`); - * returns an empty list (and *any_exists=false) when no file exists. */ -static FilterRuleList* read_dir_filters(const ScannerOptions* options, const char* dir_path, - const char* rel, bool* any_exists, char* err, - size_t err_size) { - if (err && err_size > 0) - err[0] = '\0'; - const FilterRuleList* base = options->base_filters; - bool have_names = options->per_dir_filters || (base && base->dir_merge_count > 0); - if (any_exists) - *any_exists = false; - if (!have_names) - return NULL; - FilterRuleList* own = filter_rule_list_create(); - if (!own) { - snprintf(err, err_size, "memory allocation failed"); - return NULL; - } - FilterParseOptions opts = {.delete_excluded = options->delete_excluded, .cvs_exclude = false}; - bool exists = false; - if (options->per_dir_filters) { - if (!filter_file_append(own, dir_path, ".rsync-filter", rel, &opts, &exists, err, err_size)) - goto fail; - if (exists && any_exists) - *any_exists = true; - } - if (base) { - for (int i = 0; i < base->dir_merge_count; i++) { - if (!filter_file_append(own, dir_path, base->dir_merge_names[i], rel, &opts, &exists, err, - err_size)) - goto fail; - if (exists && any_exists) - *any_exists = true; - } - } - return own; -fail: - filter_rule_list_free(own); - return NULL; -} - -/* Merge the open directory's own per-directory filter files (the default - * .rsync-filter when -F is active, plus every "dir-merge NAME" registered on the - * base rule list) into the inherited context, returning the context used for - * this directory's entries. On a parse error the scanner is marked failed. - * Returns 0 on success, -1 on failure. */ -static int open_directory_filter_context(DirectoryScanner* scanner, const FilterNode* inherited) { - char err[256]; - bool any_exists = false; - FilterRuleList* own = read_dir_filters(&scanner->options, scanner->current_path, - scanner->current_rel ? scanner->current_rel : "", - &any_exists, err, sizeof(err)); - if (!own) { - /* read_dir_filters() leaves `err` set on a parse/allocation failure even - when an earlier merge file in the same directory existed (any_exists true); - key off the error text rather than any_exists so an invalid per-directory - filter file can never be silently ignored. */ - if (err[0] == '\0') { - scanner->current_node = (FilterNode*)inherited; - return 0; - } - char* escaped_path = output_escape(scanner->current_path, log_get_8_bit_output()); - log_message(LOG_LEVEL_ERROR, "invalid per-directory filter in %s: %s", - escaped_path ? escaped_path : "", err); - free(escaped_path); - scanner->failed = true; - return -1; - } - if (any_exists && (own->count > 0 || own->dir_merge_count > 0)) { - FilterNode* node = filter_node_alloc((FilterNode*)inherited, own); - if (!node || !array_list_add(scanner->filter_nodes, node)) { - filter_node_destroy(node); - scanner->failed = true; - return -1; - } - scanner->current_node = node; - } else { - filter_rule_list_free(own); - scanner->current_node = (FilterNode*)inherited; - } - return 0; -} - -/* Inspect symlinks, resolve the entry type, and apply file filters once for both scanners. - * `link_rel` is the entry's path relative to the transfer root (including its - * name), used for the lexical rsync unsafe-symlink test. */ -static int scanner_inspect_entry(const ScannerOptions* options, const char* containing_dir, - const char* link_rel, const char* name, ScannerEntry* entry) { - entry->excluded = false; - entry->size_excluded = false; - entry->referent_error = false; - entry->is_symlink = false; - entry->link_target = NULL; - entry->path = path_cat(containing_dir, name); - if (!entry->path) - return -1; - - struct stat link_stats; - if (lstat(entry->path, &link_stats) != 0) { - free(entry->path); - return 0; - } - if (!S_ISLNK(link_stats.st_mode)) - goto regular; - - char link_target[4096]; - switch (scanner_link_action(options, entry->path, link_rel, link_target, sizeof(link_target))) { - case LINK_ACTION_SKIP: - goto skip; - case LINK_ACTION_SKIP_PROTECTED: - /* --safe-links ignored the link, but rsync still counts it as present in - the transfer, so its destination mirror survives --delete. Record it as - an excluded path (the same delete-protection channel as a filter prune). */ - entry->excluded = true; - goto skip; - case LINK_ACTION_DEREF: - if (stat(entry->path, &entry->stats) != 0) { - /* rsync reports "symlink has no referent" and continues with a partial - transfer (exit 23); record the error so the run exits 23 too. */ - char* escaped = output_escape(entry->path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, "symlink has no referent: %s", - escaped ? escaped : ""); - free(escaped); - entry->referent_error = true; - goto skip; - } - entry->is_directory = S_ISDIR(entry->stats.st_mode); - if (entry->is_directory) - return 1; - goto apply_filters; - case LINK_ACTION_CARRY: - break; - } - - /* Carry the link as a symlink. --munge-links is applied by the RECEIVER (it - prefixes every stored target with /rsyncd-munged/); when the SOURCE already - holds a munged value the sender strips it so the receiver re-munges a clean - target, round-tripping a munged tree exactly like rsync. */ - entry->is_symlink = true; - entry->stats = link_stats; - entry->is_directory = false; - entry->link_target = str_dup(link_target); - if (!entry->link_target) - goto skip; - if (options->munge_links) - file_symlink_unmunge(entry->link_target); - goto apply_filters; - -regular: - /* Not a symlink: the lstat() above already described this entry, and lstat - and stat are identical for every non-symlink, so reuse that result instead - of issuing a redundant stat() on the scanner hot path. stat() is still - used on the dereference paths above/below for actual symlinks (copy-links, - safe/copy-unsafe links, and -k symlinks-to-directories). */ - entry->stats = link_stats; - entry->is_directory = S_ISDIR(link_stats.st_mode); - if (entry->is_directory) - return 1; - -apply_filters: - for (int i = 0; i < options->exclude_count; i++) - if (glob_match(options->exclude_patterns[i], name)) { - entry->excluded = true; - goto skip; - } - if (options->include_count > 0) { - bool included = false; - for (int i = 0; i < options->include_count; i++) - if (glob_match(options->include_patterns[i], name)) - included = true; - if (!included) { - entry->excluded = true; - goto skip; - } - } - if ((options->max_size > 0 && (unsigned long long)entry->stats.st_size > options->max_size) || - (options->min_size > 0 && (unsigned long long)entry->stats.st_size < options->min_size)) { - entry->excluded = true; - entry->size_excluded = true; - goto skip; - } - return 1; - -skip: - free(entry->path); - entry->path = NULL; - free(entry->link_target); - entry->link_target = NULL; - return 0; -} - static void sorted_entry_destroy(void* item) { SortedEntry* se = (SortedEntry*)item; if (!se) @@ -1012,12 +309,11 @@ static Chunk* chunk_data_to_chunk(ArrayList* chunk_data) { * append for the parallel scanner's shared workers. An unstattable or * non-directory path is silently skipped (the transfer is unaffected); an * allocation failure is fatal and reported to the caller. */ -static bool scanner_capture_dir_time(ArrayList* dir_entries, mtx_t* mutex, const char* root_path, - const char* fs_path, bool relative_mode, - const char* relative_prefix, bool preserve_atimes, - bool preserve_crtimes, bool preserve_xattrs, - bool preserve_acls, bool no_implied_dirs, - const FileListSet* file_list) { +bool scanner_capture_dir_time(ArrayList* dir_entries, mtx_t* mutex, const char* root_path, + const char* fs_path, bool relative_mode, const char* relative_prefix, + bool preserve_atimes, bool preserve_crtimes, bool preserve_xattrs, + bool preserve_acls, bool no_implied_dirs, + const FileListSet* file_list) { if (!dir_entries || !root_path || !fs_path) return true; struct stat st; @@ -1588,6 +884,296 @@ static Chunk* directory_scanner_next_dirs(DirectoryScanner* scanner) { return dirs_flush_batch(scanner); } +/* Result of processing one inspected entry inside directory_scanner_next(). */ +typedef enum { + SCANNER_ACTION_CONTINUE, /* advance to the next buffered entry */ + SCANNER_ACTION_BREAK, /* stop the scan loop (failure recorded) */ + SCANNER_ACTION_CHUNK, /* return the Chunk produced in *out_chunk */ +} ScannerAction; + +/* Reconstruct the delete-protection path for a skipped (inspection == 0) entry + * whose destination mirror must be protected. */ +static char* scanner_entry_protected_path(DirectoryScanner* scanner, const char* name) { + if (scanner->relative_mode) + return child_rel_path(scanner->current_rel, name); + if (scanner->options.relative_prefix) { + char* relc = child_rel_path(scanner->current_rel, name); + char* prefixed = relc ? scanner_prefix_send_path(scanner->options.relative_prefix, relc) : NULL; + free(relc); + return prefixed; + } + return path_cat(scanner->current_path, name); +} + +/* Handle a buffered entry that scanner_inspect_entry() skipped (inspection == + * 0): record a partial-transfer I/O error and protect the destination mirror + * of a user-selection or size prune. Returns 0 to continue, -1 on failure. */ +static int scanner_handle_skipped_entry(DirectoryScanner* scanner, const ScannerEntry* inspected, + const char* name) { + /* A dereferenced symlink with no referent is a partial-transfer error + (rsync exit 23): record it as a non-fatal scan I/O error. */ + if (inspected->referent_error) + scanner->io_error = true; + /* A user-selection exclude protects its destination mirror from --delete + unless --delete-excluded; a size prune is always protected. Other + skips (unreadable, symlink policy) protect nothing. Under -R + + --files-from the protected prefix must be the entry's bare relative + wire path, not its source path (which would not match the destination + layout and would leave the mirror deletable). */ + if (inspected->excluded) { + char* protected_path = scanner_entry_protected_path(scanner, name); + if (!protected_path) { + scanner->failed = true; + return -1; + } + if (inspected->size_excluded) + scanner_record_size_skipped(scanner, protected_path); + else + scanner_record_excluded(scanner, protected_path); + free(protected_path); + } + return 0; +} + +/* Record the delete-protection prefix for an entry dropped by the --files-from + * allow-set or a filter rule (sender-hide or receiver-protect). Returns 0 on + * success, -1 on allocation failure (caller reports it). */ +static int scanner_record_selection_protection(DirectoryScanner* scanner, bool protect, + bool passes_selection, const char* rel, + const char* cur_path) { + if (passes_selection && !protect) + return 0; + /* --files-from subset pruning is not a filter exclusion: its delete + semantics stay keep-set-only (an unlisted source path is treated as + absent, so its destination mirror is a deletable extra). A rule-based + exclusion is recorded as a protected prefix. -R + --files-from bare + wire paths are never recorded (see ScannerOptions.excluded_paths). */ + bool files_from_prune = + scanner->options.file_list && !file_list_affects(scanner->options.file_list, rel); + if (protect && scanner->relative_mode) { + /* -R + --files-from: the destination/wire path is the bare relative + name, so the protected mirror prefix must be `rel` (not the source + path) for the delete walker to match it. */ + scanner_record_excluded(scanner, rel); + } else if (!files_from_prune && !scanner->relative_mode) { + if (scanner->options.relative_prefix) { + char* wrel = scanner_prefix_send_path(scanner->options.relative_prefix, rel); + if (!wrel) + return -1; + scanner_record_excluded(scanner, wrel); + free(wrel); + } else { + scanner_record_excluded(scanner, cur_path); + } + } + return 0; +} + +/* -x/--one-file-system handling for a directory entry: 1 when the entry was + * fully handled (caller continues), 0 when it is on the same filesystem as the + * root (caller descends), -1 on a fatal allocation failure. */ +static int scanner_handle_mount_dir(DirectoryScanner* scanner, ArrayList* chunk_data, + const char* cur_path, const struct stat* stats) { + if (scanner_same_filesystem(scanner->options.one_file_system, scanner->root_dev, stats->st_dev)) + return 0; + if (scanner->options.one_file_system > 1) { + /* rsync's -xx drops the mount-point directory entirely (the plain -x + path below keeps it as an empty directory) and prints the + --info=mount line when that category is enabled. */ + scanner_note_mount(&scanner->options, cur_path); + return 1; + } + /* rsync's -x/--one-file-system emits the mount-point directory entry + itself (so the destination gets an empty directory) but does NOT + descend into it. Build a payload-less directory File and hand it to + the caller; never enqueue it for traversal. */ + File* mount = scanner_build_dir_file(cur_path, stats, &scanner->options); + if (mount == NULL || !array_list_add(chunk_data, mount)) { + file_destroy(mount); + scanner->failed = true; + return -1; + } + scanner->current_dir_produced = true; + return 1; +} + +/* Finish the open directory (exhausted entries): emit an empty-directory entry + * when appropriate, push its pending children and reset the per-directory + * state. Returns false when the scanner failed. */ +static bool scanner_finish_current_directory(DirectoryScanner* scanner, ArrayList* chunk_data) { + /* The directory is exhausted: if nothing was transferred or descended + from it, recreate it at the destination as an explicit entry. */ + if (scanner->options.emit_empty_dirs && !scanner->current_dir_produced && + !scanner->options.prune_empty_dirs && !scanner->options.list_dirs && + scanner->options.file_list == NULL) { + if (!scanner_emit_empty_dir(scanner, chunk_data)) + scanner->failed = true; + } + scanner_push_pending_dirs(scanner); + closedir(scanner->current_dir); + scanner->current_dir = NULL; + free(scanner->current_path); + scanner->current_path = NULL; + scanner_free_sorted(scanner); + return !scanner->failed; +} + +/* Handle one kept buffered entry (inspection == 1): selection recording, + * directory descent and regular-file emission. `*chunk_data_size` tracks the + * accumulated payload so a chunk is cut at the same point as before. */ +static ScannerAction scanner_process_entry(DirectoryScanner* scanner, ArrayList* chunk_data, + unsigned long long* chunk_data_size, SortedEntry* sorted, + Chunk** out_chunk) { + const char* name = sorted->name; + ScannerEntry* inspected = &sorted->entry; + char* cur_path = inspected->path; + struct stat stats = inspected->stats; + + /* --files-from allow-set and the filter layer apply to files and to + * directories (an excluded directory is not descended into). */ + bool is_dir = inspected->is_directory; + char* rel = child_rel_path(scanner->current_rel, name); + if (!rel) { + scanner->failed = true; + return SCANNER_ACTION_BREAK; + } + bool protect = false; + bool passes_selection = entry_passes_selection( + scanner->options.file_list, scanner->options.base_filters, scanner->current_node, rel, name, + is_dir, scanner->options.per_dir_filters, scanner->options.exclude_per_dir_filter_files, + &protect); + /* A sender-side hide leaves the entry out of the transfer; an independent + receiver-side protect rule keeps a transferred entry's destination mirror + from being deleted. Both are recorded in the same protection set. */ + if (!passes_selection || protect) { + if (scanner_record_selection_protection(scanner, protect, passes_selection, rel, cur_path) != + 0) { + free(rel); + scanner->failed = true; + return SCANNER_ACTION_BREAK; + } + } + /* With -R the wire/destination path is a reconstructed relative path, not + the source path; keep `rel` alive to build it for a transferred file. */ + bool needs_rel = scanner->relative_mode || scanner->options.relative_prefix != NULL; + char* rel_copy = needs_rel ? str_dup(rel) : NULL; + free(rel); + if (rel_copy == NULL && needs_rel) { + scanner->failed = true; + return SCANNER_ACTION_BREAK; + } + if (!passes_selection) { + scanner_note_filter(&scanner->options, name); + free(rel_copy); + return SCANNER_ACTION_CONTINUE; + } + + if (is_dir) { + free(rel_copy); + int mount = scanner_handle_mount_dir(scanner, chunk_data, cur_path, &stats); + if (mount < 0) + return SCANNER_ACTION_BREAK; + if (mount > 0) + return SCANNER_ACTION_CONTINUE; + /* --list-only: list directory entries too (rsync prints them), even + though a real transfer never sends them explicitly. */ + if (scanner->options.list_dirs) { + File* dir = scanner_build_dir_file(cur_path, &stats, &scanner->options); + if (dir == NULL || !array_list_add(chunk_data, dir)) { + file_destroy(dir); + scanner->failed = true; + return SCANNER_ACTION_BREAK; + } + } + scanner->current_dir_produced = true; + int next_depth = scanner->current_depth + 1; + if (scanner->options.max_depth <= 0 || next_depth < scanner->options.max_depth) { + DirEntry* de = dir_entry_create(cur_path, next_depth, scanner->current_node); + if (!de || !array_list_add((ArrayList*)scanner->pending_dirs, de)) { + dir_entry_destroy(de); + scanner->failed = true; + } + } + return SCANNER_ACTION_CONTINUE; + } + + if (scanner->options.max_depth > 0 && scanner->current_depth + 1 > scanner->options.max_depth) { + free(rel_copy); + return SCANNER_ACTION_CONTINUE; + } + File* file = file_create(cur_path); + if (file == NULL) { + free(rel_copy); + free(inspected->link_target); + inspected->link_target = NULL; + scanner->failed = true; + return SCANNER_ACTION_CONTINUE; + } + if (inspected->is_symlink) { + file->is_symlink = true; + file->symlink_target = inspected->link_target; + inspected->link_target = NULL; + } else { + file->data->size = stats.st_size; + } + if (scanner->relative_mode) { + file->send_path = rel_copy; + rel_copy = NULL; + } else if (scanner->options.relative_prefix) { + file->send_path = scanner_prefix_send_path(scanner->options.relative_prefix, rel_copy); + free(rel_copy); + rel_copy = NULL; + if (!file->send_path) { + file_destroy(file); + scanner->failed = true; + return SCANNER_ACTION_BREAK; + } + } + /* --devices/--specials: a device/FIFO/socket entry marked for preservation + becomes a node to recreate (is_special, no data, rdev captured); an + unrequested non-regular entry is skipped (rsync default). */ + ScannerSpecial special = + scanner_prepare_special(scanner->options.preserve_devices, scanner->options.preserve_specials, + scanner->options.copy_devices, file, &stats); + if (special == SCANNER_SPECIAL_SKIP) { + scanner_note_nonreg(&scanner->options, file->path); + free(rel_copy); + file_destroy(file); + return SCANNER_ACTION_CONTINUE; + } + if (scanner->options.hardlinks && S_ISREG(stats.st_mode)) + scanner_assign_hardlink(scanner, scanner->options.hardlinks, file, &stats); + if (scanner->options.use_metadata) + file->metadata = file_metadata_create(file->path, &stats, scanner->options.preserve_atimes, + scanner->options.preserve_crtimes); + if (scanner->options.use_metadata && !file->metadata) { + free(rel_copy); + file_destroy(file); + scanner->failed = true; + return SCANNER_ACTION_BREAK; + } + if (!(file->link_group != 0 && !file->link_first)) + scanner_capture_xattrs(scanner, file); + if (!array_list_add(chunk_data, file)) { + free(rel_copy); + file_destroy(file); + scanner->failed = true; + return SCANNER_ACTION_BREAK; + } + scanner->current_dir_produced = true; + *chunk_data_size += file->data->size; + if (*chunk_data_size > scanner->options.chunk_size) { + free(rel_copy); + Chunk* result = chunk_data_to_chunk(chunk_data); + if (!result) + scanner->failed = true; + *out_chunk = result; + return SCANNER_ACTION_CHUNK; + } + free(rel_copy); + return SCANNER_ACTION_CONTINUE; +} + Chunk* directory_scanner_next(DirectoryScanner* scanner) { if (scanner && scanner->options.dirs) return directory_scanner_next_dirs(scanner); @@ -1613,21 +1199,7 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { } if (scanner->sorted_index >= scanner->sorted_count) { - /* The directory is exhausted: if nothing was transferred or descended - from it, recreate it at the destination as an explicit entry. */ - if (scanner->options.emit_empty_dirs && !scanner->current_dir_produced && - !scanner->options.prune_empty_dirs && !scanner->options.list_dirs && - scanner->options.file_list == NULL) { - if (!scanner_emit_empty_dir(scanner, chunk_data)) - scanner->failed = true; - } - scanner_push_pending_dirs(scanner); - closedir(scanner->current_dir); - scanner->current_dir = NULL; - free(scanner->current_path); - scanner->current_path = NULL; - scanner_free_sorted(scanner); - if (scanner->failed) { + if (!scanner_finish_current_directory(scanner, chunk_data)) { array_list_delete(chunk_data); return NULL; } @@ -1635,226 +1207,21 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { } SortedEntry* sorted = &((SortedEntry*)scanner->sorted_entries)[scanner->sorted_index++]; - const char* name = sorted->name; - ScannerEntry* inspected = &sorted->entry; int inspection = sorted->inspection; if (inspection == 0) { - /* A dereferenced symlink with no referent is a partial-transfer error - (rsync exit 23): record it as a non-fatal scan I/O error. */ - if (inspected->referent_error) - scanner->io_error = true; - /* A user-selection exclude protects its destination mirror from --delete - unless --delete-excluded; a size prune is always protected. Other - skips (unreadable, symlink policy) protect nothing. Under -R + - --files-from the protected prefix must be the entry's bare relative - wire path, not its source path (which would not match the destination - layout and would leave the mirror deletable). */ - if (inspected->excluded) { - char* protected_path; - if (scanner->relative_mode) { - protected_path = child_rel_path(scanner->current_rel, name); - } else if (scanner->options.relative_prefix) { - char* relc = child_rel_path(scanner->current_rel, name); - protected_path = - relc ? scanner_prefix_send_path(scanner->options.relative_prefix, relc) : NULL; - free(relc); - } else { - protected_path = path_cat(scanner->current_path, name); - } - if (!protected_path) { - scanner->failed = true; - break; - } - if (inspected->size_excluded) - scanner_record_size_skipped(scanner, protected_path); - else - scanner_record_excluded(scanner, protected_path); - free(protected_path); - } - continue; - } - char* cur_path = inspected->path; - struct stat stats = inspected->stats; - - /* --files-from allow-set and the filter layer apply to files and to - * directories (an excluded directory is not descended into). */ - bool is_dir = inspected->is_directory; - char* rel = child_rel_path(scanner->current_rel, name); - if (!rel) { - scanner->failed = true; - break; - } - bool protect = false; - bool passes_selection = entry_passes_selection( - scanner->options.file_list, scanner->options.base_filters, scanner->current_node, rel, name, - is_dir, scanner->options.per_dir_filters, scanner->options.exclude_per_dir_filter_files, - &protect); - /* A sender-side hide leaves the entry out of the transfer; an independent - receiver-side protect rule keeps a transferred entry's destination mirror - from being deleted. Both are recorded in the same protection set. */ - if (!passes_selection || protect) { - /* --files-from subset pruning is not a filter exclusion: its delete - semantics stay keep-set-only (an unlisted source path is treated as - absent, so its destination mirror is a deletable extra). A rule-based - exclusion is recorded as a protected prefix. -R + --files-from bare - wire paths are never recorded (see ScannerOptions.excluded_paths). */ - bool files_from_prune = - scanner->options.file_list && !file_list_affects(scanner->options.file_list, rel); - if (protect && scanner->relative_mode) { - /* -R + --files-from: the destination/wire path is the bare relative - name, so the protected mirror prefix must be `rel` (not the source - path) for the delete walker to match it. */ - scanner_record_excluded(scanner, rel); - } else if (!files_from_prune && !scanner->relative_mode) { - if (scanner->options.relative_prefix) { - char* wrel = scanner_prefix_send_path(scanner->options.relative_prefix, rel); - if (!wrel) { - free(rel); - scanner->failed = true; - break; - } - scanner_record_excluded(scanner, wrel); - free(wrel); - } else { - scanner_record_excluded(scanner, cur_path); - } - } - } - /* With -R the wire/destination path is a reconstructed relative path, not - the source path; keep `rel` alive to build it for a transferred file. */ - bool needs_rel = scanner->relative_mode || scanner->options.relative_prefix != NULL; - char* rel_copy = needs_rel ? str_dup(rel) : NULL; - free(rel); - if (rel_copy == NULL && needs_rel) { - scanner->failed = true; - break; - } - if (!passes_selection) { - scanner_note_filter(&scanner->options, name); - free(rel_copy); + if (scanner_handle_skipped_entry(scanner, &sorted->entry, sorted->name) != 0) + break; continue; } - if (is_dir) { - free(rel_copy); - if (!scanner_same_filesystem(scanner->options.one_file_system, scanner->root_dev, - stats.st_dev)) { - if (scanner->options.one_file_system > 1) { - /* rsync's -xx drops the mount-point directory entirely (the plain -x - path below keeps it as an empty directory) and prints the - --info=mount line when that category is enabled. */ - scanner_note_mount(&scanner->options, cur_path); - continue; - } - /* rsync's -x/--one-file-system emits the mount-point directory entry - itself (so the destination gets an empty directory) but does NOT - descend into it. Build a payload-less directory File and hand it to - the caller; never enqueue it for traversal. */ - File* mount = scanner_build_dir_file(cur_path, &stats, &scanner->options); - if (mount == NULL || !array_list_add(chunk_data, mount)) { - file_destroy(mount); - scanner->failed = true; - break; - } - scanner->current_dir_produced = true; - continue; - } - /* --list-only: list directory entries too (rsync prints them), even - though a real transfer never sends them explicitly. */ - if (scanner->options.list_dirs) { - File* dir = scanner_build_dir_file(cur_path, &stats, &scanner->options); - if (dir == NULL || !array_list_add(chunk_data, dir)) { - file_destroy(dir); - scanner->failed = true; - break; - } - } - scanner->current_dir_produced = true; - int next_depth = scanner->current_depth + 1; - if (scanner->options.max_depth <= 0 || next_depth < scanner->options.max_depth) { - DirEntry* de = dir_entry_create(cur_path, next_depth, scanner->current_node); - if (!de || !array_list_add((ArrayList*)scanner->pending_dirs, de)) { - dir_entry_destroy(de); - scanner->failed = true; - } - } - } else { - if (scanner->options.max_depth > 0 && - scanner->current_depth + 1 > scanner->options.max_depth) { - free(rel_copy); - continue; - } - File* file = file_create(cur_path); - if (file == NULL) { - free(rel_copy); - free(inspected->link_target); - inspected->link_target = NULL; - scanner->failed = true; - continue; - } - if (inspected->is_symlink) { - file->is_symlink = true; - file->symlink_target = inspected->link_target; - inspected->link_target = NULL; - } else { - file->data->size = stats.st_size; - } - if (scanner->relative_mode) { - file->send_path = rel_copy; - rel_copy = NULL; - } else if (scanner->options.relative_prefix) { - file->send_path = scanner_prefix_send_path(scanner->options.relative_prefix, rel_copy); - free(rel_copy); - rel_copy = NULL; - if (!file->send_path) { - file_destroy(file); - scanner->failed = true; - break; - } - } - /* --devices/--specials: a device/FIFO/socket entry marked for preservation - becomes a node to recreate (is_special, no data, rdev captured); an - unrequested non-regular entry is skipped (rsync default). */ - ScannerSpecial special = scanner_prepare_special(scanner->options.preserve_devices, - scanner->options.preserve_specials, - scanner->options.copy_devices, file, &stats); - if (special == SCANNER_SPECIAL_SKIP) { - scanner_note_nonreg(&scanner->options, file->path); - free(rel_copy); - file_destroy(file); - continue; - } - if (scanner->options.hardlinks && S_ISREG(stats.st_mode)) - scanner_assign_hardlink(scanner, scanner->options.hardlinks, file, &stats); - if (scanner->options.use_metadata) - file->metadata = file_metadata_create(file->path, &stats, scanner->options.preserve_atimes, - scanner->options.preserve_crtimes); - if (scanner->options.use_metadata && !file->metadata) { - free(rel_copy); - file_destroy(file); - scanner->failed = true; - break; - } - if (!(file->link_group != 0 && !file->link_first)) - scanner_capture_xattrs(scanner, file); - if (!array_list_add(chunk_data, file)) { - free(rel_copy); - file_destroy(file); - scanner->failed = true; - break; - } - scanner->current_dir_produced = true; - chunk_data_size += file->data->size; - if (chunk_data_size > scanner->options.chunk_size) { - free(rel_copy); - Chunk* result = chunk_data_to_chunk(chunk_data); - if (!result) - scanner->failed = true; - return result; - } - free(rel_copy); - } + Chunk* result = NULL; + ScannerAction action = + scanner_process_entry(scanner, chunk_data, &chunk_data_size, sorted, &result); + if (action == SCANNER_ACTION_CHUNK) + return result; + if (action == SCANNER_ACTION_BREAK) + break; } if (chunk_data->size > 0) { @@ -1874,687 +1241,3 @@ bool directory_scanner_failed(const DirectoryScanner* scanner) { bool directory_scanner_had_io_error(const DirectoryScanner* scanner) { return scanner != NULL && (scanner->io_error || scanner->root_io_error); } - -typedef struct { - ParallelScanner* ps; - char** dirs; - int dir_count; - char* root_dir; /* the transfer root, for relative-path computation */ - ScannerOptions options; - ProtocolSession* allocation_session; -} ParallelWorkerArg; - -static int parallel_worker_thread(void* arg) { - ParallelWorkerArg* wa = (ParallelWorkerArg*)arg; - ProtocolSession* allocation_session = wa->allocation_session; - if (allocation_session) - protocol_session_bind(allocation_session); - for (int i = 0; i < wa->dir_count; i++) { - DirectoryScanner* ds = directory_scanner_create_with_options(wa->dirs[i], &wa->options); - if (!ds) { - mtx_lock(&wa->ps->result_mutex); - wa->ps->failed = true; - atomic_store(&wa->ps->cancelled, true); - cnd_broadcast(&wa->ps->result_not_empty); - cnd_broadcast(&wa->ps->result_not_full); - mtx_unlock(&wa->ps->result_mutex); - for (int j = i; j < wa->dir_count; j++) - free(wa->dirs[j]); - break; - } - /* Root .rsync-filter rules (parsed by the parallel scanner) apply to the - * contents of every assigned subdirectory. Relative paths (used by the - * allow-set and per-directory rules) are computed against the transfer - * root, not the subdirectory the worker is seeded with. Exclusion - * recording shares one caller-owned list across the workers. */ - free(ds->root_path); - ds->root_path = str_dup(wa->root_dir); - ds->seed_node = wa->ps->root_filter_node; - ds->options.excluded_mutex = &wa->ps->result_mutex; - Chunk* chunk; - while ((chunk = directory_scanner_next(ds)) != NULL) { - if (!queue_enqueue_multithreaded_cancel(wa->ps->result_queue, chunk, &wa->ps->result_mutex, - &wa->ps->result_not_empty, &wa->ps->result_not_full, - &wa->ps->cancelled)) { - chunk_destroy(chunk); - break; - } - } - if (directory_scanner_failed(ds)) { - mtx_lock(&wa->ps->result_mutex); - wa->ps->failed = true; - atomic_store(&wa->ps->cancelled, true); - cnd_broadcast(&wa->ps->result_not_empty); - cnd_broadcast(&wa->ps->result_not_full); - mtx_unlock(&wa->ps->result_mutex); - } else if (directory_scanner_had_io_error(ds)) { - /* --ignore-errors path: an unreadable directory was skipped, not fatal. */ - mtx_lock(&wa->ps->result_mutex); - wa->ps->io_error = true; - mtx_unlock(&wa->ps->result_mutex); - } - directory_scanner_destroy(ds); - free(wa->dirs[i]); - } - ParallelScanner* ps = wa->ps; - free(wa->root_dir); - free(wa->dirs); - free(wa); - mtx_lock(&ps->result_mutex); - ps->completed++; - if (ps->completed >= ps->expected_threads) { - ps->done = true; - cnd_signal(&ps->result_not_empty); - } - mtx_unlock(&ps->result_mutex); - if (allocation_session) - protocol_session_unbind(); - return thrd_success; -} - -static void parallel_scanner_creation_failed(ParallelScanner* ps) { - mtx_lock(&ps->result_mutex); - ps->failed = true; - atomic_store(&ps->cancelled, true); - ps->expected_threads = ps->created_threads; - if (ps->completed >= ps->expected_threads) - ps->done = true; - cnd_broadcast(&ps->result_not_empty); - cnd_broadcast(&ps->result_not_full); - mtx_unlock(&ps->result_mutex); -} - -/* Initialize result queue and synchronization primitives. Returns true on success. */ -static bool parallel_scanner_init(ParallelScanner* ps) { - ps->result_queue = queue_create(100, chunk_destroy); - if (!ps->result_queue) - return false; - atomic_init(&ps->cancelled, false); - int init = 0; - bool ok = true; - if (mtx_init(&ps->result_mutex, mtx_plain) != thrd_success) - ok = false; - if (ok) { - init++; - if (cnd_init(&ps->result_not_empty) != thrd_success) - ok = false; - } - if (ok) { - // cppcheck-suppress unreadVariable - init++; - if (cnd_init(&ps->result_not_full) != thrd_success) - ok = false; - } - if (!ok) { - if (init >= 3) - cnd_destroy(&ps->result_not_full); - if (init >= 2) - cnd_destroy(&ps->result_not_empty); - if (init >= 1) - mtx_destroy(&ps->result_mutex); - queue_destroy(ps->result_queue); - ps->result_queue = NULL; - return false; - } - return true; -} - -/* Split files into chunks of roughly chunk_size bytes. Returns the first chunk (also stored - * chunks beyond the first are enqueued on `queue`). Nulls out consumed entries in `files`. - * Sets *failed on allocation/enqueue errors. */ -static Chunk* batch_files(ArrayList* files, unsigned long long chunk_size, Queue* queue, - bool* failed) { - Chunk* first = NULL; - if (files->size <= 0) - return NULL; - ArrayList* batch = array_list_create(NULL); - if (!batch) { - *failed = true; - return NULL; - } - unsigned long long batch_size = 0; - for (int i = 0; i < files->size; i++) { - File* f = (File*)files->items[i]; - if (!array_list_add(batch, f)) { - *failed = true; - break; - } - batch_size += f->data->size; - if (batch_size >= chunk_size || i == files->size - 1) { - void** items = array_list_to_array(batch); - if (!items) { - *failed = true; - array_list_delete(batch); - batch = NULL; - break; - } - Chunk* c = chunk_create((File**)items, batch->size); - free(items); - if (!c) { - *failed = true; - array_list_delete(batch); - batch = NULL; - break; - } - int batch_start = i - batch->size + 1; - for (int j = batch_start; j <= i; j++) - files->items[j] = NULL; - batch->item_destroyer = NULL; - array_list_delete(batch); - batch = NULL; - if (!first) { - first = c; - } else { - if (!queue_enqueue(queue, c)) { - chunk_destroy(c); - *failed = true; - } - } - if (i < files->size - 1) { - batch = array_list_create(NULL); - if (!batch) { - *failed = true; - break; - } - batch_size = 0; - } - } - } - if (batch) { - batch->item_destroyer = NULL; - array_list_delete(batch); - } - return first; -} - -/* Scan one root-directory entry into either the subdirs or files list. */ -static void scan_root_entry(const ScannerOptions* options, const FilterNode* root_node, - const char* root_directory, const struct dirent* entry, - ArrayList* root_files, ArrayList* subdirs, dev_t root_dev, - ParallelScanner* ps) { - ScannerEntry inspected; - int inspection = - scanner_inspect_entry(options, root_directory, entry->d_name, entry->d_name, &inspected); - if (inspection < 0) { - ps->failed = true; - return; - } - if (inspection == 0) { - if (inspected.referent_error) - ps->io_error = true; - ArrayList* sink = NULL; - if (inspected.excluded) - sink = inspected.size_excluded ? options->size_skipped_paths : options->excluded_paths; - if (sink) { - /* A root-level prune protects the destination mirror of the entry's wire - path: under -R + --files-from that is the bare relative name, otherwise - it is the full source path with a leading '/' removed (matching the - send_path/file_wire_path the scanner hands the sender). */ - if (options->relative && options->file_list != NULL) { - if (!excluded_sink_append(sink, options->excluded_mutex, entry->d_name)) - ps->failed = true; - } else if (options->relative_prefix) { - char* wrel = scanner_prefix_send_path(options->relative_prefix, entry->d_name); - if (!wrel) { - ps->failed = true; - } else { - if (!excluded_sink_append(sink, options->excluded_mutex, wrel)) - ps->failed = true; - free(wrel); - } - } else { - char* abs_path = path_cat(root_directory, entry->d_name); - if (!abs_path) { - ps->failed = true; - } else { - const char* rel = *abs_path == '/' ? abs_path + 1 : abs_path; - if (!excluded_sink_append(sink, options->excluded_mutex, rel)) - ps->failed = true; - free(abs_path); - } - } - } - return; - } - char* cur_path = inspected.path; - struct stat st = inspected.stats; - bool is_dir = inspected.is_directory; - char* rel = str_dup(entry->d_name); - if (!rel) { - free(cur_path); - ps->failed = true; - return; - } - bool protect = false; - bool passes = entry_passes_selection(options->file_list, options->base_filters, root_node, rel, - entry->d_name, is_dir, options->per_dir_filters, - options->exclude_per_dir_filter_files, &protect); - /* -R + --files-from: root-level files keep their bare relative send path. */ - bool use_rel = options->relative && options->file_list != NULL; - if (!passes || protect) { - /* --files-from subset pruning is not a filter exclusion; -R bare-wire-path - exclusions are never recorded (see ScannerOptions.excluded_paths). */ - bool files_from_prune = options->file_list && !file_list_affects(options->file_list, rel); - if ((!files_from_prune && !use_rel) || protect) { - const char* rel_path; - char* prefixed = NULL; - if (use_rel) { - /* -R + --files-from: the destination/wire path is the bare relative - name, not the source path. */ - rel_path = rel; - } else if (options->relative_prefix) { - prefixed = scanner_prefix_send_path(options->relative_prefix, entry->d_name); - if (!prefixed) { - free(rel); - free(cur_path); - ps->failed = true; - return; - } - rel_path = prefixed; - } else { - rel_path = *cur_path == '/' ? cur_path + 1 : cur_path; - } - if (options->excluded_paths && - !excluded_sink_append(options->excluded_paths, options->excluded_mutex, rel_path)) - ps->failed = true; - free(prefixed); - } - if (!passes) { - scanner_note_filter(options, entry->d_name); - free(rel); - free(cur_path); - return; - } - } - if (is_dir) { - if (!scanner_same_filesystem(options->one_file_system, root_dev, st.st_dev)) { - if (options->one_file_system > 1) { - /* -xx: drop the mount-point directory entirely (rsync) and print the - --info=mount line when enabled. */ - scanner_note_mount(options, cur_path); - free(rel); - free(cur_path); - return; - } - /* -x/--one-file-system: emit the mount-point directory entry (empty) but - do not descend into it (see the sequential scanner for the same rule). */ - File* mount = file_create(cur_path); - free(cur_path); - if (mount == NULL) { - free(rel); - ps->failed = true; - return; - } - mount->is_dir = true; - if (options->use_metadata) { - mount->metadata = file_metadata_create(mount->path, &st, options->preserve_atimes, - options->preserve_crtimes); - if (!mount->metadata) { - free(rel); - file_destroy(mount); - ps->failed = true; - return; - } - } - if (options->relative_prefix) { - mount->send_path = scanner_prefix_send_path(options->relative_prefix, rel); - if (!mount->send_path) { - free(rel); - file_destroy(mount); - ps->failed = true; - return; - } - } - free(rel); - if (!array_list_add(root_files, mount)) { - file_destroy(mount); - ps->failed = true; - } - return; - } - free(rel); - if (!array_list_add(subdirs, cur_path)) { - free(cur_path); - ps->failed = true; - } - return; - } - File* file = file_create(cur_path); - free(cur_path); - if (!file) { - free(rel); - free(inspected.link_target); - inspected.link_target = NULL; - ps->failed = true; - return; - } - if (inspected.is_symlink) { - file->is_symlink = true; - file->symlink_target = inspected.link_target; - inspected.link_target = NULL; - } else { - file->data->size = st.st_size; - } - if (use_rel) { - file->send_path = rel; - rel = NULL; - } else if (options->relative_prefix) { - file->send_path = scanner_prefix_send_path(options->relative_prefix, rel); - free(rel); - rel = NULL; - if (!file->send_path) { - file_destroy(file); - ps->failed = true; - return; - } - } - ScannerSpecial special = scanner_prepare_special( - options->preserve_devices, options->preserve_specials, options->copy_devices, file, &st); - if (special == SCANNER_SPECIAL_SKIP) { - scanner_note_nonreg(ps->options, file->path); - free(rel); - file_destroy(file); - return; - } - if (options->hardlinks && S_ISREG(st.st_mode)) { - int gid; - bool is_first; - char* first_path = NULL; - if (!hardlink_table_assign((HardLinkTable*)options->hardlinks, file_wire_path(file), st.st_dev, - st.st_ino, &gid, &is_first, &first_path)) { - ps->failed = true; - } else { - file->link_group = gid; - file->link_first = is_first; - if (!is_first) { - file->hardlink_target = first_path; - file->data->size = 0; - } else { - free(first_path); - } - } - } - if (options->use_metadata) - file->metadata = - file_metadata_create(file->path, &st, options->preserve_atimes, options->preserve_crtimes); - if (options->use_metadata && !file->metadata) { - free(rel); - file_destroy(file); - ps->failed = true; - return; - } - if ((options->preserve_xattrs || options->preserve_acls) && - !(file->link_group != 0 && !file->link_first)) - file->xattrs = xattr_capture_path(file->path, options->preserve_acls); - if (!array_list_add(root_files, file)) { - free(rel); - file_destroy(file); - ps->failed = true; - return; - } - free(rel); -} - -/* Scan the root directory itself, collecting root files and subdirectories. - * Returns false if the root directory could not be opened. */ -static bool scan_root_directory(ParallelScanner* ps, const char* root_directory, - const ScannerOptions* options, const FilterNode* root_node, - dev_t root_dev, ArrayList* root_files, ArrayList* subdirs) { - DIR* dir = opendir(root_directory); - if (!dir) { - log_perror("Could not open root directory for parallel scan"); - return false; - } - /* The parallel scanner opens the transfer root directly (not through - open_next_directory), so record it as synchronized here. */ - if (!scanner_record_synced_dir(options, root_directory, "", - options->relative && options->file_list != NULL)) { - closedir(dir); - ps->failed = true; - return false; - } - log_debug_message(LOG_DEBUG_FLIST, "flist: scanning %s", root_directory); - const struct dirent* entry; - while ((entry = readdir(dir)) != NULL) { - if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) - continue; - scan_root_entry(options, root_node, root_directory, entry, root_files, subdirs, root_dev, ps); - } - closedir(dir); - return true; -} - -/* Spawn worker threads, one per group of subdirectories. */ -static void spawn_parallel_workers(ParallelScanner* ps, ArrayList* subdirs, - const ScannerOptions* options, const char* root_directory, - unsigned long long cs) { - if (subdirs->size <= 0) - return; - int n = options->num_threads > 0 ? options->num_threads : 4; - if (n > subdirs->size) - n = subdirs->size; - - ps->num_threads = n; - ps->expected_threads = n; - ps->threads = calloc(n, sizeof(thrd_t)); - if (!ps->threads) { - ps->num_threads = 0; - ps->expected_threads = 0; - ps->failed = true; - return; - } - int dirs_per_thread = subdirs->size / n; - int remainder = subdirs->size % n; - int start = 0; - ps->num_threads = 0; - for (int t = 0; t < n; t++) { - int count = dirs_per_thread + (t < remainder ? 1 : 0); - if (count == 0) - break; - ParallelWorkerArg* wa = calloc(1, sizeof(ParallelWorkerArg)); - if (!wa) { - parallel_scanner_creation_failed(ps); - break; - } - wa->ps = ps; - wa->dirs = calloc(count, sizeof(char*)); - wa->root_dir = str_dup(root_directory); - if (!wa->dirs || !wa->root_dir) { - free(wa->root_dir); - free(wa->dirs); - free(wa); - parallel_scanner_creation_failed(ps); - break; - } - bool dup_ok = true; - for (int j = 0; j < count; j++) { - wa->dirs[j] = str_dup((char*)subdirs->items[start + j]); - if (!wa->dirs[j]) - dup_ok = false; - } - if (!dup_ok) { - for (int j = 0; j < count; j++) - free(wa->dirs[j]); - free(wa->root_dir); - free(wa->dirs); - free(wa); - parallel_scanner_creation_failed(ps); - break; - } - wa->dir_count = count; - wa->options = *options; - wa->options.chunk_size = cs; - wa->allocation_session = ps->allocation_session; - start += count; - if (thrd_create(&ps->threads[t], parallel_worker_thread, wa) != thrd_success) { - for (int j = 0; j < count; j++) - free(wa->dirs[j]); - free(wa->root_dir); - free(wa->dirs); - free(wa); - parallel_scanner_creation_failed(ps); - break; - } - ps->num_threads++; - ps->created_threads++; - } -} - -ParallelScanner* parallel_scanner_create_with_options(const char* root_directory, - const ScannerOptions* options, - ProtocolSession* allocation_session) { - if (!root_directory || !options) - return NULL; - ParallelScanner* ps = calloc(1, sizeof(ParallelScanner)); - if (!ps) - return NULL; - if (!parallel_scanner_init(ps)) { - free(ps); - return NULL; - } - ps->allocation_session = allocation_session; - ps->options = options; - - ArrayList* root_files = array_list_create(file_destroy); - ArrayList* subdirs = array_list_create(free); - if (!root_files || !subdirs) { - array_list_delete(root_files); - array_list_delete(subdirs); - parallel_scanner_destroy(ps); - return NULL; - } - - dev_t root_dev = 0; - if (options->one_file_system) { - struct stat root_stats; - if (stat(root_directory, &root_stats) != 0) { - log_perror("Could not stat source directory"); - array_list_delete(root_files); - array_list_delete(subdirs); - parallel_scanner_destroy(ps); - return NULL; - } - root_dev = root_stats.st_dev; - } - - /* Build the root directory's per-directory filter context once; workers seed - * their scanners with it so per-dir rules behave identically to the sequential - * scanner. */ - FilterNode* root_node = NULL; - { - char err[256]; - bool any_exists = false; - FilterRuleList* own = - read_dir_filters(options, root_directory, "", &any_exists, err, sizeof(err)); - if (!own) { - /* A parse/allocation failure must fail the scan even when an earlier - merge file in the same directory existed (see the sequential scanner). */ - if (err[0] != '\0') { - log_message(LOG_LEVEL_ERROR, "invalid per-directory filter in %s: %s", root_directory, err); - array_list_delete(root_files); - array_list_delete(subdirs); - parallel_scanner_destroy(ps); - return NULL; - } - /* no files exist: leave root_node NULL */ - } else if (any_exists && (own->count > 0 || own->dir_merge_count > 0)) { - root_node = filter_node_alloc(NULL, own); - if (!root_node) { - filter_rule_list_free(own); - array_list_delete(root_files); - array_list_delete(subdirs); - parallel_scanner_destroy(ps); - return NULL; - } - } else { - filter_rule_list_free(own); - } - } - ps->root_filter_node = root_node; - - if (!scan_root_directory(ps, root_directory, options, root_node, root_dev, root_files, subdirs)) { - array_list_delete(root_files); - array_list_delete(subdirs); - parallel_scanner_destroy(ps); - return NULL; - } - /* The root itself is a traversed directory (rsync counts it in - `Number of files`); the worker DirectoryScanners account for every - subdirectory below it. */ - scanner_dir_count_count(options); - /* P7 Wave D: the parallel scanner never runs a DirectoryScanner over the - transfer root itself (it hands the root's immediate subdirectories to - workers), so capture the root's directory time here. */ - if (options->capture_dir_times && - !scanner_capture_dir_time( - options->dir_entries, options->dir_entries_mutex, root_directory, root_directory, - options->relative && options->file_list != NULL, options->relative_prefix, - options->preserve_atimes, options->preserve_crtimes, options->preserve_xattrs, - options->preserve_acls, options->no_implied_dirs, options->file_list)) { - array_list_delete(root_files); - array_list_delete(subdirs); - parallel_scanner_destroy(ps); - return NULL; - } - - unsigned long long cs = options->chunk_size > 0 ? options->chunk_size : DESIRED_CHUNK_SIZE; - ps->initial_chunk = batch_files(root_files, cs, ps->result_queue, &ps->failed); - array_list_delete(root_files); - - spawn_parallel_workers(ps, subdirs, options, root_directory, cs); - array_list_delete(subdirs); - return ps; -} - -Chunk* parallel_scanner_next(ParallelScanner* ps) { - if (ps->initial_chunk) { - Chunk* c = ps->initial_chunk; - ps->initial_chunk = NULL; - return c; - } - if (ps->num_threads == 0) { - mtx_lock(&ps->result_mutex); - if (!queue_is_empty(ps->result_queue)) { - Chunk* chunk = queue_dequeue(ps->result_queue); - mtx_unlock(&ps->result_mutex); - return chunk; - } - ps->done = true; - mtx_unlock(&ps->result_mutex); - return NULL; - } - Chunk* chunk = queue_dequeue_multithreaded( - ps->result_queue, &ps->result_mutex, &ps->result_not_empty, &ps->result_not_full, &ps->done); - return chunk; -} - -bool parallel_scanner_failed(const ParallelScanner* ps) { - return ps == NULL || ps->failed; -} - -bool parallel_scanner_had_io_error(const ParallelScanner* ps) { - return ps != NULL && ps->io_error; -} - -void parallel_scanner_destroy(ParallelScanner* ps) { - if (!ps) - return; - mtx_lock(&ps->result_mutex); - ps->done = true; - atomic_store(&ps->cancelled, true); - cnd_broadcast(&ps->result_not_empty); - cnd_broadcast(&ps->result_not_full); - mtx_unlock(&ps->result_mutex); - for (int i = 0; i < ps->num_threads; i++) - thrd_join(ps->threads[i], NULL); - free(ps->threads); - if (ps->root_filter_node) - filter_node_destroy(ps->root_filter_node); - if (ps->initial_chunk) - chunk_destroy(ps->initial_chunk); - queue_destroy(ps->result_queue); - mtx_destroy(&ps->result_mutex); - cnd_destroy(&ps->result_not_empty); - cnd_destroy(&ps->result_not_full); - free(ps); -} diff --git a/src/client/scanner_filter.c b/src/client/scanner_filter.c new file mode 100644 index 0000000..225d9f2 --- /dev/null +++ b/src/client/scanner_filter.c @@ -0,0 +1,671 @@ +#include "log.h" +#include "scanner.h" +#include "scanner_internal.h" +#include "array_list.h" +#include "chunk.h" +#include "file.h" +#include "queue.h" +#include "utils.h" +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "xattr.h" + +/* A chain node: `own` holds the .rsync-filter rules of one directory, `parent` + * the context that directory inherited (nearest ancestor with a filter file). + * The chain for a directory's contents runs from that directory's own node up + * to the root; the command-line base rules are evaluated after the whole + * chain. */ +struct FilterNode { + FilterNode* parent; + FilterRuleList* own; +}; + +void filter_node_destroy(void* item) { + if (item) { + FilterNode* node = (FilterNode*)item; + if (node->own) + filter_rule_list_free(node->own); + free(node); + } +} + +FilterNode* filter_node_alloc(FilterNode* parent, FilterRuleList* own) { + FilterNode* node = malloc(sizeof(FilterNode)); + if (!node) + return NULL; + node->parent = parent; + node->own = own; + return node; +} + +/* Evaluate a rule chain for one entry. rsync precedence, highest first: the + * innermost (current) directory's .rsync-filter rules, then each ancestor's, + * then the root's, and finally the command-line base rules (--filter/-C). The + * sender-side verdict decides whether the entry is hidden from the transfer; + * the receiver-side verdict decides whether its destination mirror is protected + * from --delete. Each side takes the FIRST matching rule independently. */ +typedef struct { + bool hide; /* sender-side exclude matched */ + bool protect; /* receiver-side exclude matched */ +} FilterOutcome; + +static void chain_rules_outcome(const FilterRuleList* base, const FilterNode* node, const char* rel, + const char* leaf, bool is_dir, FilterOutcome* out) { + memset(out, 0, sizeof(*out)); + bool sender_decided = false; + bool receiver_decided = false; + const FilterNode* n = node; + while (!sender_decided || !receiver_decided) { + const FilterRuleList* list = n ? n->own : base; + if (list) { + if (!sender_decided) { + FilterAction action = filter_rules_apply_side(list, rel, leaf, is_dir, FILTER_SIDE_SENDER); + if (action != FILTER_ACTION_NONE) { + out->hide = action == FILTER_ACTION_EXCLUDE; + sender_decided = true; + } + } + if (!receiver_decided) { + FilterAction action = + filter_rules_apply_side(list, rel, leaf, is_dir, FILTER_SIDE_RECEIVER); + if (action != FILTER_ACTION_NONE) { + out->protect = action == FILTER_ACTION_PROTECT; + receiver_decided = true; + } + } + } + if (!n) + break; + n = n->parent; + } +} + +static bool entry_allowed(const FilterRuleList* base, const FilterNode* node, const char* rel, + const char* leaf, bool is_dir, bool exclude_filter_files, + bool* protect_out) { + /* -FF: per-directory .rsync-filter files are never transferred (single -F + transfers them, matching rsync). */ + if (exclude_filter_files && !is_dir && strcmp(leaf, ".rsync-filter") == 0) { + if (protect_out) + *protect_out = false; + return false; + } + FilterOutcome outcome; + chain_rules_outcome(base, node, rel, leaf, is_dir, &outcome); + if (protect_out) + *protect_out = outcome.protect; + return !outcome.hide; +} + +void dir_entry_destroy(void* item) { + if (item) { + DirEntry* de = (DirEntry*)item; + free(de->path); + free(de); + } +} + +DirEntry* dir_entry_create(const char* path, int depth, FilterNode* context) { + DirEntry* de = malloc(sizeof(DirEntry)); + if (!de) + return NULL; + de->path = str_dup(path); + if (!de->path) { + free(de); + return NULL; + } + de->depth = depth; + de->context = context; + return de; +} + +/* Apply rsync's symlink-resolution precedence to one S_ISLNK entry: + * --copy-links dereferences every symlink; + * --copy-unsafe-links dereferences only targets unsafe_symlink() flags; + * -k/--copy-dirlinks dereferences only a symlink whose referent is a dir; + * --safe-links (receiver-side in rsync; modelled here) ignores an unsafe + * target that would otherwise be carried; with --munge-links + * every stored target becomes absolute, so --safe-links then + * ignores every symlink, exactly as rsync documents; + * -l/--links carries the link. + * `link_rel` is the symlink's transfer-relative path (incl. name) and is used + * only for the lexical unsafe test. `target` receives the raw link value. */ +LinkAction scanner_link_action(const ScannerOptions* options, const char* path, + const char* link_rel, char* target, size_t target_size) { + if (!options->follow_symlinks && !options->copy_links && !options->safe_links && + !options->copy_unsafe_links && !options->copy_dirlinks) + return LINK_ACTION_SKIP; + ssize_t length = readlink(path, target, target_size - 1); + if (length < 0) + return LINK_ACTION_SKIP; + target[length] = '\0'; + + bool unsafe = file_symlink_unsafe(target, link_rel); + if (options->copy_links || (options->copy_unsafe_links && unsafe)) + return LINK_ACTION_DEREF; + if (options->copy_dirlinks) { + struct stat ref; + if (stat(path, &ref) == 0 && S_ISDIR(ref.st_mode)) + return LINK_ACTION_DEREF; + } + if (options->safe_links && (unsafe || options->munge_links)) + return LINK_ACTION_SKIP_PROTECTED; + if (!options->follow_symlinks || target[0] == '\0') + return LINK_ACTION_SKIP; + return LINK_ACTION_CARRY; +} + +/* --one-file-system (-x) decision. Only directories can carry a different + * device than their parent (mount points), so this is checked when a child + * directory is about to be descended into. */ +bool scanner_same_filesystem(int one_file_system, dev_t root_device, dev_t entry_device) { + return one_file_system <= 0 || entry_device == root_device; +} + +/* Build a payload-less directory File carrying the captured metadata (when + * requested). Used by -x mount-point emission and --list-only directory + * entries. Returns NULL on allocation failure. */ +File* scanner_build_dir_file(const char* path, const struct stat* stats, + const ScannerOptions* options) { + File* dir = file_create(path); + if (dir == NULL) + return NULL; + dir->is_dir = true; + if (options->use_metadata) { + dir->metadata = + file_metadata_create(dir->path, stats, options->preserve_atimes, options->preserve_crtimes); + if (!dir->metadata) { + file_destroy(dir); + return NULL; + } + } + return dir; +} + +/* Relative path of an on-disk path below `root`. The transfer root may be + * given with a trailing slash; the returned rel path never has one and is "" + * for the root itself. A root of "/" is handled (its children start at "/"). + * Exposed so tests can exercise the mapping directly. */ +char* scanner_path_relative(const char* root, const char* fs_path) { + size_t root_len = strlen(root); + while (root_len > 1 && root[root_len - 1] == '/') + root_len--; + if (strncmp(root, fs_path, root_len) != 0) + return NULL; + if (root_len == 1 && root[0] == '/') { + if (fs_path[1] == '\0') + return str_dup(""); + return str_dup(fs_path + 1); + } + if (fs_path[root_len] == '\0') + return str_dup(""); + if (fs_path[root_len] != '/') + return NULL; + return str_dup(fs_path + root_len + 1); +} + +/* -R/--relative destination-relative prefix reconstructed from a source spec: + * everything after the first '.' path component (rsync's '/./' cut point), + * with leading/trailing slashes removed; or the whole spec (normalized) when + * there is no cut. Returns "" for the receive root. Exposed for tests. */ +char* scanner_relative_prefix(const char* spec) { + if (!spec || spec[0] == '\0') + return NULL; + const char* after = spec; + if (spec[0] == '.' && spec[1] == '/') { + after = spec + 2; + } else { + const char* cut = strstr(spec, "/./"); + if (cut) + after = cut + 3; + } + size_t cap = strlen(spec) + 1; + char* out = malloc(cap); + if (!out) + return NULL; + size_t len = 0; + for (const char* s = after; *s;) { + while (*s == '/') + s++; + const char* comp = s; + while (*s && *s != '/') + s++; + size_t clen = (size_t)(s - comp); + if (clen == 0 || (clen == 1 && comp[0] == '.')) + continue; + if (len) + out[len++] = '/'; + memcpy(out + len, comp, clen); + len += clen; + } + out[len] = '\0'; + return out; +} + +/* Relative path of a child entry below the current directory. */ +char* child_rel_path(const char* parent_rel, const char* name) { + if (!parent_rel || parent_rel[0] == '\0') + return str_dup(name); + return path_cat(parent_rel, name); +} + +/* Destination-relative wire path for an entry under an -R prefix. */ +char* scanner_prefix_send_path(const char* prefix, const char* rel) { + if (prefix[0] == '\0') + return str_dup(rel); + if (rel[0] == '\0') + return str_dup(prefix); + return path_cat(prefix, rel); +} + +/* Apply the --files-from allow-set and the filter layer to one entry. On + * return `*protect_out` is true when a receiver-side rule protects the entry's + * destination mirror from deletion. */ +bool entry_passes_selection(const FileListSet* file_list, const FilterRuleList* base, + const FilterNode* node, const char* rel, const char* leaf, bool is_dir, + bool per_dir_filters, bool exclude_filter_files, bool* protect_out) { + if (protect_out) + *protect_out = false; + if (file_list && !file_list_affects(file_list, rel)) + return false; + if (base || per_dir_filters) + return entry_allowed(base, node, rel, leaf, is_dir, exclude_filter_files, protect_out); + return true; +} + +/* Best-effort capture of the file's whitelisted xattrs (-X/-A). A failure to + * read xattrs is non-fatal: the file is transferred without them. */ +void scanner_capture_xattrs(const DirectoryScanner* scanner, File* file) { + if (!scanner || !file || !(scanner->options.preserve_xattrs || scanner->options.preserve_acls)) + return; + file->xattrs = xattr_capture_path(file->path, scanner->options.preserve_acls); +} + +/* Apply --hard-links (-H) detection to one regular File. On a sibling (a + * later member of an already-seen source inode) the File keeps the group id + * and the first member's wire path but carries NO data payload (size 0); the + * first member is left untouched (data present, link_first). Allocation + * failure is fatal: the scanner is marked failed. */ +void scanner_assign_hardlink(DirectoryScanner* scanner, HardLinkTable* table, File* file, + const struct stat* stats) { + if (!table || !file || !stats) + return; + int gid; + bool is_first; + char* first_path = NULL; + if (!hardlink_table_assign(table, file_wire_path(file), stats->st_dev, stats->st_ino, &gid, + &is_first, &first_path)) { + if (scanner) + scanner->failed = true; + return; + } + file->link_group = gid; + file->link_first = is_first; + if (!is_first) { + file->hardlink_target = first_path; + file->data->size = 0; + } else { + free(first_path); + } +} + +/* Phase 4 special/devices decision for one non-regular entry, matching rsync: + - a char/block device is RECREATED as a node under -D/--devices, unless + --copy-devices asks for its content to be copied into a regular file; + - a FIFO/socket is RECREATED under --specials; + - when the matching flag is absent the entry is SKIPPED ("skipping + non-regular file"), exactly like rsync's default, instead of being + silently copied as a zero-length regular file; + - anything else (regular/directory) is left to the normal data path. */ +ScannerSpecial scanner_prepare_special(bool preserve_devices, bool preserve_specials, + bool copy_devices, File* file, const struct stat* stats) { + if (!file || !stats) + return SCANNER_SPECIAL_REGULAR; + bool is_device = S_ISCHR(stats->st_mode) || S_ISBLK(stats->st_mode); + bool is_fifo = S_ISFIFO(stats->st_mode); + bool is_socket = S_ISSOCK(stats->st_mode); + if (!is_device && !is_fifo && !is_socket) + return SCANNER_SPECIAL_REGULAR; + if (is_device && copy_devices) + return SCANNER_SPECIAL_REGULAR; /* copy device content as a regular file */ + bool preserve = is_device ? preserve_devices : preserve_specials; + if (!preserve) + return SCANNER_SPECIAL_SKIP; + file->is_special = true; + file->data->size = 0; + file->data->data = NULL; + if (is_device) { + file->rdev_major = (int32_t)major(stats->st_rdev); + file->rdev_minor = (int32_t)minor(stats->st_rdev); + } + return SCANNER_SPECIAL_RECREATE; +} + +/* Append `rel` to the caller's exclusion sink, taking `mtx` when shared across + parallel worker threads. Returns false on allocation failure (list left + unchanged). */ +bool excluded_sink_append(ArrayList* list, mtx_t* mtx, const char* rel) { + if (!list) + return true; + char* dup = str_dup(rel); + if (!dup) + return false; + if (mtx) + mtx_lock(mtx); + bool ok = array_list_add(list, dup); + if (mtx) + mtx_unlock(mtx); + if (!ok) + free(dup); + return ok; +} + +/* Record one pruned filesystem path in a delete-protection sink. The stored + form is the entry's wire/destination-relative path (a single leading '/' + removed, exactly how manifest keep entries are stored), so the receiver's + walker prefixes match the destination layout. An allocation failure is a + fatal scan error. */ +static void scanner_record_protected(DirectoryScanner* scanner, const char* fs_path, + ArrayList* sink) { + if (!sink || !fs_path) + return; + const char* rel = *fs_path == '/' ? fs_path + 1 : fs_path; + if (!excluded_sink_append(sink, scanner->options.excluded_mutex, rel)) + scanner->failed = true; +} + +/* rsync's `--info=nonreg` line for a non-regular entry that is not being + * preserved: `skipping non-regular file "NAME"`. The name is the path relative + * to the transfer root, so it matches rsync's displayed name. */ +void scanner_note_nonreg(const ScannerOptions* options, const char* fs_path) { + if (!options || !options->note_nonreg || !fs_path) + return; + const char* rel = utils_strip_transfer_root(fs_path, options->send_directory); + char* escaped = output_escape(rel, options->eight_bit_output); + printf("skipping non-regular file \"%s\"\n", escaped ? escaped : rel); + free(escaped); + fflush(stdout); +} + +/* rsync 3.4.1's `--info=mount` line, emitted when `-xx` drops a mount-point + * directory: `[sender] skipping mount-point dir NAME` (the client is the + * sender). Plain `-x` keeps the empty directory and prints nothing, matching + * rsync. */ +void scanner_note_mount(const ScannerOptions* options, const char* fs_path) { + if (!options || !options->note_mount || !fs_path) + return; + const char* rel = utils_strip_transfer_root(fs_path, options->send_directory); + char* escaped = output_escape(rel, options->eight_bit_output); + printf("[sender] skipping mount-point dir %s\n", escaped ? escaped : rel); + free(escaped); + fflush(stdout); +} + +/* --debug=filter: a selection/filter decision dropped an entry. */ +void scanner_note_filter(const ScannerOptions* options, const char* name) { + if (!options || !log_debug_enabled(LOG_DEBUG_FILTER) || !name) + return; + log_debug_message(LOG_DEBUG_FILTER, "filter: excluded %s", name); +} + +/* Account for a directory that will not be represented by an inline directory + * entry. Paired with scanner_dir_count_uncount for empty directories that are + * emitted inline, so every traversed directory is counted exactly once. */ +void scanner_dir_count_count(const ScannerOptions* options) { + if (options && options->dir_count) + atomic_fetch_add(options->dir_count, 1); +} + +void scanner_dir_count_uncount(const ScannerOptions* options) { + if (options && options->dir_count) + atomic_fetch_sub(options->dir_count, 1); +} + +/* A user-selection exclusion (--filter/-C/per-dir or --exclude/--include). */ +void scanner_record_excluded(DirectoryScanner* scanner, const char* fs_path) { + scanner_record_protected(scanner, fs_path, scanner->options.excluded_paths); +} + +/* A --max-size/--min-size prune (always protected, even under --delete-excluded). */ +void scanner_record_size_skipped(DirectoryScanner* scanner, const char* fs_path) { + scanner_record_protected(scanner, fs_path, scanner->options.size_skipped_paths); +} + +/* Record a directory the scan synchronized. `fs_path` is its absolute path and + `rel` its path relative to the transfer root ("" for the root); the stored + form matches the wire layout (the bare relative path in -R+--files-from, else + the source path with a leading '/' removed, with "." for the receive root). + Returns false on allocation failure. */ +bool scanner_record_synced_dir(const ScannerOptions* options, const char* fs_path, const char* rel, + bool relative_mode) { + if (!options->synced_dirs && !options->plan_dirs) + return true; + if (!file_list_dir_in_scope(options->file_list, rel)) + return true; + char* prefixed = NULL; + const char* dest; + if (relative_mode) { + dest = rel; + } else if (options->relative_prefix) { + prefixed = scanner_prefix_send_path(options->relative_prefix, rel); + if (!prefixed) + return false; + dest = prefixed; + } else { + dest = fs_path; + } + if (dest[0] == '/') + dest++; + if (dest[0] == '\0') + dest = "."; + bool ok = true; + if (options->synced_dirs) + ok = excluded_sink_append(options->synced_dirs, options->excluded_mutex, dest); + /* The delete-plan keep set needs an entry for every traversed source + directory, including empty ones, so its destination mirror is kept rather + than deleted as an extra; the receive root (".") is implicit. */ + if (ok && options->plan_dirs && strcmp(dest, ".") != 0) + ok = excluded_sink_append(options->plan_dirs, options->excluded_mutex, dest); + free(prefixed); + return ok; +} + +/* Read every per-directory filter file that applies to `dir_path` (its + * .rsync-filter when -F is active, plus each registered "dir-merge NAME") into a + * fresh list. Returns NULL on allocation/parse failure (message in `err`); + * returns an empty list (and *any_exists=false) when no file exists. */ +FilterRuleList* read_dir_filters(const ScannerOptions* options, const char* dir_path, + const char* rel, bool* any_exists, char* err, size_t err_size) { + if (err && err_size > 0) + err[0] = '\0'; + const FilterRuleList* base = options->base_filters; + bool have_names = options->per_dir_filters || (base && base->dir_merge_count > 0); + if (any_exists) + *any_exists = false; + if (!have_names) + return NULL; + FilterRuleList* own = filter_rule_list_create(); + if (!own) { + snprintf(err, err_size, "memory allocation failed"); + return NULL; + } + FilterParseOptions opts = {.delete_excluded = options->delete_excluded, .cvs_exclude = false}; + bool exists = false; + if (options->per_dir_filters) { + if (!filter_file_append(own, dir_path, ".rsync-filter", rel, &opts, &exists, err, err_size)) + goto fail; + if (exists && any_exists) + *any_exists = true; + } + if (base) { + for (int i = 0; i < base->dir_merge_count; i++) { + if (!filter_file_append(own, dir_path, base->dir_merge_names[i], rel, &opts, &exists, err, + err_size)) + goto fail; + if (exists && any_exists) + *any_exists = true; + } + } + return own; +fail: + filter_rule_list_free(own); + return NULL; +} + +/* Merge the open directory's own per-directory filter files (the default + * .rsync-filter when -F is active, plus every "dir-merge NAME" registered on the + * base rule list) into the inherited context, returning the context used for + * this directory's entries. On a parse error the scanner is marked failed. + * Returns 0 on success, -1 on failure. */ +int open_directory_filter_context(DirectoryScanner* scanner, const FilterNode* inherited) { + char err[256]; + bool any_exists = false; + FilterRuleList* own = read_dir_filters(&scanner->options, scanner->current_path, + scanner->current_rel ? scanner->current_rel : "", + &any_exists, err, sizeof(err)); + if (!own) { + /* read_dir_filters() leaves `err` set on a parse/allocation failure even + when an earlier merge file in the same directory existed (any_exists true); + key off the error text rather than any_exists so an invalid per-directory + filter file can never be silently ignored. */ + if (err[0] == '\0') { + scanner->current_node = (FilterNode*)inherited; + return 0; + } + char* escaped_path = output_escape(scanner->current_path, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "invalid per-directory filter in %s: %s", + escaped_path ? escaped_path : "", err); + free(escaped_path); + scanner->failed = true; + return -1; + } + if (any_exists && (own->count > 0 || own->dir_merge_count > 0)) { + FilterNode* node = filter_node_alloc((FilterNode*)inherited, own); + if (!node || !array_list_add(scanner->filter_nodes, node)) { + filter_node_destroy(node); + scanner->failed = true; + return -1; + } + scanner->current_node = node; + } else { + filter_rule_list_free(own); + scanner->current_node = (FilterNode*)inherited; + } + return 0; +} + +/* Inspect symlinks, resolve the entry type, and apply file filters once for both scanners. + * `link_rel` is the entry's path relative to the transfer root (including its + * name), used for the lexical rsync unsafe-symlink test. */ +int scanner_inspect_entry(const ScannerOptions* options, const char* containing_dir, + const char* link_rel, const char* name, ScannerEntry* entry) { + entry->excluded = false; + entry->size_excluded = false; + entry->referent_error = false; + entry->is_symlink = false; + entry->link_target = NULL; + entry->path = path_cat(containing_dir, name); + if (!entry->path) + return -1; + + struct stat link_stats; + if (lstat(entry->path, &link_stats) != 0) { + free(entry->path); + return 0; + } + if (!S_ISLNK(link_stats.st_mode)) + goto regular; + + char link_target[4096]; + switch (scanner_link_action(options, entry->path, link_rel, link_target, sizeof(link_target))) { + case LINK_ACTION_SKIP: + goto skip; + case LINK_ACTION_SKIP_PROTECTED: + /* --safe-links ignored the link, but rsync still counts it as present in + the transfer, so its destination mirror survives --delete. Record it as + an excluded path (the same delete-protection channel as a filter prune). */ + entry->excluded = true; + goto skip; + case LINK_ACTION_DEREF: + if (stat(entry->path, &entry->stats) != 0) { + /* rsync reports "symlink has no referent" and continues with a partial + transfer (exit 23); record the error so the run exits 23 too. */ + char* escaped = output_escape(entry->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "symlink has no referent: %s", + escaped ? escaped : ""); + free(escaped); + entry->referent_error = true; + goto skip; + } + entry->is_directory = S_ISDIR(entry->stats.st_mode); + if (entry->is_directory) + return 1; + goto apply_filters; + case LINK_ACTION_CARRY: + break; + } + + /* Carry the link as a symlink. --munge-links is applied by the RECEIVER (it + prefixes every stored target with /rsyncd-munged/); when the SOURCE already + holds a munged value the sender strips it so the receiver re-munges a clean + target, round-tripping a munged tree exactly like rsync. */ + entry->is_symlink = true; + entry->stats = link_stats; + entry->is_directory = false; + entry->link_target = str_dup(link_target); + if (!entry->link_target) + goto skip; + if (options->munge_links) + file_symlink_unmunge(entry->link_target); + goto apply_filters; + +regular: + /* Not a symlink: the lstat() above already described this entry, and lstat + and stat are identical for every non-symlink, so reuse that result instead + of issuing a redundant stat() on the scanner hot path. stat() is still + used on the dereference paths above/below for actual symlinks (copy-links, + safe/copy-unsafe links, and -k symlinks-to-directories). */ + entry->stats = link_stats; + entry->is_directory = S_ISDIR(link_stats.st_mode); + if (entry->is_directory) + return 1; + +apply_filters: + for (int i = 0; i < options->exclude_count; i++) + if (glob_match(options->exclude_patterns[i], name)) { + entry->excluded = true; + goto skip; + } + if (options->include_count > 0) { + bool included = false; + for (int i = 0; i < options->include_count; i++) + if (glob_match(options->include_patterns[i], name)) + included = true; + if (!included) { + entry->excluded = true; + goto skip; + } + } + if ((options->max_size > 0 && (unsigned long long)entry->stats.st_size > options->max_size) || + (options->min_size > 0 && (unsigned long long)entry->stats.st_size < options->min_size)) { + entry->excluded = true; + entry->size_excluded = true; + goto skip; + } + return 1; + +skip: + free(entry->path); + entry->path = NULL; + free(entry->link_target); + entry->link_target = NULL; + return 0; +} diff --git a/src/client/scanner_internal.h b/src/client/scanner_internal.h new file mode 100644 index 0000000..c772d9d --- /dev/null +++ b/src/client/scanner_internal.h @@ -0,0 +1,107 @@ +#ifndef SCANNER_INTERNAL_H +#define SCANNER_INTERNAL_H + +/* Internal declarations shared between the scanner translation units + * (scanner_filter.c, scanner.c, scanner_parallel.c). Nothing here is part of + * the public scanner façade (scanner.h); every symbol stays internal to the + * client module. */ + +#include "array_list.h" +#include "file.h" +#include "scanner.h" +#include +#include +#include + +typedef struct { + char* path; + int depth; + FilterNode* context; /* inherited per-directory filter context */ +} DirEntry; + +/* How rsync's readlink_stat()/generator resolves one source symlink. */ +typedef enum { + LINK_ACTION_SKIP, /* not transferred (no link option) */ + LINK_ACTION_SKIP_PROTECTED, /* ignored as unsafe by --safe-links; rsync keeps + it in the transfer, so its destination mirror + must be protected from --delete */ + LINK_ACTION_DEREF, /* follow the referent (--copy-links, an unsafe + target under --copy-unsafe-links, or -k dir) */ + LINK_ACTION_CARRY, /* transmit the link itself (-l) */ +} LinkAction; + +typedef struct { + char* path; + struct stat stats; + bool is_directory; + /* True when the entry should be carried through as a SYMLINK (is_symlink) + rather than a dereferenced file/directory. When true, `link_target` holds + the owned target string to transmit (sender-munged under --munge-links); + ownership transfers to the File built from this entry. */ + bool is_symlink; + char* link_target; + /* True when the entry was pruned by a user selection rule (--filter/-C/per-dir + rules or the --exclude/--include layer) rather than skipped for another + reason (unreadable, symlink policy, not applicable). */ + bool excluded; + /* True when the entry was skipped specifically by --max-size/--min-size. + Size pruning protects the destination mirror even under --delete-excluded, + so it is recorded into a separate sink from `excluded`. */ + bool size_excluded; + /* True when a symlink selected for dereferencing (-L/--copy-links or an + unsafe target under --copy-unsafe-links) had no usable referent (a broken + link or a stat() failure). rsync still reports this as a partial transfer + (exit 23) even though the entry is skipped, so the scanner records it as a + non-fatal I/O error. */ + bool referent_error; +} ScannerEntry; + +typedef enum { + SCANNER_SPECIAL_REGULAR, /* ordinary file: transfer content */ + SCANNER_SPECIAL_RECREATE, /* is_special node to recreate on the receiver */ + SCANNER_SPECIAL_SKIP, /* non-regular entry not requested: skip */ +} ScannerSpecial; + +/* scanner_filter.c */ +void filter_node_destroy(void* item); +FilterNode* filter_node_alloc(FilterNode* parent, FilterRuleList* own); +void dir_entry_destroy(void* item); +DirEntry* dir_entry_create(const char* path, int depth, FilterNode* context); +LinkAction scanner_link_action(const ScannerOptions* options, const char* path, + const char* link_rel, char* target, size_t target_size); +File* scanner_build_dir_file(const char* path, const struct stat* stats, + const ScannerOptions* options); +char* child_rel_path(const char* parent_rel, const char* name); +char* scanner_prefix_send_path(const char* prefix, const char* rel); +bool entry_passes_selection(const FileListSet* file_list, const FilterRuleList* base, + const FilterNode* node, const char* rel, const char* leaf, bool is_dir, + bool per_dir_filters, bool exclude_filter_files, bool* protect_out); +void scanner_capture_xattrs(const DirectoryScanner* scanner, File* file); +void scanner_assign_hardlink(DirectoryScanner* scanner, HardLinkTable* table, File* file, + const struct stat* stats); +ScannerSpecial scanner_prepare_special(bool preserve_devices, bool preserve_specials, + bool copy_devices, File* file, const struct stat* stats); +bool excluded_sink_append(ArrayList* list, mtx_t* mtx, const char* rel); +void scanner_note_nonreg(const ScannerOptions* options, const char* fs_path); +void scanner_note_mount(const ScannerOptions* options, const char* fs_path); +void scanner_note_filter(const ScannerOptions* options, const char* name); +void scanner_dir_count_count(const ScannerOptions* options); +void scanner_dir_count_uncount(const ScannerOptions* options); +void scanner_record_excluded(DirectoryScanner* scanner, const char* fs_path); +void scanner_record_size_skipped(DirectoryScanner* scanner, const char* fs_path); +bool scanner_record_synced_dir(const ScannerOptions* options, const char* fs_path, const char* rel, + bool relative_mode); +FilterRuleList* read_dir_filters(const ScannerOptions* options, const char* dir_path, + const char* rel, bool* any_exists, char* err, size_t err_size); +int open_directory_filter_context(DirectoryScanner* scanner, const FilterNode* inherited); +int scanner_inspect_entry(const ScannerOptions* options, const char* containing_dir, + const char* link_rel, const char* name, ScannerEntry* entry); + +/* scanner.c */ +bool scanner_capture_dir_time(ArrayList* dir_entries, mtx_t* mutex, const char* root_path, + const char* fs_path, bool relative_mode, const char* relative_prefix, + bool preserve_atimes, bool preserve_crtimes, bool preserve_xattrs, + bool preserve_acls, bool no_implied_dirs, + const FileListSet* file_list); + +#endif diff --git a/src/client/scanner_parallel.c b/src/client/scanner_parallel.c new file mode 100644 index 0000000..9fa5151 --- /dev/null +++ b/src/client/scanner_parallel.c @@ -0,0 +1,703 @@ +#include "log.h" +#include "scanner.h" +#include "scanner_internal.h" +#include "array_list.h" +#include "chunk.h" +#include "file.h" +#include "queue.h" +#include "utils.h" +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "xattr.h" + +typedef struct { + ParallelScanner* ps; + char** dirs; + int dir_count; + char* root_dir; /* the transfer root, for relative-path computation */ + ScannerOptions options; + ProtocolSession* allocation_session; +} ParallelWorkerArg; + +static int parallel_worker_thread(void* arg) { + ParallelWorkerArg* wa = (ParallelWorkerArg*)arg; + ProtocolSession* allocation_session = wa->allocation_session; + if (allocation_session) + protocol_session_bind(allocation_session); + for (int i = 0; i < wa->dir_count; i++) { + DirectoryScanner* ds = directory_scanner_create_with_options(wa->dirs[i], &wa->options); + if (!ds) { + mtx_lock(&wa->ps->result_mutex); + wa->ps->failed = true; + atomic_store(&wa->ps->cancelled, true); + cnd_broadcast(&wa->ps->result_not_empty); + cnd_broadcast(&wa->ps->result_not_full); + mtx_unlock(&wa->ps->result_mutex); + for (int j = i; j < wa->dir_count; j++) + free(wa->dirs[j]); + break; + } + /* Root .rsync-filter rules (parsed by the parallel scanner) apply to the + * contents of every assigned subdirectory. Relative paths (used by the + * allow-set and per-directory rules) are computed against the transfer + * root, not the subdirectory the worker is seeded with. Exclusion + * recording shares one caller-owned list across the workers. */ + free(ds->root_path); + ds->root_path = str_dup(wa->root_dir); + ds->seed_node = wa->ps->root_filter_node; + ds->options.excluded_mutex = &wa->ps->result_mutex; + Chunk* chunk; + while ((chunk = directory_scanner_next(ds)) != NULL) { + if (!queue_enqueue_multithreaded_cancel(wa->ps->result_queue, chunk, &wa->ps->result_mutex, + &wa->ps->result_not_empty, &wa->ps->result_not_full, + &wa->ps->cancelled)) { + chunk_destroy(chunk); + break; + } + } + if (directory_scanner_failed(ds)) { + mtx_lock(&wa->ps->result_mutex); + wa->ps->failed = true; + atomic_store(&wa->ps->cancelled, true); + cnd_broadcast(&wa->ps->result_not_empty); + cnd_broadcast(&wa->ps->result_not_full); + mtx_unlock(&wa->ps->result_mutex); + } else if (directory_scanner_had_io_error(ds)) { + /* --ignore-errors path: an unreadable directory was skipped, not fatal. */ + mtx_lock(&wa->ps->result_mutex); + wa->ps->io_error = true; + mtx_unlock(&wa->ps->result_mutex); + } + directory_scanner_destroy(ds); + free(wa->dirs[i]); + } + ParallelScanner* ps = wa->ps; + free(wa->root_dir); + free(wa->dirs); + free(wa); + mtx_lock(&ps->result_mutex); + ps->completed++; + if (ps->completed >= ps->expected_threads) { + ps->done = true; + cnd_signal(&ps->result_not_empty); + } + mtx_unlock(&ps->result_mutex); + if (allocation_session) + protocol_session_unbind(); + return thrd_success; +} + +static void parallel_scanner_creation_failed(ParallelScanner* ps) { + mtx_lock(&ps->result_mutex); + ps->failed = true; + atomic_store(&ps->cancelled, true); + ps->expected_threads = ps->created_threads; + if (ps->completed >= ps->expected_threads) + ps->done = true; + cnd_broadcast(&ps->result_not_empty); + cnd_broadcast(&ps->result_not_full); + mtx_unlock(&ps->result_mutex); +} + +/* Initialize result queue and synchronization primitives. Returns true on success. */ +static bool parallel_scanner_init(ParallelScanner* ps) { + ps->result_queue = queue_create(100, chunk_destroy); + if (!ps->result_queue) + return false; + atomic_init(&ps->cancelled, false); + int init = 0; + bool ok = true; + if (mtx_init(&ps->result_mutex, mtx_plain) != thrd_success) + ok = false; + if (ok) { + init++; + if (cnd_init(&ps->result_not_empty) != thrd_success) + ok = false; + } + if (ok) { + // cppcheck-suppress unreadVariable + init++; + if (cnd_init(&ps->result_not_full) != thrd_success) + ok = false; + } + if (!ok) { + if (init >= 3) + cnd_destroy(&ps->result_not_full); + if (init >= 2) + cnd_destroy(&ps->result_not_empty); + if (init >= 1) + mtx_destroy(&ps->result_mutex); + queue_destroy(ps->result_queue); + ps->result_queue = NULL; + return false; + } + return true; +} + +/* Split files into chunks of roughly chunk_size bytes. Returns the first chunk (also stored + * chunks beyond the first are enqueued on `queue`). Nulls out consumed entries in `files`. + * Sets *failed on allocation/enqueue errors. */ +static Chunk* batch_files(ArrayList* files, unsigned long long chunk_size, Queue* queue, + bool* failed) { + Chunk* first = NULL; + if (files->size <= 0) + return NULL; + ArrayList* batch = array_list_create(NULL); + if (!batch) { + *failed = true; + return NULL; + } + unsigned long long batch_size = 0; + for (int i = 0; i < files->size; i++) { + File* f = (File*)files->items[i]; + if (!array_list_add(batch, f)) { + *failed = true; + break; + } + batch_size += f->data->size; + if (batch_size >= chunk_size || i == files->size - 1) { + void** items = array_list_to_array(batch); + if (!items) { + *failed = true; + array_list_delete(batch); + batch = NULL; + break; + } + Chunk* c = chunk_create((File**)items, batch->size); + free(items); + if (!c) { + *failed = true; + array_list_delete(batch); + batch = NULL; + break; + } + int batch_start = i - batch->size + 1; + for (int j = batch_start; j <= i; j++) + files->items[j] = NULL; + batch->item_destroyer = NULL; + array_list_delete(batch); + batch = NULL; + if (!first) { + first = c; + } else { + if (!queue_enqueue(queue, c)) { + chunk_destroy(c); + *failed = true; + } + } + if (i < files->size - 1) { + batch = array_list_create(NULL); + if (!batch) { + *failed = true; + break; + } + batch_size = 0; + } + } + } + if (batch) { + batch->item_destroyer = NULL; + array_list_delete(batch); + } + return first; +} + +/* Scan one root-directory entry into either the subdirs or files list. */ +static void scan_root_entry(const ScannerOptions* options, const FilterNode* root_node, + const char* root_directory, const struct dirent* entry, + ArrayList* root_files, ArrayList* subdirs, dev_t root_dev, + ParallelScanner* ps) { + ScannerEntry inspected; + int inspection = + scanner_inspect_entry(options, root_directory, entry->d_name, entry->d_name, &inspected); + if (inspection < 0) { + ps->failed = true; + return; + } + if (inspection == 0) { + if (inspected.referent_error) + ps->io_error = true; + ArrayList* sink = NULL; + if (inspected.excluded) + sink = inspected.size_excluded ? options->size_skipped_paths : options->excluded_paths; + if (sink) { + /* A root-level prune protects the destination mirror of the entry's wire + path: under -R + --files-from that is the bare relative name, otherwise + it is the full source path with a leading '/' removed (matching the + send_path/file_wire_path the scanner hands the sender). */ + if (options->relative && options->file_list != NULL) { + if (!excluded_sink_append(sink, options->excluded_mutex, entry->d_name)) + ps->failed = true; + } else if (options->relative_prefix) { + char* wrel = scanner_prefix_send_path(options->relative_prefix, entry->d_name); + if (!wrel) { + ps->failed = true; + } else { + if (!excluded_sink_append(sink, options->excluded_mutex, wrel)) + ps->failed = true; + free(wrel); + } + } else { + char* abs_path = path_cat(root_directory, entry->d_name); + if (!abs_path) { + ps->failed = true; + } else { + const char* rel = *abs_path == '/' ? abs_path + 1 : abs_path; + if (!excluded_sink_append(sink, options->excluded_mutex, rel)) + ps->failed = true; + free(abs_path); + } + } + } + return; + } + char* cur_path = inspected.path; + struct stat st = inspected.stats; + bool is_dir = inspected.is_directory; + char* rel = str_dup(entry->d_name); + if (!rel) { + free(cur_path); + ps->failed = true; + return; + } + bool protect = false; + bool passes = entry_passes_selection(options->file_list, options->base_filters, root_node, rel, + entry->d_name, is_dir, options->per_dir_filters, + options->exclude_per_dir_filter_files, &protect); + /* -R + --files-from: root-level files keep their bare relative send path. */ + bool use_rel = options->relative && options->file_list != NULL; + if (!passes || protect) { + /* --files-from subset pruning is not a filter exclusion; -R bare-wire-path + exclusions are never recorded (see ScannerOptions.excluded_paths). */ + bool files_from_prune = options->file_list && !file_list_affects(options->file_list, rel); + if ((!files_from_prune && !use_rel) || protect) { + const char* rel_path; + char* prefixed = NULL; + if (use_rel) { + /* -R + --files-from: the destination/wire path is the bare relative + name, not the source path. */ + rel_path = rel; + } else if (options->relative_prefix) { + prefixed = scanner_prefix_send_path(options->relative_prefix, entry->d_name); + if (!prefixed) { + free(rel); + free(cur_path); + ps->failed = true; + return; + } + rel_path = prefixed; + } else { + rel_path = *cur_path == '/' ? cur_path + 1 : cur_path; + } + if (options->excluded_paths && + !excluded_sink_append(options->excluded_paths, options->excluded_mutex, rel_path)) + ps->failed = true; + free(prefixed); + } + if (!passes) { + scanner_note_filter(options, entry->d_name); + free(rel); + free(cur_path); + return; + } + } + if (is_dir) { + if (!scanner_same_filesystem(options->one_file_system, root_dev, st.st_dev)) { + if (options->one_file_system > 1) { + /* -xx: drop the mount-point directory entirely (rsync) and print the + --info=mount line when enabled. */ + scanner_note_mount(options, cur_path); + free(rel); + free(cur_path); + return; + } + /* -x/--one-file-system: emit the mount-point directory entry (empty) but + do not descend into it (see the sequential scanner for the same rule). */ + File* mount = file_create(cur_path); + free(cur_path); + if (mount == NULL) { + free(rel); + ps->failed = true; + return; + } + mount->is_dir = true; + if (options->use_metadata) { + mount->metadata = file_metadata_create(mount->path, &st, options->preserve_atimes, + options->preserve_crtimes); + if (!mount->metadata) { + free(rel); + file_destroy(mount); + ps->failed = true; + return; + } + } + if (options->relative_prefix) { + mount->send_path = scanner_prefix_send_path(options->relative_prefix, rel); + if (!mount->send_path) { + free(rel); + file_destroy(mount); + ps->failed = true; + return; + } + } + free(rel); + if (!array_list_add(root_files, mount)) { + file_destroy(mount); + ps->failed = true; + } + return; + } + free(rel); + if (!array_list_add(subdirs, cur_path)) { + free(cur_path); + ps->failed = true; + } + return; + } + File* file = file_create(cur_path); + free(cur_path); + if (!file) { + free(rel); + free(inspected.link_target); + inspected.link_target = NULL; + ps->failed = true; + return; + } + if (inspected.is_symlink) { + file->is_symlink = true; + file->symlink_target = inspected.link_target; + inspected.link_target = NULL; + } else { + file->data->size = st.st_size; + } + if (use_rel) { + file->send_path = rel; + rel = NULL; + } else if (options->relative_prefix) { + file->send_path = scanner_prefix_send_path(options->relative_prefix, rel); + free(rel); + rel = NULL; + if (!file->send_path) { + file_destroy(file); + ps->failed = true; + return; + } + } + ScannerSpecial special = scanner_prepare_special( + options->preserve_devices, options->preserve_specials, options->copy_devices, file, &st); + if (special == SCANNER_SPECIAL_SKIP) { + scanner_note_nonreg(ps->options, file->path); + free(rel); + file_destroy(file); + return; + } + if (options->hardlinks && S_ISREG(st.st_mode)) { + int gid; + bool is_first; + char* first_path = NULL; + if (!hardlink_table_assign((HardLinkTable*)options->hardlinks, file_wire_path(file), st.st_dev, + st.st_ino, &gid, &is_first, &first_path)) { + ps->failed = true; + } else { + file->link_group = gid; + file->link_first = is_first; + if (!is_first) { + file->hardlink_target = first_path; + file->data->size = 0; + } else { + free(first_path); + } + } + } + if (options->use_metadata) + file->metadata = + file_metadata_create(file->path, &st, options->preserve_atimes, options->preserve_crtimes); + if (options->use_metadata && !file->metadata) { + free(rel); + file_destroy(file); + ps->failed = true; + return; + } + if ((options->preserve_xattrs || options->preserve_acls) && + !(file->link_group != 0 && !file->link_first)) + file->xattrs = xattr_capture_path(file->path, options->preserve_acls); + if (!array_list_add(root_files, file)) { + free(rel); + file_destroy(file); + ps->failed = true; + return; + } + free(rel); +} + +/* Scan the root directory itself, collecting root files and subdirectories. + * Returns false if the root directory could not be opened. */ +static bool scan_root_directory(ParallelScanner* ps, const char* root_directory, + const ScannerOptions* options, const FilterNode* root_node, + dev_t root_dev, ArrayList* root_files, ArrayList* subdirs) { + DIR* dir = opendir(root_directory); + if (!dir) { + log_perror("Could not open root directory for parallel scan"); + return false; + } + /* The parallel scanner opens the transfer root directly (not through + open_next_directory), so record it as synchronized here. */ + if (!scanner_record_synced_dir(options, root_directory, "", + options->relative && options->file_list != NULL)) { + closedir(dir); + ps->failed = true; + return false; + } + log_debug_message(LOG_DEBUG_FLIST, "flist: scanning %s", root_directory); + const struct dirent* entry; + while ((entry = readdir(dir)) != NULL) { + if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) + continue; + scan_root_entry(options, root_node, root_directory, entry, root_files, subdirs, root_dev, ps); + } + closedir(dir); + return true; +} + +/* Spawn worker threads, one per group of subdirectories. */ +static void spawn_parallel_workers(ParallelScanner* ps, ArrayList* subdirs, + const ScannerOptions* options, const char* root_directory, + unsigned long long cs) { + if (subdirs->size <= 0) + return; + int n = options->num_threads > 0 ? options->num_threads : 4; + if (n > subdirs->size) + n = subdirs->size; + + ps->num_threads = n; + ps->expected_threads = n; + ps->threads = calloc(n, sizeof(thrd_t)); + if (!ps->threads) { + ps->num_threads = 0; + ps->expected_threads = 0; + ps->failed = true; + return; + } + int dirs_per_thread = subdirs->size / n; + int remainder = subdirs->size % n; + int start = 0; + ps->num_threads = 0; + for (int t = 0; t < n; t++) { + int count = dirs_per_thread + (t < remainder ? 1 : 0); + if (count == 0) + break; + ParallelWorkerArg* wa = calloc(1, sizeof(ParallelWorkerArg)); + if (!wa) { + parallel_scanner_creation_failed(ps); + break; + } + wa->ps = ps; + wa->dirs = calloc(count, sizeof(char*)); + wa->root_dir = str_dup(root_directory); + if (!wa->dirs || !wa->root_dir) { + free(wa->root_dir); + free(wa->dirs); + free(wa); + parallel_scanner_creation_failed(ps); + break; + } + bool dup_ok = true; + for (int j = 0; j < count; j++) { + wa->dirs[j] = str_dup((char*)subdirs->items[start + j]); + if (!wa->dirs[j]) + dup_ok = false; + } + if (!dup_ok) { + for (int j = 0; j < count; j++) + free(wa->dirs[j]); + free(wa->root_dir); + free(wa->dirs); + free(wa); + parallel_scanner_creation_failed(ps); + break; + } + wa->dir_count = count; + wa->options = *options; + wa->options.chunk_size = cs; + wa->allocation_session = ps->allocation_session; + start += count; + if (thrd_create(&ps->threads[t], parallel_worker_thread, wa) != thrd_success) { + for (int j = 0; j < count; j++) + free(wa->dirs[j]); + free(wa->root_dir); + free(wa->dirs); + free(wa); + parallel_scanner_creation_failed(ps); + break; + } + ps->num_threads++; + ps->created_threads++; + } +} + +ParallelScanner* parallel_scanner_create_with_options(const char* root_directory, + const ScannerOptions* options, + ProtocolSession* allocation_session) { + if (!root_directory || !options) + return NULL; + ParallelScanner* ps = calloc(1, sizeof(ParallelScanner)); + if (!ps) + return NULL; + if (!parallel_scanner_init(ps)) { + free(ps); + return NULL; + } + ps->allocation_session = allocation_session; + ps->options = options; + + ArrayList* root_files = array_list_create(file_destroy); + ArrayList* subdirs = array_list_create(free); + if (!root_files || !subdirs) { + array_list_delete(root_files); + array_list_delete(subdirs); + parallel_scanner_destroy(ps); + return NULL; + } + + dev_t root_dev = 0; + if (options->one_file_system) { + struct stat root_stats; + if (stat(root_directory, &root_stats) != 0) { + log_perror("Could not stat source directory"); + array_list_delete(root_files); + array_list_delete(subdirs); + parallel_scanner_destroy(ps); + return NULL; + } + root_dev = root_stats.st_dev; + } + + /* Build the root directory's per-directory filter context once; workers seed + * their scanners with it so per-dir rules behave identically to the sequential + * scanner. */ + FilterNode* root_node = NULL; + { + char err[256]; + bool any_exists = false; + FilterRuleList* own = + read_dir_filters(options, root_directory, "", &any_exists, err, sizeof(err)); + if (!own) { + /* A parse/allocation failure must fail the scan even when an earlier + merge file in the same directory existed (see the sequential scanner). */ + if (err[0] != '\0') { + log_message(LOG_LEVEL_ERROR, "invalid per-directory filter in %s: %s", root_directory, err); + array_list_delete(root_files); + array_list_delete(subdirs); + parallel_scanner_destroy(ps); + return NULL; + } + /* no files exist: leave root_node NULL */ + } else if (any_exists && (own->count > 0 || own->dir_merge_count > 0)) { + root_node = filter_node_alloc(NULL, own); + if (!root_node) { + filter_rule_list_free(own); + array_list_delete(root_files); + array_list_delete(subdirs); + parallel_scanner_destroy(ps); + return NULL; + } + } else { + filter_rule_list_free(own); + } + } + ps->root_filter_node = root_node; + + if (!scan_root_directory(ps, root_directory, options, root_node, root_dev, root_files, subdirs)) { + array_list_delete(root_files); + array_list_delete(subdirs); + parallel_scanner_destroy(ps); + return NULL; + } + /* The root itself is a traversed directory (rsync counts it in + `Number of files`); the worker DirectoryScanners account for every + subdirectory below it. */ + scanner_dir_count_count(options); + /* P7 Wave D: the parallel scanner never runs a DirectoryScanner over the + transfer root itself (it hands the root's immediate subdirectories to + workers), so capture the root's directory time here. */ + if (options->capture_dir_times && + !scanner_capture_dir_time( + options->dir_entries, options->dir_entries_mutex, root_directory, root_directory, + options->relative && options->file_list != NULL, options->relative_prefix, + options->preserve_atimes, options->preserve_crtimes, options->preserve_xattrs, + options->preserve_acls, options->no_implied_dirs, options->file_list)) { + array_list_delete(root_files); + array_list_delete(subdirs); + parallel_scanner_destroy(ps); + return NULL; + } + + unsigned long long cs = options->chunk_size > 0 ? options->chunk_size : DESIRED_CHUNK_SIZE; + ps->initial_chunk = batch_files(root_files, cs, ps->result_queue, &ps->failed); + array_list_delete(root_files); + + spawn_parallel_workers(ps, subdirs, options, root_directory, cs); + array_list_delete(subdirs); + return ps; +} + +Chunk* parallel_scanner_next(ParallelScanner* ps) { + if (ps->initial_chunk) { + Chunk* c = ps->initial_chunk; + ps->initial_chunk = NULL; + return c; + } + if (ps->num_threads == 0) { + mtx_lock(&ps->result_mutex); + if (!queue_is_empty(ps->result_queue)) { + Chunk* chunk = queue_dequeue(ps->result_queue); + mtx_unlock(&ps->result_mutex); + return chunk; + } + ps->done = true; + mtx_unlock(&ps->result_mutex); + return NULL; + } + Chunk* chunk = queue_dequeue_multithreaded( + ps->result_queue, &ps->result_mutex, &ps->result_not_empty, &ps->result_not_full, &ps->done); + return chunk; +} + +bool parallel_scanner_failed(const ParallelScanner* ps) { + return ps == NULL || ps->failed; +} + +bool parallel_scanner_had_io_error(const ParallelScanner* ps) { + return ps != NULL && ps->io_error; +} + +void parallel_scanner_destroy(ParallelScanner* ps) { + if (!ps) + return; + mtx_lock(&ps->result_mutex); + ps->done = true; + atomic_store(&ps->cancelled, true); + cnd_broadcast(&ps->result_not_empty); + cnd_broadcast(&ps->result_not_full); + mtx_unlock(&ps->result_mutex); + for (int i = 0; i < ps->num_threads; i++) + thrd_join(ps->threads[i], NULL); + free(ps->threads); + if (ps->root_filter_node) + filter_node_destroy(ps->root_filter_node); + if (ps->initial_chunk) + chunk_destroy(ps->initial_chunk); + queue_destroy(ps->result_queue); + mtx_destroy(&ps->result_mutex); + cnd_destroy(&ps->result_not_empty); + cnd_destroy(&ps->result_not_full); + free(ps); +} From d2d1b63f4478266126a410df34a1370b6bb2eb5e Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 13:38:54 +0200 Subject: [PATCH 03/10] refactor(receive): split file_save/incremental_check/delete_commit out Pure structural split of src/shared/file_receive.c into focused translation units behind the unchanged file_receive.h facade: - file_save.c : save-to-disk, special nodes, --delay-updates staging - incremental_check.c : xattr/delta/basis/fuzzy receive + check state machine - delete_commit.c : manifest receive + delete budget walkers - file_receive.c : wire receive dispatch + deferred dir metadata The shared receive_file_xattrs helper and MAX_FILE_DATA_SIZE are declared in incremental_check.h. file_save_to_disk_full_ex is decomposed into static helpers (validation, special dispatch, dir/symlink creation, path resolution, pre-write policies, data install) routed through one cleanup epilogue. No behavior change. --- CMakeLists.txt | 3 + src/shared/delete_commit.c | 635 ++++++ src/shared/delete_commit.h | 110 ++ src/shared/file_receive.c | 3287 +------------------------------- src/shared/file_receive.h | 151 +- src/shared/file_save.c | 1164 +++++++++++ src/shared/file_save.h | 40 + src/shared/incremental_check.c | 1676 ++++++++++++++++ src/shared/incremental_check.h | 41 + 9 files changed, 3680 insertions(+), 3427 deletions(-) create mode 100644 src/shared/delete_commit.c create mode 100644 src/shared/delete_commit.h create mode 100644 src/shared/file_save.c create mode 100644 src/shared/file_save.h create mode 100644 src/shared/incremental_check.c create mode 100644 src/shared/incremental_check.h diff --git a/CMakeLists.txt b/CMakeLists.txt index a55d93e..dbddf3b 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -99,17 +99,20 @@ set(SHARED_SRCS src/shared/daemon_limits.c src/shared/data.c src/shared/delay_updates.c + src/shared/delete_commit.c src/shared/delete_plan.c src/shared/delta.c src/shared/file.c src/shared/file_list.c src/shared/file_receive.c + src/shared/file_save.c src/shared/file_send.c src/shared/file_store.c src/shared/filter.c src/shared/format.c src/shared/hardlink.c src/shared/identity.c + src/shared/incremental_check.c src/shared/log.c src/shared/metadata.c src/shared/motd.c diff --git a/src/shared/delete_commit.c b/src/shared/delete_commit.c new file mode 100644 index 0000000..a522dd8 --- /dev/null +++ b/src/shared/delete_commit.c @@ -0,0 +1,635 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "array_list.h" +#include "charset.h" +#include "chmod.h" +#include "chunk.h" +#include "compression.h" +#include "config.h" +#include "data.h" +#include "delay_updates.h" +#include "delete_commit.h" +#include "delta.h" +#include "file.h" +#include "format.h" +#include "identity.h" +#include "log.h" +#include "metadata.h" +#include "protocol.h" +#include "utils.h" +#include "xattr.h" + +#define MAX_SERVER_DELETE_COUNT 100000U +/* Retained cost of one delete-manifest entry beyond its path bytes: the + ArrayList pointer slot plus an approximate malloc header/rounding for the + heap copy. Charged against MAX_MANIFEST_BYTES so a frame full of tiny paths + cannot retain far more than the byte budget (B5). */ +#define MANIFEST_ENTRY_OVERHEAD (sizeof(char*) + 16) + +/* Read a delete-manifest frame (the STATUS_MANIFEST leading code has already + been consumed): a keep-set entry count followed by that many + destination-relative paths, then a protected-prefix count followed by that + many destination-relative prefixes, then a missing-args count followed by that + many destination-relative delete paths, then (protocol 2.23.0) a + synchronized-directory count followed by that many destination-relative + directory paths (the receive root is the "." sentinel). The frame is + self-delimiting (the counts are authoritative), so the caller decides what to + do next and continues reading the following STATUS_* frame. Every section is + validated identically: an entry must be non-empty, relative and traversal-free + and the aggregate length across ALL sections is capped by MAX_MANIFEST_BYTES + (so the missing-args deletion requests are confined like the rest of the + manifest). Returns an owned DeleteManifest, or NULL after sending STATUS_ERROR + when the frame is malformed (bad count, empty/absolute path, path traversal, + or an aggregate size beyond MAX_MANIFEST_BYTES). */ +static bool receive_manifest_section(int fd, ArrayList* list, size_t* manifest_bytes, + size_t* manifest_entries) { + int count; + if (!receive_int(fd, &count)) { + send_status(fd, STATUS_ERROR); + return false; + } + if (count < 0 || count > MAX_MANIFEST_ENTRIES || + (size_t)count > MAX_MANIFEST_ENTRIES - *manifest_entries) { + send_status(fd, STATUS_ERROR); + return false; + } + for (int i = 0; i < count; i++) { + char* s = receive_wire_str(fd); + size_t entry_size = s ? strlen(s) + MANIFEST_ENTRY_OVERHEAD : 0; + if (!s || s[0] == '\0' || s[0] == '/' || has_path_traversal(s) || + entry_size > MAX_MANIFEST_BYTES - *manifest_bytes || + (*manifest_bytes += entry_size) > MAX_MANIFEST_BYTES || !array_list_add(list, s)) { + free(s); + send_status(fd, STATUS_ERROR); + return false; + } + } + *manifest_entries += (size_t)count; + return true; +} + +DeleteManifest* receive_manifest_entries(int fd) { + DeleteManifest* manifest = calloc(1, sizeof(DeleteManifest)); + if (!manifest) { + send_status(fd, STATUS_ERROR); + return NULL; + } + manifest->keeps = array_list_create(free); + manifest->protected = array_list_create(free); + manifest->missing = array_list_create(free); + manifest->dirs = array_list_create(free); + if (!manifest->keeps || !manifest->protected || !manifest->missing || !manifest->dirs) { + delete_manifest_free(manifest); + send_status(fd, STATUS_ERROR); + return NULL; + } + size_t manifest_bytes = 0; + size_t manifest_entries = 0; + if (!receive_manifest_section(fd, manifest->keeps, &manifest_bytes, &manifest_entries) || + !receive_manifest_section(fd, manifest->protected, &manifest_bytes, &manifest_entries) || + !receive_manifest_section(fd, manifest->missing, &manifest_bytes, &manifest_entries) || + !receive_manifest_section(fd, manifest->dirs, &manifest_bytes, &manifest_entries)) { + delete_manifest_free(manifest); + return NULL; + } + return manifest; +} + +void delete_manifest_free(DeleteManifest* manifest) { + if (!manifest) + return; + array_list_delete(manifest->keeps); + array_list_delete(manifest->protected); + array_list_delete(manifest->missing); + array_list_delete(manifest->dirs); + free(manifest); +} + +/* Shared --max-delete budget for one receiver-side deletion commit. Both the + --delete-missing-args exact-path removals and the ordinary extras walk draw + from the same tally, matching rsync (whose --max-delete counts every deleted + file or directory). `max_delete` is SIZE_MAX for an unlimited budget. */ +typedef struct { + size_t max_delete; + size_t deleted; + size_t skipped; + bool limit_hit; +} DeleteBudgetState; + +/* Build the delete-walk protection prefix for one basis directory. The walker + compares paths relative to the receive root, so a relative entry is already + in the right form; an absolute entry that lies below the root is converted to + its root-relative form, and one outside the root returns NULL (the walk + cannot reach it, and it is not protected data beneath the root). Exposed so + tests can exercise the root-of-"/" child mapping directly. */ +char* file_receive_basis_delete_relative(const Config* config, const char* path) { + if (!path) + return NULL; + if (path[0] != '/') + return str_dup(path); + const char* root = config->receive_root_directory; + if (!root || root[0] != '/') + return NULL; + size_t root_len = strlen(root); + while (root_len > 1 && root[root_len - 1] == '/') + root_len--; + if (strncmp(path, root, root_len) != 0) + return NULL; + if (root_len == 1) { + /* `root` is "/" (the only single-character absolute root): every absolute + path is below it, and the child relative form is everything after the + leading '/'. */ + if (path[1] == '\0') + return NULL; /* identical to the root, not a child */ + return str_dup(path + 1); + } + if (path[root_len] != '/') + return NULL; /* identical or a sibling sharing a name prefix */ + return str_dup(path + root_len + 1); +} + +/* Remove every destination entry under the receive root that is not in the + keep-set, bounded by the shared budget (a smaller client --max-delete=NUM + replaces the server hard bound; rsync deletes up to the bound and skips the + rest). With --delay-updates the not-yet-published staging directory is a + direct child of the receive root and must not be treated as a set of extras; + the manifest's protected prefixes (paths excluded on the source), the + size-pruned prefixes (--max-size/--min-size, always protected) and the + alternate basis directories are never destination content and are skipped at + any depth. Returns true unless a traversal/unlink error aborted the walk; + the budget's limit_hit/skipped fields report a cap-stopped run. */ +static bool delete_extras_budgeted_observed(const Config* config, DeleteManifest* manifest, + DeleteBudgetState* budget, DeletePathObserver observer, + void* observer_context) { + if (!config || !manifest || !manifest->keeps) + return false; + fprintf(stderr, "Deleting files not in manifest...\n"); + /* Protected entries: + - the --delay-updates staging name, protected only as a DIRECT child of the + receive root (a nested destination directory that happens to be named + .fastsync-stage is ordinary content); + - alternate basis directories (--compare-dest / --copy-dest / --link-dest) + at any depth: they are extra comparison snapshots the user pointed at, + not destination content, and deleting them would destroy the very files a + --link-dest run just linked into place; + - the sender-side protected prefixes (source paths excluded by filters and + paths pruned by --max-size/--min-size), at any depth, so their destination + mirror survives --delete unless --delete-excluded opts back into removing + the filter-excluded ones (size-pruned entries are always protected). */ + int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count + + (manifest->protected ? manifest->protected->size : 0); + DeleteSkipEntry* skips = NULL; + char** owned_prefixes = NULL; + int used = 0; + if (skip_count > 0) { + skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); + owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); + if (!skips || (config->basis_count > 0 && !owned_prefixes)) { + free(skips); + free(owned_prefixes); + return false; + } + int idx = 0; + if (config->delay_updates) { + skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; + skips[idx].top_level_only = true; + idx++; + } + for (int i = 0; i < config->basis_count; i++) { + /* An absolute basis outside the receive root is unreachable by this walk, + so it contributes no protection prefix (and no slot). */ + char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); + if (!prefix) + continue; + owned_prefixes[i] = prefix; + skips[idx].prefix = prefix; + skips[idx].top_level_only = false; + idx++; + } + for (int i = 0; i < manifest->protected->size; i++) { + skips[idx].prefix = (const char*)manifest->protected->items[i]; + skips[idx].top_level_only = false; + idx++; + } + used = idx; + } + /* Clamp rather than subtract: an accounting bug where deleted already exceeds + max_delete must never underflow into an effectively unlimited budget. */ + size_t remaining; + if (budget->max_delete == SIZE_MAX) + remaining = SIZE_MAX; + else if (budget->deleted >= budget->max_delete) + remaining = 0; + else + remaining = budget->max_delete - budget->deleted; + size_t deleted = 0; + size_t skipped = 0; + DeleteWalkResult result = delete_extras_limited_observed( + config->receive_root_directory, manifest->keeps, manifest->dirs, remaining, skips, used, + config->protect_rules, &deleted, &skipped, observer, observer_context); + if (owned_prefixes) { + for (int i = 0; i < config->basis_count; i++) + free(owned_prefixes[i]); + } + free(owned_prefixes); + free(skips); + budget->deleted += deleted; + budget->skipped += skipped; + if (result == DELETE_WALK_LIMIT_REACHED) { + budget->limit_hit = true; + return true; + } + if (result != DELETE_WALK_OK) { + log_message(LOG_LEVEL_ERROR, "deletion failed while removing extraneous files"); + return false; + } + return true; +} + +static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifest, + DeleteBudgetState* budget) { + return delete_extras_budgeted_observed(config, manifest, budget, NULL, NULL); +} + +/* Prefixes every observed path with a fixed subtree root, so a nested walk + (a recursively removed missing-arg directory) reports receive-root-relative + names like the rest of the delete output. */ +typedef struct { + DeletePathObserver inner; + void* inner_context; + const char* prefix; +} PrefixedDeleteObserver; + +static void prefixed_delete_observer(void* context, const char* rel) { + PrefixedDeleteObserver* prefixed = context; + if (!prefixed->inner || !rel) + return; + char* joined = path_cat((char*)prefixed->prefix, rel); + if (joined) { + prefixed->inner(prefixed->inner_context, joined); + free(joined); + } +} + +/* --delete-missing-args exact-path deletions: each destination mirror in + manifest->missing is an explicit user request, so it is removed even when the + ordinary extras walk (with its protected prefixes) would leave it alone. The + --delay-updates staging directory and basis snapshots are receiver artifacts + and stay protected exactly as in the extras walker. A regular file or + symlink is unlinked, an empty directory removed, and a NON-empty directory is + removed recursively only when --delete or --force is in effect (rsync parity: + the man page says a non-empty directory mirror is only deleted with --force + or --delete); otherwise it is left with a warning and the run continues. A + mirror that does not exist is a no-op. Each removal draws from the shared + --max-delete budget: once it is exhausted the remaining requests are skipped + and counted. Returns false only on a genuine error (a confinement failure on + a validated path or an I/O error), which fails the run. */ +static bool delete_missing_args_budgeted_observed(const Config* config, DeleteManifest* manifest, + DeleteBudgetState* budget, + DeletePathObserver observer, + void* observer_context) { + if (!config || !manifest) + return false; + if (!manifest->missing || manifest->missing->size == 0) + return true; + fprintf(stderr, "Deleting destination mirrors of missing source arguments...\n"); + int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count; + DeleteSkipEntry* skips = NULL; + char** owned_prefixes = NULL; + int used = 0; + if (skip_count > 0) { + skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); + owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); + if (!skips || (config->basis_count > 0 && !owned_prefixes)) { + free(skips); + free(owned_prefixes); + return false; + } + int idx = 0; + if (config->delay_updates) { + skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; + skips[idx].top_level_only = true; + idx++; + } + for (int i = 0; i < config->basis_count; i++) { + char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); + if (!prefix) + continue; + owned_prefixes[i] = prefix; + skips[idx].prefix = prefix; + skips[idx].top_level_only = false; + idx++; + } + used = idx; + } + bool ok = true; + for (int i = 0; i < manifest->missing->size; i++) { + const char* rel = (const char*)manifest->missing->items[i]; + if (!rel || *rel == '\0' || *rel == '/' || has_path_traversal(rel)) { + /* Defensive only: receive_manifest_entries already validated every + section identically, so a controlled peer never reaches this branch. */ + log_message(LOG_LEVEL_ERROR, "invalid missing-args delete path"); + ok = false; + continue; + } + bool at_root = strchr(rel, '/') == NULL; + if (path_under_skip_prefix(rel, at_root, skips, used)) { + char* escaped = output_escape(rel, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, + "missing-args path '%s' is protected (staging directory or basis snapshot); " + "not deleting", + escaped ? escaped : ""); + free(escaped); + continue; + } + char* full = path_cat(config->receive_root_directory, rel); + if (!full) { + ok = false; + continue; + } + char* leaf = NULL; + int parent_fd = file_open_secure_parent(full, &leaf, false); + if (parent_fd < 0) { + /* The mirror's parent directory may itself not exist on the destination + (a deeper missing entry whose leading directories were never created). + That is a no-op -- there is nothing to delete -- matching + file_remove_tree_secure's absent-path handling; only a genuine I/O + error (EACCES, a symlink loop, ...) fails the run. */ + bool absent = errno == ENOENT || errno == ENOTDIR; + free(full); + free(leaf); + if (!absent) + ok = false; + continue; + } + struct stat st; + if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) != 0) { + /* Already absent: nothing to delete (a no-op, not a deletion). */ + if (errno != ENOENT) + ok = false; + close(parent_fd); + free(leaf); + free(full); + continue; + } + /* An entry that exists is one deletion: skip it (and count it) when the + shared --max-delete budget is already exhausted. */ + if (budget->deleted >= budget->max_delete) { + budget->limit_hit = true; + budget->skipped++; + close(parent_fd); + free(leaf); + free(full); + continue; + } + bool removed = false; + if (S_ISDIR(st.st_mode)) { + if (unlinkat(parent_fd, leaf, AT_REMOVEDIR) == 0) { + removed = true; + } else if (errno == ENOTEMPTY || errno == EEXIST) { + close(parent_fd); + parent_fd = -1; + free(leaf); + leaf = NULL; + if (config->use_delete || config->force_delete) { + /* Remove the contents entry-by-entry through the budgeted extras + walker so every deleted file/dir counts toward --max-delete (rsync + parity); the now-empty directory itself costs one more. A run that + hits the cap leaves the remaining entries in place. */ + ArrayList* no_keeps = array_list_create(free); + /* Never let an accounting slip (deleted > max_delete) underflow the + remaining budget into SIZE_MAX, which would grant unlimited + deletions. */ + size_t remaining = + budget->deleted >= budget->max_delete ? 0 : budget->max_delete - budget->deleted; + size_t contents_deleted = 0; + size_t contents_skipped = 0; + PrefixedDeleteObserver nested = {observer, observer_context, rel}; + DeleteWalkResult walk = + no_keeps ? delete_extras_limited_observed(full, no_keeps, NULL, remaining, NULL, 0, + NULL, &contents_deleted, &contents_skipped, + observer ? prefixed_delete_observer : NULL, + observer ? &nested : NULL) + : DELETE_WALK_ERROR; + if (no_keeps) + array_list_delete(no_keeps); + budget->deleted += contents_deleted; + budget->skipped += contents_skipped; + if (walk == DELETE_WALK_LIMIT_REACHED) { + budget->limit_hit = true; + } else if (walk != DELETE_WALK_OK) { + ok = false; + } else if (budget->deleted >= budget->max_delete) { + budget->limit_hit = true; + budget->skipped++; + } else if (file_remove_tree_secure(full)) { + /* The shared `if (removed)` tail charges this directory exactly + once; counting it here too would consume two budget units. */ + removed = true; + } else { + ok = false; + } + } else { + char* escaped = output_escape(rel, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, + "missing-args destination '%s' is a non-empty directory; use --force or " + "--delete to remove it", + escaped ? escaped : ""); + free(escaped); + } + } else if (errno != ENOENT) { + ok = false; + } + } else { + if (unlinkat(parent_fd, leaf, 0) == 0) { + removed = true; + } else if (errno != ENOENT) { + ok = false; + } + } + if (removed) { + budget->deleted++; + if (observer) + observer(observer_context, rel); + char* escaped = output_escape(rel, log_get_8_bit_output()); + fprintf(stderr, " Deleted: %s\n", escaped ? escaped : ""); + free(escaped); + } + if (parent_fd >= 0) + close(parent_fd); + free(leaf); + free(full); + if (!ok) + break; + } + if (owned_prefixes) { + for (int i = 0; i < config->basis_count; i++) + free(owned_prefixes[i]); + } + free(owned_prefixes); + free(skips); + return ok; +} + +/* Public wrappers used outside the commit path (and by unit tests): no + --max-delete budget. */ +bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, ArrayList* out, + size_t* count_out) { + if (count_out) + *count_out = 0; + if (!config || !manifest || !manifest->keeps || !out) + return false; + int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count + + (manifest->protected ? manifest->protected->size : 0); + DeleteSkipEntry* skips = NULL; + char** owned_prefixes = NULL; + int used = 0; + if (skip_count > 0) { + skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); + owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); + if (!skips || (config->basis_count > 0 && !owned_prefixes)) { + free(skips); + free(owned_prefixes); + return false; + } + int idx = 0; + if (config->delay_updates) { + skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; + skips[idx].top_level_only = true; + idx++; + } + for (int i = 0; i < config->basis_count; i++) { + /* Normalize exactly like the real commit path: a relative entry is + already root-relative, an absolute one inside the receive root is + converted, and one outside contributes no protection prefix. */ + char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); + if (!prefix) + continue; + owned_prefixes[i] = prefix; + skips[idx].prefix = prefix; + skips[idx].top_level_only = false; + idx++; + } + for (int i = 0; i < manifest->protected->size; i++) { + skips[idx].prefix = (const char*)manifest->protected->items[i]; + skips[idx].top_level_only = false; + idx++; + } + used = idx; + } + bool ok = delete_extras_list(config->receive_root_directory, manifest->keeps, manifest->dirs, + skips, used, config->protect_rules, out, count_out); + if (owned_prefixes) { + for (int i = 0; i < config->basis_count; i++) + free(owned_prefixes[i]); + } + free(owned_prefixes); + free(skips); + return ok; +} + +bool manifest_delete_extras(const Config* config, DeleteManifest* manifest) { + DeleteBudgetState budget = { + .max_delete = SIZE_MAX, .deleted = 0, .skipped = 0, .limit_hit = false}; + return delete_extras_budgeted(config, manifest, &budget); +} + +bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest) { + DeleteBudgetState budget = { + .max_delete = SIZE_MAX, .deleted = 0, .skipped = 0, .limit_hit = false}; + return delete_missing_args_budgeted_observed(config, manifest, &budget, NULL, NULL); +} + +bool manifest_delete_missing_args_limited(const Config* config, DeleteManifest* manifest, + size_t max_delete, size_t* deleted, size_t* skipped, + bool* limit_hit) { + return manifest_delete_missing_args_limited_observed(config, manifest, max_delete, deleted, + skipped, limit_hit, NULL, NULL); +} + +bool manifest_delete_missing_args_limited_observed(const Config* config, DeleteManifest* manifest, + size_t max_delete, size_t* deleted, + size_t* skipped, bool* limit_hit, + DeletePathObserver observer, + void* observer_context) { + DeleteBudgetState budget = { + .max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false}; + bool ok = + delete_missing_args_budgeted_observed(config, manifest, &budget, observer, observer_context); + if (deleted) + *deleted = budget.deleted; + if (skipped) + *skipped = budget.skipped; + if (limit_hit) + *limit_hit = budget.limit_hit; + return ok; +} + +/* Commit every deletion family the manifest carries. The --delete-missing-args + exact-path deletions run FIRST: they are explicit user requests and must not + be blocked by the extras walker's filter-exclusion protection (a protected + leftover inside a missing-argument directory must not make that user-requested + removal fail). The ordinary extras walk then runs when --delete is active. + Both draw from one --max-delete budget; the result reports a cap-stopped + (partial) commit distinctly so the client can exit 25 like rsync. */ +DeleteCommitResult manifest_delete_all(const Config* config, DeleteManifest* manifest) { + return manifest_delete_all_counted(config, manifest, NULL); +} + +DeleteCommitResult manifest_delete_all_counted(const Config* config, DeleteManifest* manifest, + size_t* deleted) { + return manifest_delete_all_observed(config, manifest, deleted, NULL, NULL); +} + +DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteManifest* manifest, + size_t* deleted, DeletePathObserver observer, + void* observer_context) { + if (deleted) + *deleted = 0; + if (!config || !manifest) + return DELETE_COMMIT_ERROR; + /* Central no-mutation guard: a dry-run never deletes. No manifest is sent on + the dry-run path, but a hostile/buggy peer could; treat it as a no-op so + the receiver can never remove anything. */ + if (config->dry_run) + return DELETE_COMMIT_OK; + /* A client --max-delete=NUM smaller than the server's hard bound replaces it + for this run; both still bound the commit. */ + bool user_limited = + config->max_delete >= 0 && (size_t)config->max_delete < MAX_SERVER_DELETE_COUNT; + DeleteBudgetState budget = {.max_delete = user_limited ? (size_t)config->max_delete + : MAX_SERVER_DELETE_COUNT, + .deleted = 0, + .skipped = 0, + .limit_hit = false}; + if (config->delete_missing_args && + !delete_missing_args_budgeted_observed(config, manifest, &budget, observer, observer_context)) + return DELETE_COMMIT_ERROR; + if (config->use_delete && + !delete_extras_budgeted_observed(config, manifest, &budget, observer, observer_context)) + return DELETE_COMMIT_ERROR; + if (deleted) + *deleted = budget.deleted; + if (budget.limit_hit) { + if (user_limited) { + log_message(LOG_LEVEL_ERROR, "Deletions stopped due to --max-delete limit (%zu skipped)", + budget.skipped); + } else { + log_message(LOG_LEVEL_ERROR, + "Deletions stopped due to the server deletion limit of %u (%zu skipped)", + (unsigned)MAX_SERVER_DELETE_COUNT, budget.skipped); + } + return DELETE_COMMIT_LIMIT_REACHED; + } + return DELETE_COMMIT_OK; +} diff --git a/src/shared/delete_commit.h b/src/shared/delete_commit.h new file mode 100644 index 0000000..c2d374b --- /dev/null +++ b/src/shared/delete_commit.h @@ -0,0 +1,110 @@ +#ifndef DELETE_COMMIT_H +#define DELETE_COMMIT_H + +#include "array_list.h" +#include "config.h" +#include "utils.h" +#include + +/* Delete-commit module: delete-manifest receive plus the budgeted extras and + * --delete-missing-args walkers. These declarations are re-exported by the + * file_receive.h facade. */ + +/* A received delete-manifest frame: the keep-set (`keeps`, destination-relative + paths the sender transferred/keeps) plus `protected`, destination-relative + prefixes the sender asks the receiver never to delete (paths excluded on the + source, protected at any depth). When --delete-excluded is given the sender + transmits an empty protected list so excluded destination mirrors are treated + as ordinary extras. With --delete-missing-args a third section (`missing`) + carries the destination mirrors of explicitly-listed source entries that do + not exist: each is an exact deletion request, independent of the ordinary + extras walk (never blocked by the protected prefixes) and processed when the + manifest is committed. */ +typedef struct DeleteManifest { + ArrayList* keeps; + ArrayList* protected; + ArrayList* missing; + /* Destination-relative paths of the directories the sender synchronized for + this run. The extras walker only removes entries directly inside one of + these (the receive root is the "." sentinel); `--files-from` runs therefore + leave untransmitted directories and the unlisted parts of listed ones + alone, matching rsync's "delete only in synchronized directories". */ + ArrayList* dirs; +} DeleteManifest; + +void delete_manifest_free(DeleteManifest* manifest); +/* Read a delete-manifest frame (protocol 2.23.0): keep count + keeps, then + protected count + protected prefixes, then missing count + missing paths, + then synchronized-directory count + directory paths (self-delimiting; the + leading STATUS_MANIFEST code has been consumed). Returns an owned + DeleteManifest, or NULL after signalling STATUS_ERROR on a malformed frame. */ +DeleteManifest* receive_manifest_entries(int fd); +/* Remove destination entries under config->receive_root_directory that are not + in `manifest` (bounded, all-or-nothing walk; staging-dir, basis-dir and + protected-prefix skips). `--max-delete` and `--force` are honored here. The + caller decides WHEN to run it based on the negotiated delete timing. Returns + false (and the transfer fails) when the deletion cannot be committed. */ +bool manifest_delete_extras(const Config* config, DeleteManifest* manifest); +/* --delete-missing-args exact-path deletions: remove each destination mirror + in `manifest->missing` (never blocked by the protected prefixes, staging dir + and basis dirs excluded). A regular file/symlink is unlinked; an empty + directory is removed; a NON-empty directory is removed recursively only when + --delete or --force is in effect, otherwise it is left with a warning (rsync + parity). A missing path is a no-op. Returns false only on a genuine + confinement or I/O error (the run then fails); tolerated per-path cases are + reported and skipped. */ +bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest); +/* Budgeted form of manifest_delete_missing_args for the per-directory delete + session: each removed mirror draws from `max_delete` (SIZE_MAX = unlimited) + and the tallies are accumulated into `*deleted`/`*skipped`. `*limit_hit` is set + when the budget stopped the pass with entries left over. Returns false only + on a genuine deletion error. */ +bool manifest_delete_missing_args_limited(const Config* config, DeleteManifest* manifest, + size_t max_delete, size_t* deleted, size_t* skipped, + bool* limit_hit); +/* Observer-aware form of manifest_delete_missing_args_limited: `observer` (may + be NULL) is invoked for every destination-relative path truly removed. */ +bool manifest_delete_missing_args_limited_observed(const Config* config, DeleteManifest* manifest, + size_t max_delete, size_t* deleted, + size_t* skipped, bool* limit_hit, + DeletePathObserver observer, + void* observer_context); +/* Outcome of committing a delete manifest. LIMIT_REACHED reports rsync's + partial --max-delete result: the budget allowed some deletions and the rest + were skipped (the run still stores all file data but the client exits 25). */ +typedef enum { + DELETE_COMMIT_OK = 0, + DELETE_COMMIT_LIMIT_REACHED, + DELETE_COMMIT_ERROR +} DeleteCommitResult; + +/* Run every deletion family the manifest carries: the --delete-missing-args + exact-path deletions first (user requests are not blocked by exclusion + protection), then the ordinary extras walk when --delete is active. Both + share one --max-delete budget. Returns DELETE_COMMIT_OK when nothing was to + do or everything committed, DELETE_COMMIT_LIMIT_REACHED when the budget + stopped part of the work, or DELETE_COMMIT_ERROR on a genuine failure. */ +DeleteCommitResult manifest_delete_all(const Config* config, DeleteManifest* manifest); +/* Like manifest_delete_all, but reports how many destination entries the commit + removed (for the end-of-transfer wire stats). `deleted` may be NULL. */ +DeleteCommitResult manifest_delete_all_counted(const Config* config, DeleteManifest* manifest, + size_t* deleted); +/* Observer-aware form of manifest_delete_all_counted: `observer` (may be NULL) + is invoked for every destination-relative path truly removed. */ +DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteManifest* manifest, + size_t* deleted, DeletePathObserver observer, + void* observer_context); + +/* -n/--dry-run --delete would-delete reporting: walk the destination exactly as + the delete pass would and append (strdup'd) destination-relative paths that + WOULD be removed to `out`, without touching disk. Uses the same staging-dir, + basis-dir and protected-prefix skips as the real commit. Returns true on a + clean walk; `*count_out` receives the number of paths appended. */ +bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, ArrayList* out, + size_t* count_out); +/* Convert one basis-directory path to the receive-root-relative protection + prefix the delete walker uses (NULL when it lies outside the root). Exposed + for unit tests of the root-of-"/" and normalization edge cases. */ +char* file_receive_basis_delete_relative(const Config* config, const char* path); + +#endif diff --git a/src/shared/file_receive.c b/src/shared/file_receive.c index 3abbda1..a9e02ab 100644 --- a/src/shared/file_receive.c +++ b/src/shared/file_receive.c @@ -20,2701 +20,16 @@ #include "delay_updates.h" #include "delta.h" #include "file.h" +#include "file_receive.h" #include "format.h" #include "identity.h" +#include "incremental_check.h" #include "log.h" #include "metadata.h" #include "protocol.h" #include "utils.h" #include "xattr.h" -#define MAX_SERVER_DELETE_COUNT 100000U -#define MAX_FILE_DATA_SIZE MAX_RECEIVE_WHOLE_FILE_SIZE -/* Retained cost of one delete-manifest entry beyond its path bytes: the - ArrayList pointer slot plus an approximate malloc header/rounding for the - heap copy. Charged against MAX_MANIFEST_BYTES so a frame full of tiny paths - cannot retain far more than the byte budget (B5). */ -#define MANIFEST_ENTRY_OVERHEAD (sizeof(char*) + 16) - -bool file_save_to_disk(const char* root_directory, const File* file, const Config* config) { - return file_save_to_disk_full_ex(root_directory, file, config, NULL, NULL) != FILE_SAVE_ERROR; -} - -FileSaveResult file_save_to_disk_full(const char* root_directory, const File* file, - const Config* config) { - return file_save_to_disk_full_ex(root_directory, file, config, NULL, NULL); -} - -/* --delay-updates receiver path: write the file into a private staging tree - below the receive root instead of its final destination, and remember it so - it can be atomically renamed into place only once the whole transfer has - succeeded. Existence/update policies (--existing/--ignore-existing/--update) - are decided against the FINAL destination path at stage time so the run - decides exactly what an immediate (non-delayed) run would decide; the staged - file is then never re-checked at publication. Backups are deferred to - publication so the final destination is untouched until the transfer ends. */ -static FileSaveResult file_stage_delayed_update(const char* root_directory, - const char* destination_path, const File* file, - Config* config) { - if (!config) - return FILE_SAVE_ERROR; - bool sparse = config->preserve_sparse; - FileAttrPolicy policy = file_attr_policy_from_config(config); - - if (config->existing && !file_path_exists_secure(destination_path)) - return FILE_SAVE_SKIPPED; - if (config->ignore_existing && file_path_exists_secure(destination_path)) - return FILE_SAVE_SKIPPED; - if (config->update && file_destination_is_newer_secure(destination_path, file->metadata)) - return FILE_SAVE_SKIPPED; - - FileMetadata adjusted_metadata; - const FileMetadata* metadata = file->metadata; - if (metadata && config->chmod_spec && *config->chmod_spec) { - adjusted_metadata = *metadata; - if (!chmod_apply(adjusted_metadata.mode, config->chmod_spec, &adjusted_metadata.mode)) - return FILE_SAVE_ERROR; - metadata = &adjusted_metadata; - } - - if (!config->delay_context) { - config->delay_context = delay_updates_context_create(root_directory); - if (!config->delay_context) - return FILE_SAVE_ERROR; - } - DelayUpdatesContext* context = config->delay_context; - if (!delay_updates_prepare(context)) - return FILE_SAVE_ERROR; - - char* staged_path = path_cat(context->staging_root, file->path); - if (!staged_path) - return FILE_SAVE_ERROR; - - /* The staged location is brand new (stale leftovers from a prior crash were - wiped by prepare), so the plain atomic temp+rename engine installs the - complete file there. --temp-dir scratch is deliberately not layered on - top of the delay-updates staging tree. A --link-dest basis file is hard - linked into the staging tree (so publication's rename keeps the link). */ - bool ok; - if (file->basis_link) { - ok = file_to_disk_secure_link(staged_path, file->basis_link, file->data->data, file->data->size, - config->preallocate, metadata, policy, config->use_fsync, NULL); - } else if (file->basis_copy) { - /* --copy-dest basis hit: stream the basis into the staging tree (bounded - buffers, so an over-limit basis still stages). */ - ok = file_copy_basis_stream_attrs(staged_path, file->basis_copy, file->data->size, - config->preallocate, metadata, policy, config->update, - config->use_fsync, file->xattrs, config->fake_super, NULL); - } else { - ok = - file_to_disk_secure_attrs(staged_path, file->data->data, file->data->size, false, sparse, - config->preallocate, metadata, policy, false, false, - config->use_fsync, file->xattrs, config->fake_super, false, NULL); - } - if (!ok) { - free(staged_path); - return FILE_SAVE_ERROR; - } - - if (!delay_updates_record(context, staged_path, destination_path, file->path)) { - unlink(staged_path); - free(staged_path); - return FILE_SAVE_ERROR; - } - free(staged_path); - return FILE_SAVE_WRITTEN; -} - -/* Read the whole content of a confined regular file (used to fall back to a - byte-identical copy when a hard-link sibling's link() fails). Symlink-safe - (parent resolved via file_open_secure_parent + O_NOFOLLOW). A zero-length - file yields *out_size 0 and *out_buf NULL as a SUCCESS. Returns false only - on a real error/read failure, setting *source_absent to true when the reason - was that the path does not exist (ENOENT/ENOTDIR), so the caller can decide - between an abort and a graceful skip. */ -static bool hardlink_read_source(const char* path, void** out_buf, unsigned long long* out_size, - bool* source_absent) { - *out_buf = NULL; - *out_size = 0; - *source_absent = false; - if (!path) - return false; - char* leaf = NULL; - int parent_fd = file_open_secure_parent(path, &leaf, false); - if (parent_fd < 0) { - *source_absent = errno == ENOENT || errno == ENOTDIR; - return false; - } - /* O_NONBLOCK is a no-op for a regular file but makes openat() fail/succeed - immediately for a client-planted FIFO instead of blocking the receive - thread forever; the post-open S_ISREG gate below is the actual type check. */ - int fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); - int saved_errno = errno; - free(leaf); - close(parent_fd); - if (fd < 0) { - *source_absent = saved_errno == ENOENT || saved_errno == ENOTDIR; - return false; - } - struct stat st; - if (fstat(fd, &st) != 0 || !S_ISREG(st.st_mode)) { - close(fd); - return false; - } - unsigned long long size = (unsigned long long)st.st_size; - if (size > MAX_RECEIVE_WHOLE_FILE_SIZE || size > SIZE_MAX) { - close(fd); - return false; - } - if (size == 0) { - close(fd); - return true; - } - void* buf = protocol_alloc((size_t)size); - if (!buf) { - close(fd); - return false; - } - size_t got = 0; - while (got < (size_t)size) { - ssize_t n = read(fd, (char*)buf + got, (size_t)size - got); - if (n <= 0) { - free(buf); - close(fd); - return false; - } - got += (size_t)n; - } - close(fd); - *out_buf = buf; - *out_size = size; - return true; -} - -/* The group's first member's installed file is absent, but its destination - path was validated (a sibling is only ever processed after its group's first - member). When the sibling's OWN destination already exists it should be - left alone -- a clean skip -- rather than aborting the whole transfer (the - asymmetric --existing case: the first member was skipped because its - destination was missing, while the sibling already has one). Only when the - sibling's destination is missing too is this a genuine failure to - link/copy, which aborts. */ -static FileSaveResult hardlink_sibling_absent_first(const char* destination_path) { - if (destination_path && file_path_exists_secure(destination_path)) - return FILE_SAVE_SKIPPED; - return FILE_SAVE_ERROR; -} - -/* Install a --hard-links/-H sibling: the destination entry is atomically - replaced (temp + rename) with a hard link to the group's first member. The - first member is guaranteed already installed at `hardlink_target` under the - root because -H relies on the receiver's single-FIFO-writer pipeline (one - receive thread, one write thread, FIFO queue => wire order == write order) - plus the sender's forced sequential scan, so a sibling is always processed - after its group's first member. When link() fails (different filesystem, - filesystem refuses links) a byte-identical copy of the first member is - written instead, so the result is never partial or corrupt. With - --delay-updates the sibling is staged as a hard link to the first member's - STAGED file (publication's renames preserve the shared inode). The final - --existing/--ignore-existing/--update policies are decided against the final - destination like every normal write. */ -static FileSaveResult file_save_hardlink_sibling(const char* root_directory, const File* file, - const Config* config, bool* created) { - Config* cfg = (Config*)config; - if (!root_directory || !file || !file->path || !file->hardlink_target) - return FILE_SAVE_ERROR; - char* destination_path = path_cat(root_directory, file->path); - if (!destination_path) - return FILE_SAVE_ERROR; - bool existed = file_path_exists_secure(destination_path); - - if (cfg->existing && !file_path_exists_secure(destination_path)) { - free(destination_path); - return FILE_SAVE_SKIPPED; - } - if (cfg->ignore_existing && file_path_exists_secure(destination_path)) { - free(destination_path); - return FILE_SAVE_SKIPPED; - } - if (cfg->update && file_destination_is_newer_secure(destination_path, file->metadata)) { - free(destination_path); - return FILE_SAVE_SKIPPED; - } - - bool preallocate = cfg && cfg->preallocate; - FileAttrPolicy policy = file_attr_policy_from_config(cfg); - bool use_fsync = cfg && cfg->use_fsync; - - if (cfg->delay_updates) { - if (!cfg->delay_context) { - cfg->delay_context = delay_updates_context_create(root_directory); - if (!cfg->delay_context) { - free(destination_path); - return FILE_SAVE_ERROR; - } - } - if (!delay_updates_prepare(cfg->delay_context)) { - free(destination_path); - return FILE_SAVE_ERROR; - } - char* staged_first = path_cat(cfg->delay_context->staging_root, file->hardlink_target); - char* staged_sibling = path_cat(cfg->delay_context->staging_root, file->path); - if (!staged_first || !staged_sibling) { - free(staged_first); - free(staged_sibling); - free(destination_path); - return FILE_SAVE_ERROR; - } - void* content = NULL; - unsigned long long content_size = 0; - bool source_absent = false; - if (!hardlink_read_source(staged_first, &content, &content_size, &source_absent)) { - FileSaveResult absent_result = - source_absent ? hardlink_sibling_absent_first(destination_path) : FILE_SAVE_ERROR; - free(staged_first); - free(staged_sibling); - free(destination_path); - return absent_result; - } - FileXattrList* sibling_xattrs = - cfg->use_xattrs ? xattr_capture_path(staged_first, cfg->preserve_acls) : NULL; - bool ok = file_to_disk_secure_link_attrs(staged_sibling, staged_first, content, content_size, - preallocate, file->metadata, policy, use_fsync, - sibling_xattrs, cfg ? cfg->fake_super : false, NULL); - xattr_list_free(sibling_xattrs); - free(content); - if (ok) - ok = delay_updates_record(cfg->delay_context, staged_sibling, destination_path, file->path); - if (!ok) - unlink(staged_sibling); - free(staged_first); - free(staged_sibling); - free(destination_path); - if (ok && created && !existed) - *created = true; - return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; - } - - char* first_disk = path_cat(root_directory, file->hardlink_target); - if (!first_disk) { - free(destination_path); - return FILE_SAVE_ERROR; - } - void* content = NULL; - unsigned long long content_size = 0; - bool source_absent = false; - if (!hardlink_read_source(first_disk, &content, &content_size, &source_absent)) { - FileSaveResult absent_result = - source_absent ? hardlink_sibling_absent_first(destination_path) : FILE_SAVE_ERROR; - free(first_disk); - free(destination_path); - return absent_result; - } - /* Resolve a relative --temp-dir under the destination root, exactly as the - * primary save path does; an absolute or `..`-escaping value is rejected. */ - char* resolved_temp = NULL; - if (cfg->temp_dir) { - if (cfg->temp_dir[0] == '/' || has_path_traversal(cfg->temp_dir)) { - free(content); - free(first_disk); - free(destination_path); - return FILE_SAVE_ERROR; - } - resolved_temp = path_cat(root_directory, cfg->temp_dir); - if (!resolved_temp) { - free(content); - free(first_disk); - free(destination_path); - return FILE_SAVE_ERROR; - } - } - FileXattrList* sibling_xattrs = - cfg->use_xattrs ? xattr_capture_path(first_disk, cfg->preserve_acls) : NULL; - bool ok = file_to_disk_secure_link_attrs( - destination_path, first_disk, content, content_size, preallocate, file->metadata, policy, - use_fsync, sibling_xattrs, cfg ? cfg->fake_super : false, resolved_temp); - xattr_list_free(sibling_xattrs); - free(resolved_temp); - free(content); - free(first_disk); - free(destination_path); - if (ok && created && !existed) - *created = true; - return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; -} - -/* Validate a transmitted special rdev against the node kind implied by `mode`'s - * S_IFMT bits. Char/block devices require a legal major/minor pair (non-negative, - * range-checked); a non-device special (FIFO/socket) must carry an empty rdev. - * Used identically on the wire path and at the secure recreation site so a - * malicious/bogus rdev can never drive a dangerous node. */ -bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode) { - bool is_device = S_ISCHR(mode) || S_ISBLK(mode); - if (is_device) - return major >= 0 && minor >= 0 && major <= 0xffff && minor <= 0x00ffffff; - /* A non-device entry must actually be a special (FIFO/socket) and carry no - rdev; a regular/dir mode is never a valid special node. */ - return (S_ISFIFO(mode) || S_ISSOCK(mode)) && major == 0 && minor == 0; -} - -/* ---- Device/special node RECREATION (--devices/--specials), receiver side ---- - * - * Privilege gating: making a real device node requires CAP_MKNOD (root); making - * a FIFO works unprivileged (mkfifo). When the receiver lacks the capability, - * mknodat() fails with EPERM and the entry is SKIPPED with a warning -- the - * whole transfer must NOT abort just because the environment cannot make the - * node. CI runs non-root, so device creation is expected to skip there and - * only a FIFO is honestly assertable unprivileged. - * - * Confinement: the parent directory is opened fd-relative below the receive - * root (file_open_secure_parent: O_NOFOLLOW, no "..", root-checked) and the - * node is created with mknodat()/mkfifoat(), so it can never be placed outside - * the confined root and never follows a symlink. - * - * rdev validation: a malicious/bogus rdev (negative, out-of-range) is rejected - * here as well as on the wire (file_receive_special / chunk_deserialize), and a - * non-device entry must carry an empty rdev. - */ -static FileSaveResult file_save_special_to_disk(const char* root_directory, const File* file, - const Config* config, bool* created) { - /* The empty-path and structural checks stay unconditional; the redundant - ".." list-path re-check is skipped under --trust-sender exactly like the - receive layer (confinement is deferred to the secure parent walk below, - which is never disabled). */ - if (!root_directory || !file || !file->path || file->path[0] == '\0' || - (!file_get_trust_sender() && has_path_traversal(file->path)) || !file->metadata) - return FILE_SAVE_ERROR; - - mode_t mode = file->metadata->mode; - bool is_char = S_ISCHR(mode); - bool is_blk = S_ISBLK(mode); - bool is_fifo = S_ISFIFO(mode); - bool is_sock = S_ISSOCK(mode); - if (!is_char && !is_blk && !is_fifo && !is_sock) { - log_message(LOG_LEVEL_ERROR, "Special node has no device/FIFO/socket mode"); - return FILE_SAVE_ERROR; - } - if (is_char || is_blk) { - if (!config || !config->preserve_devices) - return FILE_SAVE_SKIPPED; - /* --super / --no-super (P7 Wave E): char/block device-node creation is a - super-user activity. --no-super forbids it even for a root receiver; - AUTO and --super attempt it (an unprivileged attempt is refused by the - kernel and skipped). The helper is evaluated against THIS config's mode - so the policy does not depend on a prior identity_set_active(). Pure - FIFO creation is unprivileged and deliberately NOT gated here. */ - if (!privilege_super_mode_permitted(config->super_mode)) { - char* escaped_path = output_escape(file->path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, - "skipping %s: super-user device-node creation is not permitted on this receiver", - escaped_path ? escaped_path : ""); - free(escaped_path); - return FILE_SAVE_SKIPPED; - } - } else if (is_fifo || is_sock) { - /* FIFOs and unix sockets are recreated by --specials. mknod(S_IFSOCK) - works unprivileged on Linux (the node carries no live socket), so unlike - a socket bound to a live fd it can be materialized. */ - if (!config || !config->preserve_specials) - return FILE_SAVE_SKIPPED; - } - /* Defense-in-depth rdev/type validation (also done on the wire path). */ - if (!file_special_rdev_valid(file->rdev_major, file->rdev_minor, mode)) { - log_message(LOG_LEVEL_ERROR, "Rejected out-of-range device rdev %d:%d", file->rdev_major, - file->rdev_minor); - return FILE_SAVE_ERROR; - } - - char* destination = path_cat(root_directory, file->path); - if (!destination) - return FILE_SAVE_ERROR; - bool existed = file_path_exists_secure(destination); - char* leaf = NULL; - int parent_fd = file_open_secure_parent(destination, &leaf, true); - if (parent_fd < 0) { - free(destination); - return FILE_SAVE_ERROR; - } - - /* --existing / --ignore-existing / --update decide against the node that - would be replaced, mirroring the regular-file path. */ - if (config->existing && !file_path_exists_secure(destination)) { - close(parent_fd); - free(leaf); - free(destination); - return FILE_SAVE_SKIPPED; - } - if (config->ignore_existing && file_path_exists_secure(destination)) { - close(parent_fd); - free(leaf); - free(destination); - return FILE_SAVE_SKIPPED; - } - if (config->update && file_destination_is_newer_secure(destination, file->metadata)) { - close(parent_fd); - free(leaf); - free(destination); - return FILE_SAVE_SKIPPED; - } - - dev_t rdev = 0; - mode_t create_mode; - if (is_char) { - create_mode = S_IFCHR; - rdev = makedev((unsigned)file->rdev_major, (unsigned)file->rdev_minor); - } else if (is_blk) { - create_mode = S_IFBLK; - rdev = makedev((unsigned)file->rdev_major, (unsigned)file->rdev_minor); - } else if (is_sock) { - create_mode = S_IFSOCK; - } else { - create_mode = S_IFIFO; - } - const char* node_kind = (is_char || is_blk) ? "device" : (is_fifo ? "FIFO" : "socket"); - /* Under -p/--perms rsync copies the source's permission and special bits; a - * kernel that denies setuid/setgid/sticky reports the failure rather than - * having them masked here. Without -p the node is created like any other new - * entry: source_mode & 0777 & ~umask. When super-user activities are - * forbidden, the special bits are stripped even under -p (they are - * super-user activities just like device-node creation). */ - mode_t perms = config->preserve_perms ? (mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777)) - : (mode & 0777 & ~(mode_t)file_process_umask()); - if (!privilege_super_mode_permitted(config->super_mode)) - perms &= ~(mode_t)(S_ISUID | S_ISGID | S_ISVTX); - - int rc = is_fifo ? mkfifoat(parent_fd, leaf, perms) - : mknodat(parent_fd, leaf, create_mode | perms, rdev); - if (rc != 0) { - if (errno == EEXIST) { - /* An entry already exists: only skip when it already is a matching node; - never replace an existing directory or unrelated entry with the node. */ - struct stat st; - if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) == 0 && - ((is_char && S_ISCHR(st.st_mode)) || (is_blk && S_ISBLK(st.st_mode)) || - (is_fifo && S_ISFIFO(st.st_mode)) || (is_sock && S_ISSOCK(st.st_mode)))) { - close(parent_fd); - free(leaf); - free(destination); - return FILE_SAVE_SKIPPED; - } - char* escaped_path = output_escape(file->path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, "refusing to replace existing entry with %s: %s (skipped)", - node_kind, escaped_path ? escaped_path : ""); - free(escaped_path); - } else if (errno == EPERM || errno == EACCES) { - /* Missing CAP_MKNOD / parent write permission: the environment cannot - create the node, so skip instead of failing the whole run. */ - char* escaped_path = output_escape(file->path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, - "skipping %s: cannot create %s node (%s)\n" - " --devices/--specials node creation needs privilege (CAP_MKNOD)", - escaped_path ? escaped_path : "", node_kind, strerror(errno)); - free(escaped_path); - } else { - char* escaped_path = output_escape(file->path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, "failed to create %s %s: %s (skipped)", node_kind, - escaped_path ? escaped_path : "", strerror(errno)); - free(escaped_path); - } - close(parent_fd); - free(leaf); - free(destination); - return FILE_SAVE_SKIPPED; - } - - /* Apply times on the fresh node (utimensat, no-follow) per the negotiated - * per-attribute policy: mtime only under -t, atime only under -U. The slot - * not requested stays UTIME_OMIT so it is left untouched. */ - FileAttrPolicy policy = file_attr_policy_from_config(config); - if (policy.times || (policy.atimes && file->metadata->atime_valid)) { - struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT}, - {.tv_sec = 0, .tv_nsec = UTIME_OMIT}}; - if (policy.times) { - times[1].tv_sec = file->metadata->mtime_sec; - times[1].tv_nsec = file->metadata->mtime_nsec; - } - if (policy.atimes && file->metadata->atime_valid) { - times[0].tv_sec = file->metadata->atime_sec; - times[0].tv_nsec = file->metadata->atime_nsec; - } - utimensat(parent_fd, leaf, times, AT_SYMLINK_NOFOLLOW); - } - /* P7 Wave E: apply the negotiated ownership to the node ITSELF. A FIFO is - created unprivileged, but --copy-as and explicit identity policies own - every entry (a char/block node path is already privilege-gated above). The - no-follow helper changes the node's own ownership without dereferencing it; - it is a no-op unless an identity policy is active. */ - bool owner_ok = true; - if (identity_active_enabled()) - owner_ok = identity_apply_ownership_link(parent_fd, leaf, (int32_t)file->metadata->uid, - (int32_t)file->metadata->gid); - close(parent_fd); - free(leaf); - free(destination); - /* A failed required --copy-as ownership marks the node as failed; every other - * identity policy stays best-effort. */ - if (owner_ok && created && !existed) - *created = true; - return owner_ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; -} - -/* --write-devices (receiver): write the received data directly into an EXISTING - * device node on the destination instead of creating a regular file. The node - * must already exist and be a char/block device (the device itself is opened and - * followed); it is confined to the receive root via file_open_secure_parent. - * Dangerous by nature, so deliberately restricted: a missing/non-device - * destination, or a write failure, is SKIPPED with a warning rather than - * allowed. On environments without device access the run still succeeds (the - * entry is skipped), never aborts. */ -static FileSaveResult file_save_write_device(const char* root_directory, const File* file) { - if (!root_directory || !file || !file->path || file->path[0] == '\0' || - (!file_get_trust_sender() && has_path_traversal(file->path))) - return FILE_SAVE_ERROR; - if (!file->data) - return FILE_SAVE_ERROR; - char* destination = path_cat(root_directory, file->path); - if (!destination) - return FILE_SAVE_ERROR; - char* leaf = NULL; - int parent_fd = file_open_secure_parent(destination, &leaf, false); - if (parent_fd < 0) { - free(destination); - return FILE_SAVE_SKIPPED; - } - /* O_NONBLOCK: a pre-existing FIFO at the target would otherwise block the - receive thread forever on open(2). With it the open only succeeds for a - readerless FIFO with O_RDWR (which the device fstat gate rejects anyway) - or fails with ENXIO/EAGAIN, both treated as a normal skip below. */ - int fd = openat(parent_fd, leaf, O_WRONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); - int saved_errno = errno; - free(leaf); - close(parent_fd); - if (fd < 0) { - free(destination); - char* escaped_path = output_escape(file->path, log_get_8_bit_output()); - const char* shown_path = escaped_path ? escaped_path : ""; - if (saved_errno == ENXIO || saved_errno == EAGAIN) { - /* A FIFO with no reader / an unreadable special: skip like every other - unusable write-devices target instead of blocking or failing. */ - log_message(LOG_LEVEL_WARNING, "write-devices: %s not writable (%s); skipped", shown_path, - strerror(saved_errno)); - } else { - log_message(LOG_LEVEL_WARNING, "write-devices: cannot open %s (%s); skipped", shown_path, - strerror(saved_errno)); - } - free(escaped_path); - return FILE_SAVE_SKIPPED; - } - struct stat st; - if (fstat(fd, &st) != 0 || !(S_ISCHR(st.st_mode) || S_ISBLK(st.st_mode))) { - close(fd); - free(destination); - char* escaped_path = output_escape(file->path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, "write-devices: %s is not a device node; skipped", - escaped_path ? escaped_path : ""); - free(escaped_path); - return FILE_SAVE_SKIPPED; - } - bool ok = true; - if (file->data->size > 0) { - size_t total = (size_t)file->data->size; - size_t written = 0; - while (written < total) { - ssize_t n = write(fd, (char*)file->data->data + written, total - written); - if (n <= 0) { - ok = false; - break; - } - written += (size_t)n; - } - } - close(fd); - free(destination); - return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_SKIPPED; -} - -FileSaveResult file_save_to_disk_full_ex(const char* root_directory, const File* file, - const Config* config, bool* created, - unsigned* created_dirs) { - if (created) - *created = false; - if (created_dirs) - *created_dirs = 0; - /* Central no-mutation guard: a server-contacting --dry-run (or a local batch - apply that somehow carries dry_run) must never touch the destination, no - matter which caller reached this primitive. The per-caller guards remain, - but this is the last line of defense for every save path. Report SKIPPED - so a --remove-source-files sender correctly keeps its source. */ - if (config && config->dry_run) - return FILE_SAVE_SKIPPED; - /* Backups are incompatible with ignore-existing: moving the entry first - would make a concurrent no-replace commit overwrite its old name. */ - bool backup_enabled = config && config->backup && !config->ignore_existing; - bool inplace = config && config->inplace; - bool sparse = config && config->preserve_sparse; - FileAttrPolicy policy = file_attr_policy_from_config(config); - const char* backup_suffix = (config && config->suffix) ? config->suffix : "~"; - const char* backup_dir = (config && config->backup_dir) ? config->backup_dir : NULL; - const char* partial_dir = (config && config->partial_dir) ? config->partial_dir : NULL; - const char* temp_dir = (config && config->temp_dir) ? config->temp_dir : NULL; - bool use_partial_root = partial_dir && config && config->partial; - char *confined_backup = NULL, *confined_partial = NULL, *disk_path = NULL; - char* destination_path = NULL; - char *backup_path = NULL, *parent_copy = NULL; - - if (!file || !file->path || !file->data || - (file->data->size != 0 && !file->data->data && !file->basis_link && !file->basis_copy) || - (!file_get_trust_sender() && has_path_traversal(file->path)) || - (backup_enabled && - (!backup_suffix || backup_suffix[0] == '\0' || strchr(backup_suffix, '/') != NULL || - strcmp(backup_suffix, ".") == 0 || strcmp(backup_suffix, "..") == 0))) { - log_message(LOG_LEVEL_ERROR, "Invalid file or path received"); - return FILE_SAVE_ERROR; - } - - /* P7 Wave D #1: a STATUS_DIR_TIMES entry is RECORD-ONLY. The scanner - captures every traversed directory -- including empty ones whose parents - were never created by a child write and directories pruned by - -m/--prune-empty-dirs. Creating them here would resurrect empty - directories (an -a behavior change) and could abort the whole transfer on a - pre-existing regular file/symlink at the mirror path. Short-circuit before - any device/write-devices/directory branch and report it as skipped so the - sink still accumulates its metadata for the deferred DirTimeList - application, but create nothing. */ - if (file->dir_time_only) - return FILE_SAVE_SKIPPED; - - /* Device/special node (--devices/--specials): recreate the node instead of - writing content (privilege-gated, confined, rdev-validated). */ - if (file->is_special) - return file_save_special_to_disk(root_directory, file, config, created); - /* --write-devices: write straight into an existing device node. Writing - into a device is a super-user activity, so --no-super must suppress it just - like device-node creation; the default AUTO/--super attempt it (the wide - open below keeps its own confinement and best-effort skip semantics). */ - if (config && config->write_devices) { - if (!privilege_super_mode_permitted(config->super_mode)) { - char* escaped_path = output_escape(file->path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, - "write-devices: %s skipped: super-user activities are not permitted on this " - "receiver", - escaped_path ? escaped_path : "(null)"); - free(escaped_path); - return FILE_SAVE_SKIPPED; - } - return file_save_write_device(root_directory, file); - } - - /* Explicit directory entries (--dirs) carry an empty payload; the entry is - created as a directory under the receive root, applying the same secure - mkdir-parent semantics as regular writes. Directories are created - immediately (they are never staged by --delay-updates, matching rsync, - where directory creation is not delayed). */ - if (file->is_dir) { - if (file->path[0] == '\0' || (!file_get_trust_sender() && has_path_traversal(file->path))) { - log_message(LOG_LEVEL_ERROR, "Invalid directory path received"); - return FILE_SAVE_ERROR; - } - char* dir_path = path_cat(root_directory, file->path); - if (!dir_path) - return FILE_SAVE_ERROR; - bool dir_existed = file_path_exists_secure(dir_path); - bool ok = file_ensure_directory_secure(dir_path); - /* P7 Wave E: apply the negotiated ownership to the directory ITSELF (not - just the files inside it). --copy-as and every explicit identity policy - own every entry, so a directory must not keep the receiver's owner while - its children get the policy owner. Applied no-follow on the confined - parent fd after the mkdir; identity_apply_ownership_link() is itself a - no-op unless an identity policy is active. */ - if (ok && file->metadata && identity_active_enabled()) { - char* leaf = NULL; - int parent_fd = file_open_secure_parent(dir_path, &leaf, false); - if (parent_fd >= 0) { - if (!identity_apply_ownership_link(parent_fd, leaf, (int32_t)file->metadata->uid, - (int32_t)file->metadata->gid)) - ok = false; - close(parent_fd); - } else if (identity_copy_as_active()) { - /* The directory exists (ok) but its required --copy-as ownership could - not be applied because the confined parent could not be opened. */ - ok = false; - } - free(leaf); - } - free(dir_path); - if (ok && created && !dir_existed) - *created = true; - return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; - } - - /* Symlink entry. (The process-wide --keep-dirlinks policy is set once by the - connection handler from the negotiated config, before any receiver/writer - threads start, so it is stable throughout this walk.) */ - - if (file->is_symlink) { - if (!file->symlink_target || file->path[0] == '\0' || - (!file_get_trust_sender() && has_path_traversal(file->path))) { - log_message(LOG_LEVEL_ERROR, "Invalid symlink entry received"); - return FILE_SAVE_ERROR; - } - char* link_path = path_cat(root_directory, file->path); - if (!link_path) - return FILE_SAVE_ERROR; - bool link_existed = file_path_exists_secure(link_path); - /* The link value is stored verbatim (rsync -l parity: absolute and - ".."-bearing targets are preserved; the scanner's --safe-links / - --copy-unsafe-links decide which links are sent at all). --munge-links - is a RECEIVER-side rewrite: the stored target is prefixed with - /rsyncd-munged/, making the link unusable while the referenced directory - does not exist -- exactly as rsync's receiver munges. Only the link's - own placement path is confined below the receive root. */ - bool munge = config && config->munge_links; - char* target = str_dup(file->symlink_target); - bool ok = target != NULL; - if (ok && munge) { - char* munged = file_symlink_munge(target); - free(target); - target = munged; - ok = target != NULL; - } - if (!ok) { - free(target); - free(link_path); - return FILE_SAVE_SKIPPED; - } - char* parent = str_dup(link_path); - if (parent) { - /* Propagate a failed --copy-as ownership of the parent directory this - creates; every other failure mode stays best-effort as before. */ - ok = file_ensure_directory_secure(dirname(parent)); - free(parent); - } - if (ok) - ok = file_symlink_at_secure(link_path, target); - free(target); - /* P7 Wave D: apply the symlink's own metadata with no-follow primitives - (utimensat/lchown/fchmodat AT_SYMLINK_NOFOLLOW). -J/--omit-link-times - suppresses the timestamps; ownership stays gated by the identity policy. - A symlink has no children, so this can be applied immediately. */ - if (ok && config && config->use_metadata) { - FileAttrPolicy link_policy = file_attr_policy_from_config(config); - ok = file_restore_symlink_metadata(link_path, file->metadata, link_policy, - config->omit_link_times); - } - if (ok && created && !link_existed) - *created = true; - free(link_path); - return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; - } - - /* --hard-links/-H sibling: a later member of a link group arrives with no - payload and is installed as a hard link to (or, on link() failure, a - byte-identical copy of) the group's first member. Handled entirely here, - before the normal data-write paths (which would create an empty file). */ - if (file->link_group != 0 && !file->link_first && file->hardlink_target != NULL) { - return file_save_hardlink_sibling(root_directory, file, config, created); - } - - /* These options arrive from the client. --backup-dir, --partial-dir and - --temp-dir are names below the server root, never independent filesystem - roots: an absolute or `..`-escaping value is rejected outright (rsync's - daemon confines temp-dir to the module the same way). A relative temp dir - is resolved under the receive root below; if that resolution still lands on - a different filesystem than the destination the install falls back to a - non-atomic copy (see file_to_disk_secure_impl), never an abort. */ - if ((backup_dir && (backup_dir[0] == '/' || has_path_traversal(backup_dir))) || - (partial_dir && (partial_dir[0] == '/' || has_path_traversal(partial_dir))) || - (temp_dir && (temp_dir[0] == '/' || has_path_traversal(temp_dir)))) - return FILE_SAVE_ERROR; - if (backup_dir && !(confined_backup = path_cat(root_directory, backup_dir))) - return FILE_SAVE_ERROR; - if (partial_dir && !(confined_partial = path_cat(root_directory, partial_dir))) { - free(confined_backup); - return FILE_SAVE_ERROR; - } - - const char* actual_root = use_partial_root ? confined_partial : root_directory; - destination_path = path_cat(root_directory, file->path); - disk_path = path_cat(actual_root, file->path); - if (destination_path == NULL || disk_path == NULL) { - free(confined_backup); - free(confined_partial); - free(destination_path); - free(disk_path); - return FILE_SAVE_ERROR; - } - /* Snapshot the final destination's existence BEFORE any backup/force/partial - step can move or remove it, so the receiver can report rsync's - `Number of created files` (protocol 2.28.0). */ - bool dest_existed = file_path_exists_secure(destination_path); - - /* --delay-updates diverts the whole write into the staging tree; the rest of - this function is the immediate-install path. */ - if (config && config->delay_updates) { - FileSaveResult result = - file_stage_delayed_update(root_directory, destination_path, file, (Config*)config); - if (result == FILE_SAVE_WRITTEN && created && !dest_existed) - *created = true; - free(confined_backup); - free(confined_partial); - free(destination_path); - free(disk_path); - return result; - } - - /* --existing checks the final destination, not a temporary partial path. */ - if (config && config->existing && !file_path_exists_secure(destination_path)) { - free(confined_backup); - free(confined_partial); - free(destination_path); - free(disk_path); - return FILE_SAVE_SKIPPED; - } - - /* --ignore-existing checks the final destination before partial files or - overwrite policies can modify it. */ - if (config && config->ignore_existing) { - bool exists = file_path_exists_secure(destination_path); - if (exists) { - free(confined_backup); - free(confined_partial); - free(destination_path); - free(disk_path); - return FILE_SAVE_SKIPPED; - } - } - - /* --update is receiver-side policy: never replace a newer destination. - In partial-dir mode the entry that would be replaced is the real - destination, not the temporary partial file. The secure stat does not - require read permission on the destination. */ - const char* update_target = use_partial_root ? destination_path : disk_path; - if (config && config->update && file_destination_is_newer_secure(update_target, file->metadata)) { - free(confined_backup); - free(confined_partial); - free(destination_path); - free(disk_path); - return FILE_SAVE_SKIPPED; - } - - /* --force (rsync semantics): an incoming regular file may replace a - destination DIRECTORY by removing that (possibly non-empty, symlink-safe) - tree first, so the atomic temp+rename below can install the file. Only the - immediate-install path does this: a --delay-updates run stages into its own - tree and is unaffected here (its publication renames over regular files - only). The blocking directory is removed only after the --update / - --existing / --ignore-existing decisions above, which see it as an existing - destination entry. */ - if (config && config->force_delete && !file->is_dir && - file_directory_exists_secure(destination_path)) { - if (!file_remove_tree_secure(destination_path)) - goto fail; - } - - if (backup_enabled) { - /* Back up the entry that the incoming write will replace. When writing - through a partial dir the pre-existing destination file is the one to - preserve; any stale partial file is overwritten without a backup. */ - const char* replace_target = use_partial_root ? destination_path : disk_path; - struct stat backup_stat; - if (file_stat_secure(replace_target, &backup_stat)) { - if (backup_dir) { - backup_path = path_cat(confined_backup, file->path); - } else { - size_t path_len = strlen(replace_target); - size_t suffix_len = strlen(backup_suffix); - if (path_len > SIZE_MAX - suffix_len - 1) - goto fail; - backup_path = malloc(path_len + suffix_len + 1); - if (backup_path) { - memcpy(backup_path, replace_target, path_len); - memcpy(backup_path + path_len, backup_suffix, suffix_len + 1); - } - } - if (!backup_path) - goto fail; - parent_copy = str_dup(backup_path); - if (!parent_copy || !file_ensure_directory_secure(dirname(parent_copy))) - goto fail; - free(parent_copy); - parent_copy = NULL; - if (!file_rename_secure(replace_target, backup_path)) - goto fail; - free(backup_path); - backup_path = NULL; - } - } - - FileMetadata adjusted_metadata; - const FileMetadata* metadata = file->metadata; - if (metadata && config && config->chmod_spec && *config->chmod_spec) { - adjusted_metadata = *metadata; - if (!chmod_apply(adjusted_metadata.mode, config->chmod_spec, &adjusted_metadata.mode)) - goto fail; - metadata = &adjusted_metadata; - } - - /* A configured --temp-dir sends the temporary working copy to a scratch - directory; the engine then atomically renames the completed file into the - final destination directory. A relative temp dir is resolved under the - receive root and must already exist (an absolute or `..`-escaping value was - rejected above); the engine falls back to a non-atomic copy on EXDEV. The - partial-dir flow already keeps its working copy in a separate directory and - --inplace writes directly, so neither diverts through the scratch dir - (matching rsync, where --inplace/--partial-dir supersede --temp-dir). */ - char* confined_temp = NULL; - bool use_temp_dir = temp_dir != NULL && !inplace && !use_partial_root; - if (use_temp_dir) { - confined_temp = path_cat(root_directory, temp_dir); - if (!confined_temp) - goto fail; - /* A user-supplied trailing slash would leave the scratch path ending in - "/", which has no final component to create/open. Normalize it away. */ - size_t temp_len = strlen(confined_temp); - while (temp_len > 1 && confined_temp[temp_len - 1] == '/') - confined_temp[--temp_len] = '\0'; - } - /* A --link-dest basis hit installs an atomic hard link (with a byte-copy - fallback); --inplace and the update/no-replace write variants do not - apply to a fresh hard link, whose inode attributes already match. The - existing/ignore-existing/update/backup preamble above has already made the - policy decision. */ - bool ok; - char* count_floor = file_transfer_root_floor(config); - if (config && file->basis_link) { - ok = file_to_disk_secure_link_attrs_counted( - disk_path, file->basis_link, file->data->data, file->data->size, config->preallocate, - metadata, policy, config->use_fsync, file->xattrs, config->fake_super, confined_temp, - created_dirs, count_floor); - } else if (config && file->basis_copy) { - /* --copy-dest: stream the basis bytes through a bounded buffer so a basis - larger than any whole-file bound still materializes. The source - metadata was transmitted with the check frame. */ - ok = file_copy_basis_stream_attrs( - disk_path, file->basis_copy, file->data->size, config->preallocate, metadata, policy, - config->update, config->use_fsync, file->xattrs, config->fake_super, confined_temp); - } else { - /* The plain no-replace / update / with-fsync engines, plus per-file xattr - (-X/-A) and --fake-super application on the written fd. */ - ok = file_to_disk_secure_attrs_counted( - disk_path, file->data->data, file->data->size, inplace, sparse, - config && config->preallocate, metadata, policy, config && config->update, - config && config->ignore_existing, config && config->use_fsync, file->xattrs, - config ? config->fake_super : false, config ? config->partial : false, confined_temp, - created_dirs, count_floor); - } - free(count_floor); - free(confined_temp); - confined_temp = NULL; - if (!ok) - goto fail; - - /* --partial --partial-dir writes the complete file under the partial dir so - interrupted transfers leave a resumable copy there. Once the file is - fully written it must be atomically installed at the real destination; - otherwise completed transfers would linger under the partial dir. */ - if (use_partial_root) { - if (!file_rename_secure(disk_path, destination_path)) - goto fail; - } - - free(parent_copy); - free(backup_path); - free(confined_backup); - free(confined_partial); - free(destination_path); - free(disk_path); - if (created && !dest_existed) - *created = true; - return FILE_SAVE_WRITTEN; - -fail: - free(parent_copy); - free(backup_path); - free(confined_backup); - free(confined_partial); - free(destination_path); - free(disk_path); - return FILE_SAVE_ERROR; -} - -void receiver_stats_note_saved(ReceiverStats* stats, const File* file, bool created, - unsigned created_dirs) { - if (!stats || !file) - return; - /* A basis-dir hit (--link-dest/--copy-dest) materializes bytes the sender - * never transferred. rsync reports no literal data and no created entry for - * such a file, and does not count the parent directories it creates only to - * hold it, so exclude the whole entry from the receiver tallies. */ - bool basis_sourced = file->basis_link != NULL || file->basis_copy != NULL; - if (basis_sourced) - return; - bool is_sibling = file->link_group != 0 && !file->link_first; - if (!file->is_dir && !file->is_symlink && !file->is_special && !is_sibling) { - unsigned long long literal = file->literal_bytes; - if (literal == 0 && file->matched_bytes == 0) - literal = file->data ? file->data->size : 0; - stats->literal_bytes += literal; - } - stats->created_dir += created_dirs; - if (!created) - return; - if (file->is_dir) - stats->created_dir++; - else if (file->is_symlink) - stats->created_link++; - else if (file->is_special) - stats->created_special++; - else - stats->created_reg++; -} - -/* Receive a file's xattr block (when the config enables xattr transport) and - * attach it to `file`. Returns false on a malformed/oversized frame. */ -static bool receive_file_xattrs(File* file, int fd, const Config* config) { - if (!config->use_xattrs) - return true; - int xok = 0; - FileXattrList* list = xattr_receive(fd, &xok, config->preserve_acls); - if (!xok) { - xattr_list_free(list); - return false; - } - file->xattrs = list; - return true; -} - -static File* receive_delta_file(int fd, const Config* config, const char* check_path, - void* old_data, unsigned long long old_size, bool* failed) { - if (!old_data) { - free(old_data); /* defensive: old_data is always non-NULL today */ - *failed = true; - return NULL; - } - - DeltaSignature* sig = delta_signature_create_seeded(old_data, old_size, config->delta_block_size, - (uint32_t)config->checksum_seed); - if (!sig) { - free(old_data); - *failed = true; - return NULL; - } - - Data* sig_data = delta_signature_serialize(sig); - if (!sig_data) { - delta_signature_destroy(sig); - free(old_data); - *failed = true; - return NULL; - } - - bool sig_sent = send_status(fd, STATUS_DELTA_SIGNATURE) && send_data(fd, sig_data); - data_destroy(sig_data); - - if (!sig_sent) { - delta_signature_destroy(sig); - free(old_data); - *failed = true; - return NULL; - } - - Status resp; - if (!receive_status(fd, &resp)) { - delta_signature_destroy(sig); - free(old_data); - *failed = true; - return NULL; - } - - if (resp == STATUS_DELTA_DATA) { - Data* delta_data = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); - if (!delta_data) { - delta_signature_destroy(sig); - free(old_data); - *failed = true; - return NULL; - } - - Data* raw_delta = delta_data; - if (config->use_compression && - !compression_should_skip_with_suffixes( - check_path, config->skip_compress_suffixes, - config->skip_compress_set ? config->skip_compress_count : -1)) { - ProtocolSession* owner = delta_data->owner; - raw_delta = data_decompress_limited(delta_data, MAX_RECEIVE_WHOLE_FILE_SIZE); - data_destroy(delta_data); - if (!raw_delta) { - free(old_data); - delta_signature_destroy(sig); - *failed = true; - return NULL; - } - /* Charge the decompressed delta to the connection budget (the paired - wire buffer's charge was just released). */ - if (!data_charge_session(raw_delta, owner, raw_delta->size)) { - data_destroy(raw_delta); - free(old_data); - delta_signature_destroy(sig); - *failed = true; - return NULL; - } - } - - Delta* delta = delta_deserialize(raw_delta); - data_destroy(raw_delta); - if (!delta) { - free(old_data); - delta_signature_destroy(sig); - *failed = true; - return NULL; - } - - uint64_t new_size = delta->new_file_size; - if (new_size > MAX_RECEIVE_WHOLE_FILE_SIZE || new_size > SIZE_MAX) { - delta_destroy(delta); - free(old_data); - delta_signature_destroy(sig); - send_status(fd, STATUS_ERROR); - *failed = true; - return NULL; - } - /* Wire-stats tally: bytes taken straight from the basis file (matched - delta blocks) and bytes shipped literally (protocol 2.28.0). Computed - before the delta is destroyed. */ - unsigned long long matched = 0; - unsigned long long literal = 0; - for (uint32_t k = 0; k < delta->instruction_count; k++) { - if (delta->instructions[k].type == DELTA_INSTR_BLOCK_MATCH) - matched += delta->instructions[k].match.length; - else if (delta->instructions[k].type == DELTA_INSTR_LITERAL) - literal += delta->instructions[k].literal.length; - } - void* new_data = delta_apply(old_data, old_size, delta, config->delta_block_size); - delta_destroy(delta); - - if (!new_data) { - free(old_data); - delta_signature_destroy(sig); - *failed = true; - return NULL; - } - - File* file = file_create(check_path); - if (!file) { - free(new_data); - free(old_data); - delta_signature_destroy(sig); - *failed = true; - return NULL; - } - file->matched_bytes = matched; - file->literal_bytes = literal; - - if (config->use_metadata) { - int meta_ok = 1; - file->metadata = metadata_receive(fd, &meta_ok); - if (!meta_ok) { - file_destroy(file); - free(new_data); - free(old_data); - delta_signature_destroy(sig); - *failed = true; - return NULL; - } - } - if (!receive_file_xattrs(file, fd, config)) { - file_destroy(file); - free(new_data); - free(old_data); - delta_signature_destroy(sig); - *failed = true; - return NULL; - } - - Data* replacement = data_create(new_data, (size_t)new_size); - if (replacement == NULL) { - file_destroy(file); - free(old_data); - delta_signature_destroy(sig); - send_status(fd, STATUS_ERROR); - *failed = true; - return NULL; - } - data_destroy(file->data); - file->data = replacement; - - free(old_data); - delta_signature_destroy(sig); - return file; - } - - if (resp == STATUS_NEXT) { - delta_signature_destroy(sig); - free(old_data); - - File* file = file_create(check_path); - if (!file) { - *failed = true; - return NULL; - } - - if (config->use_metadata) { - int meta_ok = 1; - file->metadata = metadata_receive(fd, &meta_ok); - if (!meta_ok) { - file_destroy(file); - *failed = true; - return NULL; - } - } - if (!receive_file_xattrs(file, fd, config)) { - file_destroy(file); - *failed = true; - return NULL; - } - - Data* file_data = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); - if (file_data == NULL) { - file_destroy(file); - *failed = true; - return NULL; - } - - if (config->use_compression && - !compression_should_skip_with_suffixes( - file->path, config->skip_compress_suffixes, - config->skip_compress_set ? config->skip_compress_count : -1)) { - Data* uncompressed = data_decompress_limited(file_data, MAX_RECEIVE_WHOLE_FILE_SIZE); - ProtocolSession* owner = file_data->owner; - data_destroy(file_data); - if (uncompressed == NULL) { - file_destroy(file); - *failed = true; - return NULL; - } - if (!data_charge_session(uncompressed, owner, uncompressed->size)) { - data_destroy(uncompressed); - file_destroy(file); - send_status(fd, STATUS_ERROR); - *failed = true; - return NULL; - } - if (uncompressed->size > MAX_FILE_DATA_SIZE) { - data_destroy(uncompressed); - file_destroy(file); - send_status(fd, STATUS_ERROR); - *failed = true; - return NULL; - } - file_data = uncompressed; - } - - data_destroy(file->data); - file->data = file_data; - return file; - } - - delta_signature_destroy(sig); - free(old_data); - send_status(fd, STATUS_ERROR); - *failed = true; - return NULL; -} - -/* ---- Alternate basis directories (--compare-dest / --copy-dest / --link-dest) ---- - * The receiver consults the ordered basis-dir list only when the destination - * entry is NOT already up to date. By default an "exact match" is rsync's - * metadata quick-check: an equal size and an equal mtime (unless --size-only). - * The FastSync-only --verify-basis additionally requires an equal whole-file - * content digest, so a hard link / local copy is only then made from - * byte-verified content. */ - -typedef struct BasisMatch { - bool hit; - BasisDestType type; - char* basis_path; /* owned absolute path of the matched basis file */ - struct stat st; /* fstat() of the matched basis file */ -} BasisMatch; - -static void basis_match_free(BasisMatch* match) { - if (!match) - return; - free(match->basis_path); - match->basis_path = NULL; - match->hit = false; - match->type = BASIS_DEST_NONE; -} - -/* Open `path` (via the secure, root-confined primitives) and require it to be - a regular file of exactly `expected_size` bytes. Returns an open read-only - descriptor and its fstat on success. */ -static bool basis_open_regular(const char* path, unsigned long long expected_size, int* out_fd, - struct stat* out_st) { - char* leaf = NULL; - int parent_fd = file_open_secure_parent(path, &leaf, false); - if (parent_fd < 0) - return false; - /* O_NONBLOCK: a client-planted FIFO must not block the receiver's openat() - forever; the fstat()/S_ISREG gate below rejects it immediately. */ - int fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); - free(leaf); - close(parent_fd); - if (fd < 0) - return false; - struct stat st; - if (fstat(fd, &st) != 0 || !S_ISREG(st.st_mode) || - (unsigned long long)st.st_size != expected_size) { - close(fd); - return false; - } - *out_fd = fd; - *out_st = st; - return true; -} - -/* --ignore-times forces every file to be updated, so no basis hit is ever - declared (matching rsync, where -I prevents link-dest from linking). */ -bool file_basis_quick_match(const Config* config, const struct stat* st, time_t check_mtime, - long check_mtime_nsec) { - if (config->size_only) - return true; - long mtime_nsec = 0; -#ifdef __linux__ - mtime_nsec = st->st_mtim.tv_nsec; -#endif - return metadata_mtime_matches(st->st_mtime, mtime_nsec, check_mtime, check_mtime_nsec, - config->modify_window); -} - -/* True when a basis hit must be confirmed by a whole-file content digest - (--verify-basis). False is the rsync-parity default: the metadata - quick-check alone decides a hit. */ -bool file_basis_content_required(const Config* config) { - return config != NULL && config->verify_basis; -} - -/* Probe one candidate basis file: open it (confined, O_NOFOLLOW) and apply - rsync's metadata quick-check; under --verify-basis also hash its bytes and - require the sender's digest. On a hit record `candidate` in `out` and return - true. The caller retains ownership of `candidate`. */ -static bool basis_match_probe(const Config* config, const char* candidate, - unsigned long long check_size, time_t check_mtime, - long check_mtime_nsec, const uint8_t* check_digest, - size_t check_digest_len, BasisDestType type, BasisMatch* out) { - int fd; - struct stat st; - if (!basis_open_regular(candidate, check_size, &fd, &st)) - return false; - bool hit = false; - if (file_basis_quick_match(config, &st, check_mtime, check_mtime_nsec)) { - hit = true; - if (file_basis_content_required(config)) { - uint8_t basis_digest[CHECKSUM_MAX_DIGEST_LEN]; - size_t basis_len = 0; - bool hashed = checksum_digest_fd((ChecksumAlgo)config->checksum_algo, config->checksum_seed, - fd, basis_digest, sizeof(basis_digest), &basis_len); - hit = hashed && basis_len == check_digest_len && check_digest_len > 0 && - memcmp(basis_digest, check_digest, check_digest_len) == 0; - } - } - close(fd); - if (!hit) - return false; - char* owned = str_dup(candidate); - if (!owned) - return false; - out->hit = true; - out->type = type; - out->basis_path = owned; - out->st = st; - return true; -} - -/* Search the basis-dir list in command-line order and return the first match. - By default (no --verify-basis) rsync's metadata quick-check is sufficient: - basis_open_regular has already required an equal size, and - file_basis_quick_match applies rsync's mtime (or --size-only) rule. - --verify-basis additionally requires the basis bytes' whole-file digest to - equal the sender's, restoring FastSync's historical content equality; that - digest is computed by streaming the open basis descriptor, so an arbitrarily - large basis is verified without buffering it. A copy/link install re-reads - the basis from its path in bounded buffers, so no content buffer is kept. - - `hash_content` gates content READS under --verify-basis: a server-contacting - --dry-run passes false because hashing a basis against a client-supplied - digest would be a 1-bit content oracle. Without --verify-basis a dry-run can - still confirm the metadata-only hit without reading any basis bytes, matching - rsync's read-only quick-check. - - Path resolution (rsync 3.4.1 parity): rsync resolves a relative - --compare-dest/--copy-dest/--link-dest DIR against the destination directory - (the receiver's cwd) and appends the file's TRANSFER-RELATIVE name, e.g. - `--compare-dest=basis` with `rsync src/ dst/` probes `dst/basis/`. - FastSync's receive root IS the destination directory, but its default transfer - mirrors the absolute source path below that root, so check_path carries the - source-root scaffolding rsync would not append. Recover rsync's spelling with - utils_strip_transfer_root for a relative DIR; under -R/--files-from the wire - path is already transfer-relative, so it is used as-is. A relative DIR also - probes the historical mirror-appended spelling as a fallback, so existing - FastSync-laid-out snapshot trees keep resolving. An absolute DIR is used - verbatim and keeps appending the destination-relative check_path (FastSync's - mirrored layout). Every candidate stays confined to the authorized root by - file_open_secure_parent. */ -static bool basis_match_find(const Config* config, const char* check_path, - unsigned long long check_size, time_t check_mtime, - long check_mtime_nsec, const uint8_t* check_digest, - size_t check_digest_len, bool hash_content, BasisMatch* out) { - memset(out, 0, sizeof(*out)); - if (!config || !config_has_basis(config) || config->ignore_times) - return false; - /* --verify-basis needs the basis content; a content-blind (dry-run) pass can - never confirm it and must not read the file, so decline without touching - the basis bytes. */ - if (file_basis_content_required(config) && !hash_content) - return false; - const char* transfer_rel = check_path; - if (!config->relative && config->files_from_set == NULL) - transfer_rel = utils_strip_transfer_root(check_path, config->send_directory); - for (int i = 0; i < config->basis_count; i++) { - const BasisDest* entry = &config->basis_dirs[i]; - /* An absolute basis path is used verbatim (rsync semantics); a relative one - is resolved below the receive root. Both remain subject to the receiver's - authorized-root confinement inside file_open_secure_parent. */ - bool absolute = entry->path[0] == '/'; - char* basis_dir = - absolute ? str_dup(entry->path) : path_cat(config->receive_root_directory, entry->path); - if (!basis_dir) - continue; - const char* names[2]; - int name_count = 0; - if (absolute) - names[name_count++] = check_path; - else - names[name_count++] = transfer_rel; - if (!absolute && strcmp(transfer_rel, check_path) != 0) - names[name_count++] = check_path; /* historical mirror-appended spelling */ - bool found = false; - for (int n = 0; n < name_count && !found; n++) { - char* candidate = path_cat(basis_dir, names[n]); - if (!candidate) - continue; - found = basis_match_probe(config, candidate, check_size, check_mtime, check_mtime_nsec, - check_digest, check_digest_len, entry->type, out); - free(candidate); - } - free(basis_dir); - if (found) - return true; - } - return false; -} - -/* --------------------------------------------------------------------------- - * -y/--fuzzy similar-file delta basis. - * - * When a file must be transferred and the destination holds no usable content - * at the exact path (the destination file is absent, or is outside the delta - * engine's size bounds), --fuzzy lets the receiver reuse an EXISTING regular - * file in the SAME destination directory as the delta basis, so the sender - * transmits only the differences instead of the whole file. This is the - * rsync "find a similar file to use as a basis for a transfer" case (e.g. a - * file recreated under a new name whose old-named sibling is still present). - * - * The delta handshake is unchanged and receiver-driven, so the sender never - * learns the basis was a different file and needs no new protocol. Byte - * exactness never depends on which bytes the basis holds: the delta protocol - * only references basis blocks whose Adler-32 + xxHash32 checksums match the - * source, delta_apply validates every reference against the basis size, and a - * basis that shares nothing simply makes the sender reply STATUS_NEXT (full - * transfer). A fuzzy basis can therefore waste bandwidth but never corrupt a - * file. - * - * Similarity heuristic (rsync 3.4.1 parity, util1.c fuzzy_distance / - * find_filename_suffix + generator.c find_fuzzy): - * * candidates are the target's sibling entries in its destination - * directory, opened through the confined root (file_open_secure_parent + - * openat O_NOFOLLOW, fstatat AT_SYMLINK_NOFOLLOW) -- symlinks are never - * followed and nothing outside the destination root is ever read; - * * dotfiles, directories, the target's own name, and the .fastsync-stage / - * temp scratch names are never candidates; - * * size gate = rsync's, NOT the ordinary delta engine's bounds: any - * non-empty regular sibling up to the receiver's whole-file buffer cap is - * eligible, regardless of the 16 KiB delta minimum or the 10x delta size - * ratio (rsync's find_fuzzy has no delta-size gate at all). The delta - * engine consumes the fuzzy basis through the same signature handshake - * whether or not it is inside delta_should_attempt's window; - * * first pass = an exact size+mtime match wins regardless of name (rsync's - * "fuzzy size/modtime match"); - * * otherwise the winner minimizes rsync's weighted Levenshtein distance - * (substitution ± byte difference, insertion UNIT+byte, 16.16 fixed point) - * plus ten times the suffix distance, accepted only when <= 25*UNIT; the - * tie-break (smallest size gap, then lexical name) keeps the result - * deterministic across filesystem readdir order (rsync leaves equal - * distances to its file-list order). - * ------------------------------------------------------------------------- */ - -/* A directory scan is linear in the number of entries; the fuzzy search stops - * after this many so a pathological huge directory cannot stall a transfer. - * The cap bounds the readdir() ITERATIONS, not the per-entry work: every - * entry that survives the (cheap) size and pre-name gates still runs an - * edit-distance DP, so the per-entry DP cost is separately bounded below by - * pre-pruning on the name length gap and the absent-character bound, and by - * trimming the common prefix/suffix before the DP runs on the middles only. */ -#define FUZZY_MAX_DIRECTORY_SCAN 4096 -/* Names longer than this never take part in fuzzy matching: the edit-distance - * DP below is O(len^2), so over-long names are bounded out of the search. */ -#define FUZZY_NAME_LIMIT 192 - -typedef struct { - char name[FUZZY_NAME_LIMIT + 1]; - unsigned long long size; - uint32_t distance; - unsigned long long size_gap; -} FuzzyCandidate; - -/* rsync's fuzzy distance is a weighted Levenshtein variant in 16.16 fixed point - * (util1.c fuzzy_distance): a substitution costs UNIT +/- the byte difference - * and an insertion costs UNIT + the inserted byte, so similar names score low. - * The search keeps only distances <= 25*UNIT. Ported verbatim for parity. */ -#define FUZZY_DIST_UNIT (1u << 16) -#define FUZZY_DIST_REJECT (0xFFFFu * FUZZY_DIST_UNIT + 1) -#define FUZZY_DIST_LIMIT (25u * FUZZY_DIST_UNIT) - -static uint32_t fuzzy_distance(const char* s1, unsigned len1, const char* s2, unsigned len2, - uint32_t upperlimit, uint32_t* scratch) { - if ((len1 > len2 ? len1 - len2 : len2 - len1) * FUZZY_DIST_UNIT > upperlimit) - return FUZZY_DIST_REJECT; - if (!len1 || !len2) { - if (!len1) { - s1 = s2; - len1 = len2; - } - uint32_t cost = 0; - for (unsigned i = 0; i < len1; i++) - cost += (uint8_t)s1[i]; - return (uint32_t)len1 * FUZZY_DIST_UNIT + cost; - } - uint32_t* a = scratch; - for (unsigned i2 = 0; i2 < len2; i2++) - a[i2] = (i2 + 1) * FUZZY_DIST_UNIT; - for (unsigned i1 = 0; i1 < len1; i1++) { - uint32_t diag = i1 * FUZZY_DIST_UNIT; - uint32_t above = (i1 + 1) * FUZZY_DIST_UNIT; - for (unsigned i2 = 0; i2 < len2; i2++) { - uint32_t left = a[i2]; - int32_t cost = (int32_t)(uint8_t)s1[i1] - (int32_t)(uint8_t)s2[i2]; - if (cost != 0) - cost = cost < 0 ? (int32_t)(FUZZY_DIST_UNIT - (uint32_t)(-cost)) - : (int32_t)(FUZZY_DIST_UNIT + (uint32_t)cost); - uint32_t diag_inc = diag + (uint32_t)cost; - uint32_t left_inc = left + FUZZY_DIST_UNIT + (uint8_t)s1[i1]; - uint32_t above_inc = above + FUZZY_DIST_UNIT + (uint8_t)s2[i2]; - a[i2] = above = left < above ? (left_inc < diag_inc ? left_inc : diag_inc) - : (above_inc < diag_inc ? above_inc : diag_inc); - diag = left; - } - } - return a[len2 - 1]; -} - -/* rsync's find_filename_suffix (util1.c): return the last significant filename - * suffix (its dot included). Leading dots are not a suffix; a trailing "~" is - * ignored; .bak/.old/.orig and a "~/" backup marker are skipped. */ -static const char* fuzzy_find_suffix(const char* fn, int fn_len, int* len_ptr) { - const char* suf; - const char* s; - bool had_tilde; - - while (fn_len && *fn == '.') { - fn++; - fn_len--; - } - if (fn_len > 1 && fn[fn_len - 1] == '~') { - fn_len--; - had_tilde = true; - } else { - had_tilde = false; - } - suf = ""; - *len_ptr = 0; - for (s = fn + fn_len; fn_len > 1;) { - int s_len; - while (--s != fn && *s != '.') { - } - if (s == fn) - break; - s_len = fn_len - (int)(s - fn); - fn_len = (int)(s - fn); - if (s_len == 4) { - if (strcmp(s + 1, "bak") == 0 || strcmp(s + 1, "old") == 0) - continue; - } else if (s_len == 5) { - if (strcmp(s + 1, "orig") == 0) - continue; - } else if (s_len > 2 && had_tilde && s[1] == '~' && isdigit((unsigned char)s[2])) { - continue; - } - *len_ptr = s_len; - suf = s; - if (s_len == 1) - break; - for (s++, s_len--; s_len > 0; s++, s_len--) { - if (!isdigit((unsigned char)*s)) - return suf; - } - s = suf; - } - return suf; -} - -/* Deterministic ordering of two fuzzy candidates with equal rsync distance: - * smallest size gap, then the lexical basename (rsync itself takes the last - * equal-distance candidate in file-list order). */ -static bool fuzzy_candidate_better(const FuzzyCandidate* cand, const FuzzyCandidate* best) { - if (!best->name[0]) - return true; - if (cand->distance != best->distance) - return cand->distance < best->distance; - if (cand->size_gap != best->size_gap) - return cand->size_gap < best->size_gap; - return strcmp(cand->name, best->name) < 0; -} - -/* Search the destination directory that will contain `check_path` for a - * similar regular file usable as a --fuzzy delta basis and return its full - * content in a malloc'd (protocol_alloc) buffer. Returns NULL (with *out_size - * = 0) when no candidate qualifies, which means the caller performs the normal - * whole-file transfer. */ -static void* fuzzy_basis_find_and_load(const Config* config, const char* check_path, - unsigned long long check_size, time_t check_mtime, - long check_mtime_nsec, unsigned long long* out_size) { - *out_size = 0; - if (!config || !config->receive_root_directory || !config->fuzzy || !config->use_delta || - !check_path || check_size > MAX_RECEIVE_WHOLE_FILE_SIZE) - return NULL; - - char* full_path = path_cat(config->receive_root_directory, check_path); - if (!full_path) - return NULL; - char* leaf = NULL; - int dir_fd = file_open_secure_parent(full_path, &leaf, false); - if (dir_fd < 0 || !leaf) { - free(leaf); - free(full_path); - return NULL; - } - size_t target_len = strlen(leaf); - /* A target basename longer than FUZZY_NAME_LIMIT can never pass the name gate - (every candidate name is bounded by the same limit), so skip the scan. */ - if (target_len > FUZZY_NAME_LIMIT) { - close(dir_fd); - free(leaf); - free(full_path); - return NULL; - } - - int scanfd = dup(dir_fd); - if (scanfd < 0) { - close(dir_fd); - free(leaf); - free(full_path); - return NULL; - } - DIR* dir = fdopendir(scanfd); - if (!dir) { - close(scanfd); - close(dir_fd); - free(leaf); - free(full_path); - return NULL; - } - - /* The weighted-distance scratch row is allocated once per scan (not once per - candidate). */ - uint32_t* dist_scratch = malloc((FUZZY_NAME_LIMIT + 1) * sizeof(uint32_t)); - if (!dist_scratch) { - closedir(dir); - close(dir_fd); - free(leaf); - free(full_path); - return NULL; - } - int fname_suf_len = 0; - const char* fname_suf = fuzzy_find_suffix(leaf, (int)target_len, &fname_suf_len); - - FuzzyCandidate best; - memset(&best, 0, sizeof(best)); - uint32_t lowest_dist = FUZZY_DIST_LIMIT; - /* rsync's fuzzy search runs an exact size+mtime pass before the name-distance - pass; such a candidate is almost certainly the same content and wins - regardless of how dissimilar its name is. The first one (directory order, - deterministic) is kept. */ - FuzzyCandidate exact; - memset(&exact, 0, sizeof(exact)); - const struct dirent* entry; - size_t scanned = 0; - /* readdir() yields entries in filesystem-dependent order, so the SET of - candidates seen is order-dependent; the winner is still deterministic - because every candidate is compared with the total ordering in - fuzzy_candidate_better (acceptable for a heuristic). */ - while (scanned < FUZZY_MAX_DIRECTORY_SCAN && (entry = readdir(dir)) != NULL) { - scanned++; - const char* name = entry->d_name; - size_t name_len = strlen(name); - if (name[0] == '.' || name_len == 0 || name_len > FUZZY_NAME_LIMIT || strcmp(name, leaf) == 0) - continue; - struct stat st; - if (fstatat(dir_fd, name, &st, AT_SYMLINK_NOFOLLOW) != 0 || !S_ISREG(st.st_mode)) - continue; - unsigned long long cand_size = (unsigned long long)st.st_size; - if (cand_size == 0 || cand_size > MAX_RECEIVE_WHOLE_FILE_SIZE) - continue; - long cand_nsec = 0; -#ifdef __linux__ - cand_nsec = st.st_mtim.tv_nsec; -#endif - if (!exact.name[0] && cand_size == check_size && - metadata_mtime_matches(st.st_mtime, cand_nsec, check_mtime, check_mtime_nsec, - config->modify_window)) { - memcpy(exact.name, name, name_len + 1); - exact.size = cand_size; - exact.size_gap = 0; - continue; - } - /* rsync's name-distance pass: a weighted Levenshtein distance over the full - basenames, plus ten times the same distance over the filename suffixes, - accepted only when it does not exceed the running lowest distance. */ - int name_suf_len = 0; - const char* name_suf = fuzzy_find_suffix(name, (int)name_len, &name_suf_len); - uint32_t distance = fuzzy_distance(name, (unsigned)name_len, leaf, (unsigned)target_len, - lowest_dist, dist_scratch); - if (distance < 0xFFFF0000U) - distance += fuzzy_distance(name_suf, (unsigned)name_suf_len, fname_suf, - (unsigned)fname_suf_len, 0xFFFF0000U, dist_scratch) * - 10; - if (distance > lowest_dist) - continue; - lowest_dist = distance; - FuzzyCandidate cand; - memcpy(cand.name, name, name_len + 1); - cand.size = cand_size; - cand.distance = distance; - cand.size_gap = cand_size > check_size ? cand_size - check_size : check_size - cand_size; - if (fuzzy_candidate_better(&cand, &best)) - best = cand; - } - closedir(dir); - free(leaf); - free(dist_scratch); - - /* Prefer the exact size+mtime candidate over any name-distance winner. */ - if (exact.name[0]) - best = exact; - - void* basis = NULL; - if (best.name[0]) { - /* O_NONBLOCK: a name raced to a FIFO between the fstatat gate and this open - would otherwise block the receive thread forever on open(2); with it the - open fails (ENXIO) and the fstat/S_ISREG gate below would reject it too. - A regular file opened with O_NONBLOCK is unaffected. */ - int fd = openat(dir_fd, best.name, O_RDONLY | O_NOFOLLOW | O_NONBLOCK | O_CLOEXEC); - if (fd >= 0) { - struct stat st; - if (fstat(fd, &st) == 0 && S_ISREG(st.st_mode) && - (unsigned long long)st.st_size == best.size && best.size <= SIZE_MAX) { - basis = protocol_alloc((size_t)best.size); - if (basis) { - size_t got = 0; - while (got < (size_t)best.size) { - ssize_t n = read(fd, (char*)basis + got, (size_t)best.size - got); - if (n <= 0) { - free(basis); - basis = NULL; - break; - } - got += (size_t)n; - } - } - } - close(fd); - } - } - close(dir_fd); - free(full_path); - if (basis) - *out_size = best.size; - return basis; -} - -/* Read the remainder of a full-file transfer after the receiver has already - * sent STATUS_NEXT: receive the metadata frame (when enabled) followed by the - * data frame, and return an owned File. Shared by the plain full-transfer path - * and the --append-verify prefix-mismatch fallback (a clean full transfer - * instead of a corrupt prefix+tail blend). */ -static File* receive_full_file(int fd, const Config* config, const char* path) { - File* file = file_create(path); - if (!file) - return NULL; - if (config->use_metadata) { - int meta_ok = 1; - file->metadata = metadata_receive(fd, &meta_ok); - if (!meta_ok) { - file_destroy(file); - return NULL; - } - } - if (!receive_file_xattrs(file, fd, config)) { - file_destroy(file); - return NULL; - } - Data* file_data = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); - if (file_data == NULL) { - file_destroy(file); - return NULL; - } - if (config->use_compression && - !compression_should_skip_with_suffixes(file->path, config->skip_compress_suffixes, - config->skip_compress_set ? config->skip_compress_count - : -1)) { - Data* uncompressed = data_decompress_limited(file_data, MAX_RECEIVE_WHOLE_FILE_SIZE); - ProtocolSession* owner = file_data->owner; - data_destroy(file_data); - if (uncompressed == NULL) { - file_destroy(file); - return NULL; - } - if (!data_charge_session(uncompressed, owner, uncompressed->size)) { - data_destroy(uncompressed); - file_destroy(file); - return NULL; - } - if (uncompressed->size > MAX_FILE_DATA_SIZE) { - data_destroy(uncompressed); - file_destroy(file); - return NULL; - } - file_data = uncompressed; - } - data_destroy(file->data); - file->data = file_data; - return file; -} - -/* --------------------------------------------------------------------------- - * receive_incremental_check() decomposition. - * - * The per-file STATUS_CHECK fast path is split into the small helpers below, - * called in order by a short linear orchestrator (receive_incremental_check_ex). - * Each helper owns one decision: request validation, secure destination open, - * metadata-only skip, server-contacting --dry-run no-mutation short-circuit, - * alternate-basis match, --append tail resume, block delta, --fuzzy basis, and - * the final "send the whole file" fallback. Every protocol send/receive and - * every resource cleanup is preserved exactly; the non-dry-run wire is - * byte-for-byte unchanged. receive_incremental_check_ex additionally exposes a - * `would_transfer` out-param for the dry-run caller; the 3-arg - * receive_incremental_check wrapper passes NULL. - * ------------------------------------------------------------------------- */ - -/* Owned state threaded through the helpers below. */ -typedef struct { - int fd; - const Config* config; - char* check_path; /* received destination-relative path */ - char* full_path; /* receive-root-prefixed destination path */ - unsigned long long check_size; - long long check_mtime; - long long check_mtime_nsec; - uint8_t check_digest[CHECKSUM_MAX_DIGEST_LEN]; - size_t check_digest_len; - /* Source metadata carried alongside the check frame whenever a basis dir is - configured (rsync keeps the whole file list; FastSync's sender-driven - incremental path otherwise never transmits metadata for a SKIPPED file). - A basis materialization applies these SOURCE attributes instead of the - basis inode's, matching rsync's "copy then fix attributes". */ - FileMetadata* source_metadata; - bool dest_exists; /* any destination entry exists (lstat succeeded) */ - bool has_old_file; - int old_fd; - struct stat old_st; - unsigned long long old_size; - void* old_data; /* snapshot of the existing destination, or NULL */ -} IncrementalCheckState; - -typedef enum { - INCREMENTAL_CONTINUE, /* proceed to the next helper */ - INCREMENTAL_ERROR, /* protocol/validation failure: return NULL */ - INCREMENTAL_SKIP, /* up to date: *skipped = true, return NULL */ - INCREMENTAL_DRY_RUN, /* --dry-run resolved: flags set, return NULL */ - INCREMENTAL_FILE, /* a File* was produced (out_file) */ -} IncrementalCheckOutcome; - -static void incremental_check_state_init(IncrementalCheckState* state, int fd, - const Config* config) { - memset(state, 0, sizeof(*state)); - state->fd = fd; - state->config = config; - state->old_fd = -1; -} - -/* Release every resource the helpers may have acquired. Idempotent, so it is - safe on every exit path exactly the way the original inline cleanup was. */ -static void incremental_check_state_cleanup(IncrementalCheckState* state) { - free(state->old_data); - state->old_data = NULL; - if (state->old_fd >= 0) - close(state->old_fd); - state->old_fd = -1; - file_metadata_destroy(state->source_metadata); - state->source_metadata = NULL; - free(state->full_path); - state->full_path = NULL; - free(state->check_path); - state->check_path = NULL; -} - -/* Receive and validate the STATUS_CHECK request frame: path, size, mtime, - nanosecond mtime, and (when negotiated) the source digest. */ -static IncrementalCheckOutcome incremental_check_receive_request(IncrementalCheckState* state) { - int fd = state->fd; - const Config* config = state->config; - char* check_path = receive_wire_str(fd); - if (check_path == NULL) - return INCREMENTAL_ERROR; - state->check_path = check_path; - - if (!receive_n_data(fd, &state->check_size, sizeof(state->check_size)) || - !receive_n_data(fd, &state->check_mtime, sizeof(state->check_mtime))) - return INCREMENTAL_ERROR; - if (!receive_n_data(fd, &state->check_mtime_nsec, sizeof(state->check_mtime_nsec)) || - state->check_mtime_nsec < 0 || state->check_mtime_nsec >= 1000000000LL) { - send_error_detail(fd, "invalid check mtime nanoseconds"); - return INCREMENTAL_ERROR; - } - if ((config->checksum || config->verify_basis)) { - uint8_t wire_len; - if (!receive_n_data(fd, &wire_len, sizeof(wire_len)) || wire_len == 0 || - wire_len > CHECKSUM_MAX_DIGEST_LEN || - wire_len != checksum_digest_len((ChecksumAlgo)config->checksum_algo)) { - send_error_detail(fd, "invalid check digest length"); - return INCREMENTAL_ERROR; - } - state->check_digest_len = wire_len; - if (!receive_n_data(fd, state->check_digest, state->check_digest_len)) - return INCREMENTAL_ERROR; - } - /* The sender transmits the source metadata with every basis-configured check - so a basis hit can be materialized with the SOURCE's attributes (rsync - copies/copies-then-fixes; the receiver would otherwise only have the basis - inode's stat). The block is symmetric and consumed unconditionally here, - whether or not this file ends up as a basis hit. */ - if (config_has_basis(config) && config->use_metadata) { - int meta_ok = 1; - state->source_metadata = metadata_receive(fd, &meta_ok); - if (!meta_ok) - return INCREMENTAL_ERROR; - } - - /* A basis-configured run may materialize a file larger than the whole-file - payload bound: a basis hit is streamed from the basis path (bounded - buffers), so the check size is not itself an allocation. Every other - path (delta/append/full) still applies MAX_RECEIVE_WHOLE_FILE_SIZE, and a - miss simply falls through to the normal transfer with its own bound. */ - if (!config_has_basis(config) && state->check_size > MAX_RECEIVE_WHOLE_FILE_SIZE) { - send_error_detail(fd, "check size exceeds receiver limit"); - return INCREMENTAL_ERROR; - } - - if (check_path[0] == '\0' || has_path_traversal(check_path)) { - char* escaped_path = output_escape(check_path, log_get_8_bit_output()); - log_message(LOG_LEVEL_ERROR, "Invalid received check path: %s", - escaped_path ? escaped_path : ""); - free(escaped_path); - return INCREMENTAL_ERROR; - } - return INCREMENTAL_CONTINUE; -} - -/* Open the existing destination entry once, confined below the receive root, - and record its stat. */ -static IncrementalCheckOutcome incremental_check_open_destination(IncrementalCheckState* state) { - char* full_path = path_cat(state->config->receive_root_directory, state->check_path); - if (!full_path) { - send_error_detail(state->fd, "could not build destination path"); - return INCREMENTAL_ERROR; - } - state->full_path = full_path; - - char* leaf = NULL; - int parent_fd = file_open_secure_parent(full_path, &leaf, false); - if (parent_fd >= 0) { - struct stat dest_st; - if (fstatat(parent_fd, leaf, &dest_st, AT_SYMLINK_NOFOLLOW) == 0) - state->dest_exists = true; - /* O_NONBLOCK: an existing FIFO at the destination must not block this - openat(); the S_ISREG gate below rejects the non-regular entry. */ - state->old_fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); - free(leaf); - close(parent_fd); - state->has_old_file = state->old_fd >= 0 && fstat(state->old_fd, &state->old_st) == 0 && - S_ISREG(state->old_st.st_mode); - } - if (!state->has_old_file && state->old_fd >= 0) { - close(state->old_fd); - state->old_fd = -1; - } - state->old_size = state->has_old_file ? (unsigned long long)state->old_st.st_size : 0; - return INCREMENTAL_CONTINUE; -} - -/* Output parity (protocol 2.23.0): when the wire config asked for it, report a - snapshot of the pre-transfer destination entry BEFORE the ordinary verdict so - the sender can render rsync-accurate -i/--out-format columns. A missing - destination is reported explicitly (existed=false) rather than omitted, so - the sender can distinguish "new" from "unknown". */ -static IncrementalCheckOutcome incremental_check_report_dest_info(IncrementalCheckState* state) { - if (!state->config->report_dest_info) - return INCREMENTAL_CONTINUE; - OutputDestState info; - memset(&info, 0, sizeof(info)); - info.known = true; - info.existed = state->has_old_file; - if (state->has_old_file) { - info.size = (unsigned long long)state->old_st.st_size; - info.mtime_sec = (long long)state->old_st.st_mtime; -#ifdef __linux__ - info.mtime_nsec = state->old_st.st_mtim.tv_nsec; -#endif - info.mode = (uint32_t)state->old_st.st_mode; - info.uid = (int32_t)state->old_st.st_uid; - info.gid = (int32_t)state->old_st.st_gid; - } - if (!send_status(state->fd, STATUS_DEST_INFO) || !format_dest_state_send(state->fd, &info)) - return INCREMENTAL_ERROR; - return INCREMENTAL_CONTINUE; -} - -/* --ignore-existing short-circuit. The receiver must answer "skip" (STATUS_OK) - BEFORE the sender transmits any payload, otherwise the whole file crosses the - wire only to be discarded at write time. rsync skips an existing destination - entry regardless of its content or type, so the reply depends only on the - lstat existence probe; the ordinary --ignore-existing checks inside - file_receive remain as defense-in-depth for the frame types that have no - per-file check (directories/symlinks/specials/hard-links). */ -static IncrementalCheckOutcome -incremental_check_ignore_existing(const IncrementalCheckState* state) { - if (!state->config->ignore_existing || !state->dest_exists) - return INCREMENTAL_CONTINUE; - if (!send_status(state->fd, STATUS_OK)) - return INCREMENTAL_ERROR; - return INCREMENTAL_SKIP; -} - -/* Metadata for a materialized basis hit: prefer the SOURCE metadata the sender - transmitted with the check frame (rsync copies then fixes the destination to - the source's attributes); fall back to the basis inode's own stat when - metadata was not negotiated. Consumes state->source_metadata on success. */ -static FileMetadata* basis_take_metadata(IncrementalCheckState* state, - const struct stat* basis_st) { - if (state->source_metadata) { - FileMetadata* meta = state->source_metadata; - state->source_metadata = NULL; - return meta; - } - return file_metadata_create(NULL, basis_st, false, false); -} - -/* --link-dest relink of an already up-to-date destination. rsync hard-links a - destination entry to a matching basis even when the entry is already correct, - so a run over an existing tree still maximizes sharing with the basis. Only a - link-dest basis triggers this (copy-dest/compare-dest leave an up-to-date - destination untouched, matching rsync). The ordinary basis path further down - handles every not-up-to-date case, so this helper only adds the relink that - the quick-skip would otherwise short-circuit. */ -static IncrementalCheckOutcome incremental_check_link_dest_relink(IncrementalCheckState* state, - File** out_file) { - const Config* config = state->config; - if (!config_has_basis(config) || config->ignore_times || config->dry_run) - return INCREMENTAL_CONTINUE; - if (!state->has_old_file) - return INCREMENTAL_CONTINUE; - BasisMatch basis; - basis_match_find(config, state->check_path, state->check_size, (time_t)state->check_mtime, - (long)state->check_mtime_nsec, state->check_digest, state->check_digest_len, - true, &basis); - /* Only a link-dest hit relinks; a copy-dest/compare-dest hit (or a miss) lets - the up-to-date check below keep the existing destination. */ - if (!basis.hit || basis.type != BASIS_DEST_LINK) { - basis_match_free(&basis); - return INCREMENTAL_CONTINUE; - } - /* Already the basis inode: nothing to do, leave the destination alone. */ - if (basis.st.st_dev == state->old_st.st_dev && basis.st.st_ino == state->old_st.st_ino) { - basis_match_free(&basis); - return INCREMENTAL_CONTINUE; - } - File* materialized = file_create(state->check_path); - if (materialized) { - data_destroy(materialized->data); - materialized->data = data_create_reserve((size_t)state->check_size); - if (!materialized->data) { - file_destroy(materialized); - materialized = NULL; - } - } - if (materialized) { - materialized->metadata = basis_take_metadata(state, &basis.st); - materialized->skip = true; - materialized->basis_link = basis.basis_path; - basis.basis_path = NULL; - if (!materialized->metadata) { - file_destroy(materialized); - materialized = NULL; - } - } - if (materialized) { - if (!send_status(state->fd, STATUS_OK)) { - basis_match_free(&basis); - file_destroy(materialized); - return INCREMENTAL_ERROR; - } - basis_match_free(&basis); - *out_file = materialized; - return INCREMENTAL_FILE; - } - basis_match_free(&basis); - return INCREMENTAL_CONTINUE; -} - -/* Metadata-only (and, when --checksum forces it, content) up-to-date decision. - Loads the old contents only when a checksum comparison or delta needs them. */ -static IncrementalCheckOutcome incremental_check_quick_skip(IncrementalCheckState* state, - bool* out_try_delta) { - int fd = state->fd; - const Config* config = state->config; - bool has_old_file = state->has_old_file; - unsigned long long old_size = state->old_size; - struct stat st = state->old_st; - - bool size_equal = has_old_file && old_size == state->check_size; - bool match_by_metadata = false; - if (size_equal && !config->ignore_times && !config->size_only) { - long long old_mtime_nsec = 0; -#ifdef __linux__ - old_mtime_nsec = st.st_mtim.tv_nsec; -#endif - match_by_metadata = - metadata_mtime_matches(st.st_mtime, old_mtime_nsec, (time_t)state->check_mtime, - (long)state->check_mtime_nsec, config->modify_window); - } - - bool try_delta = config->use_delta && !config->whole_file && has_old_file && - delta_should_attempt(old_size, state->check_size, config->delta_max_file_size); - bool checksum_needs_read = size_equal && !config->ignore_times && config->checksum; - /* --dry-run must never read the destination file's CONTENTS: a client could - otherwise use `--dry-run --checksum` against a read-only module as a - 1-bit content oracle (hash match / mismatch) and force arbitrary reads. - Decide from metadata alone; when metadata is inconclusive (checksum or - delta would have required the body) report would-transfer. The real - (non-dry-run) behavior below is unchanged. */ - bool need_old_data = !config->dry_run && (checksum_needs_read || try_delta); - if (config->dry_run) - try_delta = false; - *out_try_delta = try_delta; - - if (need_old_data && has_old_file && old_size > 0 && old_size <= MAX_RECEIVE_WHOLE_FILE_SIZE && - old_size <= SIZE_MAX) { - state->old_data = protocol_alloc((size_t)old_size); - if (state->old_data) { - size_t got = 0; - while (got < (size_t)old_size) { - ssize_t n = read(state->old_fd, (char*)state->old_data + got, (size_t)old_size - got); - if (n <= 0) { - free(state->old_data); - state->old_data = NULL; - break; - } - got += (size_t)n; - } - } - } - - bool match = false; - if (config->dry_run) { - /* Metadata-only decision: a size match plus a matching mtime is treated as - up to date; --checksum/--delta cannot be verified without reading, so an - otherwise inconclusive comparison is a would-transfer. */ - match = size_equal && !config->ignore_times && (config->size_only || match_by_metadata); - } else if (checksum_needs_read) { - uint8_t old_digest[CHECKSUM_MAX_DIGEST_LEN]; - size_t old_len = 0; - bool hashed = checksum_digest((ChecksumAlgo)config->checksum_algo, config->checksum_seed, - old_size == 0 ? "" : state->old_data, (size_t)old_size, - old_digest, sizeof(old_digest), &old_len); - match = hashed && old_len == state->check_digest_len && state->check_digest_len > 0 && - memcmp(old_digest, state->check_digest, state->check_digest_len) == 0; - } else if (size_equal && !config->ignore_times) { - match = config->size_only || match_by_metadata; - } - - if (match) { - if (!send_status(fd, STATUS_OK)) - return INCREMENTAL_ERROR; - return INCREMENTAL_SKIP; - } - return INCREMENTAL_CONTINUE; -} - -/* Server-contacting --dry-run no-mutation short-circuit. Runs after the - quick-skip decision and before any path that could touch the destination. - When dry_run is set and the file is not already up to date the receiver must - materialize nothing (no basis link/copy, no append/delta/full transfer) and - the sender must send no data, so answer STATUS_DRY_RUN_TRANSFER and stop. - - The basis lookup is content-blind: under the default metadata quick-check a - hit needs no basis bytes and is honored here just as in a real run; under - --verify-basis a real run hashes the basis against the client-supplied digest, - which in a dry-run is a 1-bit content oracle, so no basis bytes may be read - and an otherwise-matching entry is reported as would-transfer. Everything - read here (the destination file's metadata, basis candidates' metadata) is - read-only. */ -static IncrementalCheckOutcome incremental_check_dry_run_shortcut(IncrementalCheckState* state, - bool* skipped, - bool* would_transfer) { - const Config* config = state->config; - if (!config->dry_run) - return INCREMENTAL_CONTINUE; - - bool skip_via_compare = false; - if (config_has_basis(config) && !config->ignore_times) { - BasisMatch basis; - /* hash_content=false: a dry-run must not read or hash the basis file, so - under --verify-basis no compare-dest hit can be confirmed and an - otherwise-matching file is reported as would-transfer. Without - --verify-basis the metadata quick-check confirms it without touching any - basis bytes. */ - basis_match_find(config, state->check_path, state->check_size, (time_t)state->check_mtime, - (long)state->check_mtime_nsec, state->check_digest, state->check_digest_len, - false, &basis); - if (basis.hit && basis.type == BASIS_DEST_COMPARE && !state->has_old_file) - skip_via_compare = true; - basis_match_free(&basis); - } - Status reply = skip_via_compare ? STATUS_OK : STATUS_DRY_RUN_TRANSFER; - if (!send_status(state->fd, reply)) - return INCREMENTAL_ERROR; - if (skip_via_compare) - *skipped = true; - else if (would_transfer) - *would_transfer = true; - return INCREMENTAL_DRY_RUN; -} - -/* Alternate basis directories (--compare-dest/--copy-dest/--link-dest): a hit - either suppresses the transfer (compare-dest) or materializes the file from - the basis without a data frame. */ -static IncrementalCheckOutcome incremental_check_try_basis(IncrementalCheckState* state, - File** out_file) { - int fd = state->fd; - const Config* config = state->config; - if (!config_has_basis(config)) - return INCREMENTAL_CONTINUE; - - BasisMatch basis; - basis_match_find(config, state->check_path, state->check_size, (time_t)state->check_mtime, - (long)state->check_mtime_nsec, state->check_digest, state->check_digest_len, - true, &basis); - if (basis.hit) { - if (basis.type == BASIS_DEST_COMPARE) { - basis_match_free(&basis); - if (!state->has_old_file) { - if (!send_status(fd, STATUS_OK)) - return INCREMENTAL_ERROR; - return INCREMENTAL_SKIP; - } - } else { - /* Copy/link installs source their bytes from the basis PATH at install - time (bounded buffers), so no whole-file content buffer is needed here - even for an over-limit basis. */ - File* materialized = file_create(state->check_path); - if (materialized) { - data_destroy(materialized->data); - materialized->data = data_create_reserve((size_t)state->check_size); - if (!materialized->data) { - file_destroy(materialized); - materialized = NULL; - } - } - if (materialized) { - materialized->metadata = basis_take_metadata(state, &basis.st); - materialized->skip = true; /* receiver must not ack this as a data file */ - if (basis.type == BASIS_DEST_LINK) - materialized->basis_link = basis.basis_path; - else - materialized->basis_copy = basis.basis_path; - basis.basis_path = NULL; - if (!materialized->metadata) { - file_destroy(materialized); - materialized = NULL; - } - } - if (materialized) { - if (!send_status(fd, STATUS_OK)) { - basis_match_free(&basis); - file_destroy(materialized); - return INCREMENTAL_ERROR; - } - basis_match_free(&basis); - *out_file = materialized; - return INCREMENTAL_FILE; - } - /* Materialization setup failed: fall through to the normal transfer. */ - } - } - basis_match_free(&basis); - return INCREMENTAL_CONTINUE; -} - -/* --append / --append-verify tail resume: when the destination is a SHORTER - file in an append mode, negotiate the resume offset and receive only the - tail. Produces the reconstructed file, or falls through to delta/full. */ -static IncrementalCheckOutcome incremental_check_try_append_resume(IncrementalCheckState* state, - File** out_file) { - int fd = state->fd; - const Config* config = state->config; - const char* check_path = state->check_path; - unsigned long long old_size = state->old_size; - unsigned long long check_size = state->check_size; - - bool append_resume = (config->append || config->append_verify) && state->has_old_file && - append_resume_eligible(old_size, check_size); - if (!append_resume) - return INCREMENTAL_CONTINUE; - - /* Ensure the retained prefix (== the whole, shorter destination file) is in - memory; it is needed both to rebuild the full file and, for - --append-verify, to checksum it. A load failure is not fatal: the resume is - simply not possible and we fall through to the other paths. */ - if (state->old_data == NULL && old_size > 0 && old_size <= MAX_RECEIVE_WHOLE_FILE_SIZE && - old_size <= SIZE_MAX) { - state->old_data = protocol_alloc((size_t)old_size); - if (state->old_data) { - size_t got = 0; - while (got < (size_t)old_size) { - ssize_t n = read(state->old_fd, (char*)state->old_data + got, (size_t)old_size - got); - if (n <= 0) { - free(state->old_data); - state->old_data = NULL; - break; - } - got += (size_t)n; - } - } - } - if (state->old_data == NULL && old_size != 0) - return INCREMENTAL_CONTINUE; - - if (!send_status(fd, STATUS_APPEND) || !send_n_data(fd, &old_size, sizeof(old_size))) - return INCREMENTAL_ERROR; - bool verify = config->append_verify; - bool full_fallback = false; - if (verify) { - Status sig_status; - if (!receive_status(fd, &sig_status)) - return INCREMENTAL_ERROR; - if (sig_status != STATUS_APPEND_SIG) { - send_status(fd, STATUS_ERROR); - return INCREMENTAL_ERROR; - } - uint64_t src_prefix_hash; - if (!receive_n_data(fd, &src_prefix_hash, sizeof(src_prefix_hash))) - return INCREMENTAL_ERROR; - /* Compare the retained prefix against the source prefix. A mismatch must - never be silently appended to: fall back to a full transfer so the result - is a byte-identical source copy. */ - uint64_t dst_prefix_hash = - old_size == 0 ? delta_xxhash64("", 0) : delta_xxhash64(state->old_data, (size_t)old_size); - if (dst_prefix_hash == src_prefix_hash) { - if (!send_status(fd, STATUS_APPEND_OK)) - return INCREMENTAL_ERROR; - } else { - if (!send_status(fd, STATUS_NEXT)) - return INCREMENTAL_ERROR; - full_fallback = true; - } - } - - if (full_fallback) { - /* Retained prefix differed: receive the sender's full transfer. */ - free(state->old_data); - state->old_data = NULL; - if (state->old_fd >= 0) { - close(state->old_fd); - state->old_fd = -1; - } - *out_file = receive_full_file(fd, config, check_path); - return INCREMENTAL_FILE; - } - - /* Receive the tail (STATUS_APPEND_DATA + metadata + tail bytes). */ - Status tail_status; - if (!receive_status(fd, &tail_status)) - return INCREMENTAL_ERROR; - if (tail_status != STATUS_APPEND_DATA) { - send_status(fd, STATUS_ERROR); - return INCREMENTAL_ERROR; - } - FileMetadata* meta = NULL; - FileXattrList* append_xattrs = NULL; - if (config->use_metadata) { - int meta_ok = 1; - meta = metadata_receive(fd, &meta_ok); - if (!meta_ok) - return INCREMENTAL_ERROR; - } - if (config->use_xattrs) { - int xok = 0; - append_xattrs = xattr_receive(fd, &xok, config->preserve_acls); - if (!xok) { - xattr_list_free(append_xattrs); - return INCREMENTAL_ERROR; - } - } - Data* tail = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); - if (tail == NULL) { - xattr_list_free(append_xattrs); - return INCREMENTAL_ERROR; - } - if (config->use_compression && - !compression_should_skip_with_suffixes(check_path, config->skip_compress_suffixes, - config->skip_compress_set ? config->skip_compress_count - : -1)) { - Data* uncompressed = data_decompress_limited(tail, MAX_RECEIVE_WHOLE_FILE_SIZE); - ProtocolSession* owner = tail->owner; - data_destroy(tail); - if (uncompressed == NULL) { - xattr_list_free(append_xattrs); - return INCREMENTAL_ERROR; - } - if (!data_charge_session(uncompressed, owner, uncompressed->size)) { - data_destroy(uncompressed); - xattr_list_free(append_xattrs); - return INCREMENTAL_ERROR; - } - if (uncompressed->size > MAX_FILE_DATA_SIZE) { - data_destroy(uncompressed); - xattr_list_free(append_xattrs); - return INCREMENTAL_ERROR; - } - tail = uncompressed; - } - /* The tail must complete the file exactly; anything else is a protocol - violation (never a truncated or overrun file). */ - unsigned long long expected_tail; - if (!append_tail_length(old_size, check_size, &expected_tail) || - tail->size != (size_t)expected_tail) { - send_status(fd, STATUS_ERROR); - data_destroy(tail); - xattr_list_free(append_xattrs); - return INCREMENTAL_ERROR; - } - size_t full_size = (size_t)check_size; - void* full = protocol_alloc(full_size ? full_size : 1); - if (!full) { - data_destroy(tail); - xattr_list_free(append_xattrs); - return INCREMENTAL_ERROR; - } - if (old_size > 0 && state->old_data) - memcpy(full, state->old_data, (size_t)old_size); - if (tail->size > 0) - memcpy((char*)full + old_size, tail->data, tail->size); - data_destroy(tail); - free(state->old_data); - state->old_data = NULL; - - File* file = file_create(check_path); - if (!file) { - free(full); - xattr_list_free(append_xattrs); - return INCREMENTAL_ERROR; - } - file->metadata = meta; - file->xattrs = append_xattrs; - append_xattrs = NULL; - data_destroy(file->data); - file->data = data_create(full, full_size); - if (!file->data) { /* data_create already freed full on failure */ - file_destroy(file); - return INCREMENTAL_ERROR; - } - *out_file = file; - return INCREMENTAL_FILE; -} - -/* Block delta transfer against the existing destination content. */ -static IncrementalCheckOutcome incremental_check_try_delta(IncrementalCheckState* state, - bool try_delta, File** out_file) { - if (try_delta && state->old_data != NULL) { - bool delta_failed = false; - File* delta_file = receive_delta_file(state->fd, state->config, state->check_path, - state->old_data, state->old_size, &delta_failed); - state->old_data = NULL; /* receive_delta_file consumes the snapshot on every path */ - if (delta_file) { - *out_file = delta_file; - return INCREMENTAL_FILE; - } - if (delta_failed) - return INCREMENTAL_ERROR; - } - free(state->old_data); - state->old_data = NULL; - return INCREMENTAL_CONTINUE; -} - -/* -y/--fuzzy similar-file delta basis. Reaching this point means the file - must be transferred and the destination's own content could not serve as a - delta basis; try an existing similar-named sibling in the same directory. */ -static IncrementalCheckOutcome incremental_check_try_fuzzy(IncrementalCheckState* state, - File** out_file) { - const Config* config = state->config; - if (!config->fuzzy || !config->use_delta) - return INCREMENTAL_CONTINUE; - unsigned long long fuzzy_size = 0; - void* fuzzy_basis = fuzzy_basis_find_and_load(config, state->check_path, state->check_size, - (time_t)state->check_mtime, - (long)state->check_mtime_nsec, &fuzzy_size); - if (fuzzy_basis != NULL) { - bool fuzzy_failed = false; - File* fuzzy_file = receive_delta_file(state->fd, config, state->check_path, fuzzy_basis, - fuzzy_size, &fuzzy_failed); - fuzzy_basis = NULL; /* receive_delta_file consumes the buffer on every path */ - if (fuzzy_file) { - *out_file = fuzzy_file; - return INCREMENTAL_FILE; - } - if (fuzzy_failed) - return INCREMENTAL_ERROR; - } - free(fuzzy_basis); - return INCREMENTAL_CONTINUE; -} - -/* Final fallback: tell the sender to transmit the whole file and receive it. */ -static File* incremental_check_receive_full(IncrementalCheckState* state) { - if (!send_status(state->fd, STATUS_NEXT)) - return NULL; - if (state->old_fd >= 0) { - close(state->old_fd); - state->old_fd = -1; - } - return receive_full_file(state->fd, state->config, state->check_path); -} - -/* Core implementation. `would_transfer` (may be NULL) is set true only on the - * server-contacting --dry-run path, when the file is not up to date and the - * receiver answered STATUS_DRY_RUN_TRANSFER; the caller then knows no File is - * returned and nothing was stored. */ -File* receive_incremental_check_ex(int fd, const Config* config, bool* skipped, - bool* would_transfer) { - if (would_transfer) - *would_transfer = false; - if (!config || !skipped) { - send_status(fd, STATUS_ERROR); - return NULL; - } - *skipped = false; - - IncrementalCheckState state; - incremental_check_state_init(&state, fd, config); - - File* result = NULL; - bool try_delta = false; - IncrementalCheckOutcome outcome; - - outcome = incremental_check_receive_request(&state); - if (outcome == INCREMENTAL_ERROR) - goto done; - - outcome = incremental_check_open_destination(&state); - if (outcome == INCREMENTAL_ERROR) - goto done; - - outcome = incremental_check_report_dest_info(&state); - if (outcome == INCREMENTAL_ERROR) - goto done; - - /* --ignore-existing must answer before any data is requested; it takes - precedence over the metadata up-to-date check below. */ - outcome = incremental_check_ignore_existing(&state); - if (outcome == INCREMENTAL_ERROR) - goto done; - if (outcome == INCREMENTAL_SKIP) { - *skipped = true; - goto done; - } - - /* A --link-dest hit relinks even an already up-to-date destination before the - quick-skip can suppress it (rsync parity). */ - outcome = incremental_check_link_dest_relink(&state, &result); - if (outcome == INCREMENTAL_ERROR) - goto done; - if (outcome == INCREMENTAL_FILE) - goto done; - - outcome = incremental_check_quick_skip(&state, &try_delta); - if (outcome == INCREMENTAL_ERROR) - goto done; - if (outcome == INCREMENTAL_SKIP) { - *skipped = true; - goto done; - } - - /* Dry-run resolves here (no mutation) or falls through to the normal path. */ - outcome = incremental_check_dry_run_shortcut(&state, skipped, would_transfer); - if (outcome == INCREMENTAL_ERROR) - goto done; - if (outcome != INCREMENTAL_CONTINUE) - goto done; - - outcome = incremental_check_try_basis(&state, &result); - if (outcome == INCREMENTAL_ERROR) - goto done; - if (outcome == INCREMENTAL_SKIP) { - *skipped = true; - goto done; - } - if (outcome == INCREMENTAL_FILE) - goto done; - - outcome = incremental_check_try_append_resume(&state, &result); - if (outcome == INCREMENTAL_ERROR) - goto done; - if (outcome == INCREMENTAL_FILE) - goto done; - - outcome = incremental_check_try_delta(&state, try_delta, &result); - if (outcome == INCREMENTAL_ERROR) - goto done; - if (outcome == INCREMENTAL_FILE) - goto done; - - outcome = incremental_check_try_fuzzy(&state, &result); - if (outcome == INCREMENTAL_ERROR) - goto done; - if (outcome == INCREMENTAL_FILE) - goto done; - - result = incremental_check_receive_full(&state); - -done: - incremental_check_state_cleanup(&state); - return result; -} - -File* receive_incremental_check(int fd, const Config* config, bool* skipped) { - return receive_incremental_check_ex(fd, config, skipped, NULL); -} - File* file_receive(const Config* config, int file_descriptor) { char* path = receive_wire_str(file_descriptor); if (path == NULL) @@ -3235,601 +550,3 @@ File* file_receive_special(int file_descriptor) { file->rdev_minor = minor; return file; } - -/* Read a delete-manifest frame (the STATUS_MANIFEST leading code has already - been consumed): a keep-set entry count followed by that many - destination-relative paths, then a protected-prefix count followed by that - many destination-relative prefixes, then a missing-args count followed by that - many destination-relative delete paths, then (protocol 2.23.0) a - synchronized-directory count followed by that many destination-relative - directory paths (the receive root is the "." sentinel). The frame is - self-delimiting (the counts are authoritative), so the caller decides what to - do next and continues reading the following STATUS_* frame. Every section is - validated identically: an entry must be non-empty, relative and traversal-free - and the aggregate length across ALL sections is capped by MAX_MANIFEST_BYTES - (so the missing-args deletion requests are confined like the rest of the - manifest). Returns an owned DeleteManifest, or NULL after sending STATUS_ERROR - when the frame is malformed (bad count, empty/absolute path, path traversal, - or an aggregate size beyond MAX_MANIFEST_BYTES). */ -static bool receive_manifest_section(int fd, ArrayList* list, size_t* manifest_bytes, - size_t* manifest_entries) { - int count; - if (!receive_int(fd, &count)) { - send_status(fd, STATUS_ERROR); - return false; - } - if (count < 0 || count > MAX_MANIFEST_ENTRIES || - (size_t)count > MAX_MANIFEST_ENTRIES - *manifest_entries) { - send_status(fd, STATUS_ERROR); - return false; - } - for (int i = 0; i < count; i++) { - char* s = receive_wire_str(fd); - size_t entry_size = s ? strlen(s) + MANIFEST_ENTRY_OVERHEAD : 0; - if (!s || s[0] == '\0' || s[0] == '/' || has_path_traversal(s) || - entry_size > MAX_MANIFEST_BYTES - *manifest_bytes || - (*manifest_bytes += entry_size) > MAX_MANIFEST_BYTES || !array_list_add(list, s)) { - free(s); - send_status(fd, STATUS_ERROR); - return false; - } - } - *manifest_entries += (size_t)count; - return true; -} - -DeleteManifest* receive_manifest_entries(int fd) { - DeleteManifest* manifest = calloc(1, sizeof(DeleteManifest)); - if (!manifest) { - send_status(fd, STATUS_ERROR); - return NULL; - } - manifest->keeps = array_list_create(free); - manifest->protected = array_list_create(free); - manifest->missing = array_list_create(free); - manifest->dirs = array_list_create(free); - if (!manifest->keeps || !manifest->protected || !manifest->missing || !manifest->dirs) { - delete_manifest_free(manifest); - send_status(fd, STATUS_ERROR); - return NULL; - } - size_t manifest_bytes = 0; - size_t manifest_entries = 0; - if (!receive_manifest_section(fd, manifest->keeps, &manifest_bytes, &manifest_entries) || - !receive_manifest_section(fd, manifest->protected, &manifest_bytes, &manifest_entries) || - !receive_manifest_section(fd, manifest->missing, &manifest_bytes, &manifest_entries) || - !receive_manifest_section(fd, manifest->dirs, &manifest_bytes, &manifest_entries)) { - delete_manifest_free(manifest); - return NULL; - } - return manifest; -} - -void delete_manifest_free(DeleteManifest* manifest) { - if (!manifest) - return; - array_list_delete(manifest->keeps); - array_list_delete(manifest->protected); - array_list_delete(manifest->missing); - array_list_delete(manifest->dirs); - free(manifest); -} - -/* Shared --max-delete budget for one receiver-side deletion commit. Both the - --delete-missing-args exact-path removals and the ordinary extras walk draw - from the same tally, matching rsync (whose --max-delete counts every deleted - file or directory). `max_delete` is SIZE_MAX for an unlimited budget. */ -typedef struct { - size_t max_delete; - size_t deleted; - size_t skipped; - bool limit_hit; -} DeleteBudgetState; - -/* Build the delete-walk protection prefix for one basis directory. The walker - compares paths relative to the receive root, so a relative entry is already - in the right form; an absolute entry that lies below the root is converted to - its root-relative form, and one outside the root returns NULL (the walk - cannot reach it, and it is not protected data beneath the root). Exposed so - tests can exercise the root-of-"/" child mapping directly. */ -char* file_receive_basis_delete_relative(const Config* config, const char* path) { - if (!path) - return NULL; - if (path[0] != '/') - return str_dup(path); - const char* root = config->receive_root_directory; - if (!root || root[0] != '/') - return NULL; - size_t root_len = strlen(root); - while (root_len > 1 && root[root_len - 1] == '/') - root_len--; - if (strncmp(path, root, root_len) != 0) - return NULL; - if (root_len == 1) { - /* `root` is "/" (the only single-character absolute root): every absolute - path is below it, and the child relative form is everything after the - leading '/'. */ - if (path[1] == '\0') - return NULL; /* identical to the root, not a child */ - return str_dup(path + 1); - } - if (path[root_len] != '/') - return NULL; /* identical or a sibling sharing a name prefix */ - return str_dup(path + root_len + 1); -} - -/* Remove every destination entry under the receive root that is not in the - keep-set, bounded by the shared budget (a smaller client --max-delete=NUM - replaces the server hard bound; rsync deletes up to the bound and skips the - rest). With --delay-updates the not-yet-published staging directory is a - direct child of the receive root and must not be treated as a set of extras; - the manifest's protected prefixes (paths excluded on the source), the - size-pruned prefixes (--max-size/--min-size, always protected) and the - alternate basis directories are never destination content and are skipped at - any depth. Returns true unless a traversal/unlink error aborted the walk; - the budget's limit_hit/skipped fields report a cap-stopped run. */ -static bool delete_extras_budgeted_observed(const Config* config, DeleteManifest* manifest, - DeleteBudgetState* budget, DeletePathObserver observer, - void* observer_context) { - if (!config || !manifest || !manifest->keeps) - return false; - fprintf(stderr, "Deleting files not in manifest...\n"); - /* Protected entries: - - the --delay-updates staging name, protected only as a DIRECT child of the - receive root (a nested destination directory that happens to be named - .fastsync-stage is ordinary content); - - alternate basis directories (--compare-dest / --copy-dest / --link-dest) - at any depth: they are extra comparison snapshots the user pointed at, - not destination content, and deleting them would destroy the very files a - --link-dest run just linked into place; - - the sender-side protected prefixes (source paths excluded by filters and - paths pruned by --max-size/--min-size), at any depth, so their destination - mirror survives --delete unless --delete-excluded opts back into removing - the filter-excluded ones (size-pruned entries are always protected). */ - int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count + - (manifest->protected ? manifest->protected->size : 0); - DeleteSkipEntry* skips = NULL; - char** owned_prefixes = NULL; - int used = 0; - if (skip_count > 0) { - skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); - owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); - if (!skips || (config->basis_count > 0 && !owned_prefixes)) { - free(skips); - free(owned_prefixes); - return false; - } - int idx = 0; - if (config->delay_updates) { - skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; - skips[idx].top_level_only = true; - idx++; - } - for (int i = 0; i < config->basis_count; i++) { - /* An absolute basis outside the receive root is unreachable by this walk, - so it contributes no protection prefix (and no slot). */ - char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); - if (!prefix) - continue; - owned_prefixes[i] = prefix; - skips[idx].prefix = prefix; - skips[idx].top_level_only = false; - idx++; - } - for (int i = 0; i < manifest->protected->size; i++) { - skips[idx].prefix = (const char*)manifest->protected->items[i]; - skips[idx].top_level_only = false; - idx++; - } - used = idx; - } - /* Clamp rather than subtract: an accounting bug where deleted already exceeds - max_delete must never underflow into an effectively unlimited budget. */ - size_t remaining; - if (budget->max_delete == SIZE_MAX) - remaining = SIZE_MAX; - else if (budget->deleted >= budget->max_delete) - remaining = 0; - else - remaining = budget->max_delete - budget->deleted; - size_t deleted = 0; - size_t skipped = 0; - DeleteWalkResult result = delete_extras_limited_observed( - config->receive_root_directory, manifest->keeps, manifest->dirs, remaining, skips, used, - config->protect_rules, &deleted, &skipped, observer, observer_context); - if (owned_prefixes) { - for (int i = 0; i < config->basis_count; i++) - free(owned_prefixes[i]); - } - free(owned_prefixes); - free(skips); - budget->deleted += deleted; - budget->skipped += skipped; - if (result == DELETE_WALK_LIMIT_REACHED) { - budget->limit_hit = true; - return true; - } - if (result != DELETE_WALK_OK) { - log_message(LOG_LEVEL_ERROR, "deletion failed while removing extraneous files"); - return false; - } - return true; -} - -static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifest, - DeleteBudgetState* budget) { - return delete_extras_budgeted_observed(config, manifest, budget, NULL, NULL); -} - -/* Prefixes every observed path with a fixed subtree root, so a nested walk - (a recursively removed missing-arg directory) reports receive-root-relative - names like the rest of the delete output. */ -typedef struct { - DeletePathObserver inner; - void* inner_context; - const char* prefix; -} PrefixedDeleteObserver; - -static void prefixed_delete_observer(void* context, const char* rel) { - PrefixedDeleteObserver* prefixed = context; - if (!prefixed->inner || !rel) - return; - char* joined = path_cat((char*)prefixed->prefix, rel); - if (joined) { - prefixed->inner(prefixed->inner_context, joined); - free(joined); - } -} - -/* --delete-missing-args exact-path deletions: each destination mirror in - manifest->missing is an explicit user request, so it is removed even when the - ordinary extras walk (with its protected prefixes) would leave it alone. The - --delay-updates staging directory and basis snapshots are receiver artifacts - and stay protected exactly as in the extras walker. A regular file or - symlink is unlinked, an empty directory removed, and a NON-empty directory is - removed recursively only when --delete or --force is in effect (rsync parity: - the man page says a non-empty directory mirror is only deleted with --force - or --delete); otherwise it is left with a warning and the run continues. A - mirror that does not exist is a no-op. Each removal draws from the shared - --max-delete budget: once it is exhausted the remaining requests are skipped - and counted. Returns false only on a genuine error (a confinement failure on - a validated path or an I/O error), which fails the run. */ -static bool delete_missing_args_budgeted_observed(const Config* config, DeleteManifest* manifest, - DeleteBudgetState* budget, - DeletePathObserver observer, - void* observer_context) { - if (!config || !manifest) - return false; - if (!manifest->missing || manifest->missing->size == 0) - return true; - fprintf(stderr, "Deleting destination mirrors of missing source arguments...\n"); - int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count; - DeleteSkipEntry* skips = NULL; - char** owned_prefixes = NULL; - int used = 0; - if (skip_count > 0) { - skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); - owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); - if (!skips || (config->basis_count > 0 && !owned_prefixes)) { - free(skips); - free(owned_prefixes); - return false; - } - int idx = 0; - if (config->delay_updates) { - skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; - skips[idx].top_level_only = true; - idx++; - } - for (int i = 0; i < config->basis_count; i++) { - char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); - if (!prefix) - continue; - owned_prefixes[i] = prefix; - skips[idx].prefix = prefix; - skips[idx].top_level_only = false; - idx++; - } - used = idx; - } - bool ok = true; - for (int i = 0; i < manifest->missing->size; i++) { - const char* rel = (const char*)manifest->missing->items[i]; - if (!rel || *rel == '\0' || *rel == '/' || has_path_traversal(rel)) { - /* Defensive only: receive_manifest_entries already validated every - section identically, so a controlled peer never reaches this branch. */ - log_message(LOG_LEVEL_ERROR, "invalid missing-args delete path"); - ok = false; - continue; - } - bool at_root = strchr(rel, '/') == NULL; - if (path_under_skip_prefix(rel, at_root, skips, used)) { - char* escaped = output_escape(rel, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, - "missing-args path '%s' is protected (staging directory or basis snapshot); " - "not deleting", - escaped ? escaped : ""); - free(escaped); - continue; - } - char* full = path_cat(config->receive_root_directory, rel); - if (!full) { - ok = false; - continue; - } - char* leaf = NULL; - int parent_fd = file_open_secure_parent(full, &leaf, false); - if (parent_fd < 0) { - /* The mirror's parent directory may itself not exist on the destination - (a deeper missing entry whose leading directories were never created). - That is a no-op -- there is nothing to delete -- matching - file_remove_tree_secure's absent-path handling; only a genuine I/O - error (EACCES, a symlink loop, ...) fails the run. */ - bool absent = errno == ENOENT || errno == ENOTDIR; - free(full); - free(leaf); - if (!absent) - ok = false; - continue; - } - struct stat st; - if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) != 0) { - /* Already absent: nothing to delete (a no-op, not a deletion). */ - if (errno != ENOENT) - ok = false; - close(parent_fd); - free(leaf); - free(full); - continue; - } - /* An entry that exists is one deletion: skip it (and count it) when the - shared --max-delete budget is already exhausted. */ - if (budget->deleted >= budget->max_delete) { - budget->limit_hit = true; - budget->skipped++; - close(parent_fd); - free(leaf); - free(full); - continue; - } - bool removed = false; - if (S_ISDIR(st.st_mode)) { - if (unlinkat(parent_fd, leaf, AT_REMOVEDIR) == 0) { - removed = true; - } else if (errno == ENOTEMPTY || errno == EEXIST) { - close(parent_fd); - parent_fd = -1; - free(leaf); - leaf = NULL; - if (config->use_delete || config->force_delete) { - /* Remove the contents entry-by-entry through the budgeted extras - walker so every deleted file/dir counts toward --max-delete (rsync - parity); the now-empty directory itself costs one more. A run that - hits the cap leaves the remaining entries in place. */ - ArrayList* no_keeps = array_list_create(free); - /* Never let an accounting slip (deleted > max_delete) underflow the - remaining budget into SIZE_MAX, which would grant unlimited - deletions. */ - size_t remaining = - budget->deleted >= budget->max_delete ? 0 : budget->max_delete - budget->deleted; - size_t contents_deleted = 0; - size_t contents_skipped = 0; - PrefixedDeleteObserver nested = {observer, observer_context, rel}; - DeleteWalkResult walk = - no_keeps ? delete_extras_limited_observed(full, no_keeps, NULL, remaining, NULL, 0, - NULL, &contents_deleted, &contents_skipped, - observer ? prefixed_delete_observer : NULL, - observer ? &nested : NULL) - : DELETE_WALK_ERROR; - if (no_keeps) - array_list_delete(no_keeps); - budget->deleted += contents_deleted; - budget->skipped += contents_skipped; - if (walk == DELETE_WALK_LIMIT_REACHED) { - budget->limit_hit = true; - } else if (walk != DELETE_WALK_OK) { - ok = false; - } else if (budget->deleted >= budget->max_delete) { - budget->limit_hit = true; - budget->skipped++; - } else if (file_remove_tree_secure(full)) { - /* The shared `if (removed)` tail charges this directory exactly - once; counting it here too would consume two budget units. */ - removed = true; - } else { - ok = false; - } - } else { - char* escaped = output_escape(rel, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, - "missing-args destination '%s' is a non-empty directory; use --force or " - "--delete to remove it", - escaped ? escaped : ""); - free(escaped); - } - } else if (errno != ENOENT) { - ok = false; - } - } else { - if (unlinkat(parent_fd, leaf, 0) == 0) { - removed = true; - } else if (errno != ENOENT) { - ok = false; - } - } - if (removed) { - budget->deleted++; - if (observer) - observer(observer_context, rel); - char* escaped = output_escape(rel, log_get_8_bit_output()); - fprintf(stderr, " Deleted: %s\n", escaped ? escaped : ""); - free(escaped); - } - if (parent_fd >= 0) - close(parent_fd); - free(leaf); - free(full); - if (!ok) - break; - } - if (owned_prefixes) { - for (int i = 0; i < config->basis_count; i++) - free(owned_prefixes[i]); - } - free(owned_prefixes); - free(skips); - return ok; -} - -/* Public wrappers used outside the commit path (and by unit tests): no - --max-delete budget. */ -bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, ArrayList* out, - size_t* count_out) { - if (count_out) - *count_out = 0; - if (!config || !manifest || !manifest->keeps || !out) - return false; - int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count + - (manifest->protected ? manifest->protected->size : 0); - DeleteSkipEntry* skips = NULL; - char** owned_prefixes = NULL; - int used = 0; - if (skip_count > 0) { - skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); - owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); - if (!skips || (config->basis_count > 0 && !owned_prefixes)) { - free(skips); - free(owned_prefixes); - return false; - } - int idx = 0; - if (config->delay_updates) { - skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; - skips[idx].top_level_only = true; - idx++; - } - for (int i = 0; i < config->basis_count; i++) { - /* Normalize exactly like the real commit path: a relative entry is - already root-relative, an absolute one inside the receive root is - converted, and one outside contributes no protection prefix. */ - char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); - if (!prefix) - continue; - owned_prefixes[i] = prefix; - skips[idx].prefix = prefix; - skips[idx].top_level_only = false; - idx++; - } - for (int i = 0; i < manifest->protected->size; i++) { - skips[idx].prefix = (const char*)manifest->protected->items[i]; - skips[idx].top_level_only = false; - idx++; - } - used = idx; - } - bool ok = delete_extras_list(config->receive_root_directory, manifest->keeps, manifest->dirs, - skips, used, config->protect_rules, out, count_out); - if (owned_prefixes) { - for (int i = 0; i < config->basis_count; i++) - free(owned_prefixes[i]); - } - free(owned_prefixes); - free(skips); - return ok; -} - -bool manifest_delete_extras(const Config* config, DeleteManifest* manifest) { - DeleteBudgetState budget = { - .max_delete = SIZE_MAX, .deleted = 0, .skipped = 0, .limit_hit = false}; - return delete_extras_budgeted(config, manifest, &budget); -} - -bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest) { - DeleteBudgetState budget = { - .max_delete = SIZE_MAX, .deleted = 0, .skipped = 0, .limit_hit = false}; - return delete_missing_args_budgeted_observed(config, manifest, &budget, NULL, NULL); -} - -bool manifest_delete_missing_args_limited(const Config* config, DeleteManifest* manifest, - size_t max_delete, size_t* deleted, size_t* skipped, - bool* limit_hit) { - return manifest_delete_missing_args_limited_observed(config, manifest, max_delete, deleted, - skipped, limit_hit, NULL, NULL); -} - -bool manifest_delete_missing_args_limited_observed(const Config* config, DeleteManifest* manifest, - size_t max_delete, size_t* deleted, - size_t* skipped, bool* limit_hit, - DeletePathObserver observer, - void* observer_context) { - DeleteBudgetState budget = { - .max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false}; - bool ok = - delete_missing_args_budgeted_observed(config, manifest, &budget, observer, observer_context); - if (deleted) - *deleted = budget.deleted; - if (skipped) - *skipped = budget.skipped; - if (limit_hit) - *limit_hit = budget.limit_hit; - return ok; -} - -/* Commit every deletion family the manifest carries. The --delete-missing-args - exact-path deletions run FIRST: they are explicit user requests and must not - be blocked by the extras walker's filter-exclusion protection (a protected - leftover inside a missing-argument directory must not make that user-requested - removal fail). The ordinary extras walk then runs when --delete is active. - Both draw from one --max-delete budget; the result reports a cap-stopped - (partial) commit distinctly so the client can exit 25 like rsync. */ -DeleteCommitResult manifest_delete_all(const Config* config, DeleteManifest* manifest) { - return manifest_delete_all_counted(config, manifest, NULL); -} - -DeleteCommitResult manifest_delete_all_counted(const Config* config, DeleteManifest* manifest, - size_t* deleted) { - return manifest_delete_all_observed(config, manifest, deleted, NULL, NULL); -} - -DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteManifest* manifest, - size_t* deleted, DeletePathObserver observer, - void* observer_context) { - if (deleted) - *deleted = 0; - if (!config || !manifest) - return DELETE_COMMIT_ERROR; - /* Central no-mutation guard: a dry-run never deletes. No manifest is sent on - the dry-run path, but a hostile/buggy peer could; treat it as a no-op so - the receiver can never remove anything. */ - if (config->dry_run) - return DELETE_COMMIT_OK; - /* A client --max-delete=NUM smaller than the server's hard bound replaces it - for this run; both still bound the commit. */ - bool user_limited = - config->max_delete >= 0 && (size_t)config->max_delete < MAX_SERVER_DELETE_COUNT; - DeleteBudgetState budget = {.max_delete = user_limited ? (size_t)config->max_delete - : MAX_SERVER_DELETE_COUNT, - .deleted = 0, - .skipped = 0, - .limit_hit = false}; - if (config->delete_missing_args && - !delete_missing_args_budgeted_observed(config, manifest, &budget, observer, observer_context)) - return DELETE_COMMIT_ERROR; - if (config->use_delete && - !delete_extras_budgeted_observed(config, manifest, &budget, observer, observer_context)) - return DELETE_COMMIT_ERROR; - if (deleted) - *deleted = budget.deleted; - if (budget.limit_hit) { - if (user_limited) { - log_message(LOG_LEVEL_ERROR, "Deletions stopped due to --max-delete limit (%zu skipped)", - budget.skipped); - } else { - log_message(LOG_LEVEL_ERROR, - "Deletions stopped due to the server deletion limit of %u (%zu skipped)", - (unsigned)MAX_SERVER_DELETE_COUNT, budget.skipped); - } - return DELETE_COMMIT_LIMIT_REACHED; - } - return DELETE_COMMIT_OK; -} diff --git a/src/shared/file_receive.h b/src/shared/file_receive.h index e316000..c55fc4c 100644 --- a/src/shared/file_receive.h +++ b/src/shared/file_receive.h @@ -2,11 +2,19 @@ #define FILE_RECEIVE_H #include "config.h" +#include "delete_commit.h" +#include "file_save.h" #include "file_types.h" +#include "incremental_check.h" #include "utils.h" #include -/* Server-side file receive/save path. */ +/* Server-side file receive/save path. + * + * This header is the public facade for the file_receive module family: the + * wire receive dispatch (this file) plus the save-to-disk (file_save.h), the + * incremental check (incremental_check.h) and the delete-commit + * (delete_commit.h) modules. */ /* Cumulative caps for the deferred directory-time accumulator. The sender may * legitimately split a large tree across repeated STATUS_DIR_TIMES frames, so a @@ -23,25 +31,6 @@ File* file_receive_dir_time(int file_descriptor, const Config* config); File* file_receive_hardlink(int file_descriptor); File* file_receive_symlink(int file_descriptor, const Config* config); File* file_receive_special(int file_descriptor); -bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode); -/* Testable basis quick-check / verification policy. file_basis_quick_match is - * rsync's metadata quick-check for a basis candidate (equal size is required - * separately by the caller; this adds the --size-only / mtime / --modify-window - * leg). file_basis_content_required reports whether a hit must ALSO be - * confirmed by a whole-file content digest (--verify-basis; false is the - * default rsync-parity behavior). */ -bool file_basis_quick_match(const Config* config, const struct stat* st, time_t check_mtime, - long check_mtime_nsec); -bool file_basis_content_required(const Config* config); - -File* receive_incremental_check(int fd, const Config* config, bool* skipped); -/* Extended variant used by the receiver. `would_transfer` (may be NULL) is set - * true only on the server-contacting --dry-run path when the file is not up to - * date: the receiver has already sent STATUS_DRY_RUN_TRANSFER and returns NULL - * without storing anything. On that path `*skipped` is true for an up-to-date - * (STATUS_OK) file and both flags are false for a genuine error. */ -File* receive_incremental_check_ex(int fd, const Config* config, bool* skipped, - bool* would_transfer); /* P7 Wave D directory-time accumulator. The receiver collects the metadata of * every directory it creates/receives (STATUS_MKDIR with metadata and/or the @@ -85,126 +74,4 @@ bool dir_time_list_add(DirTimeList* list, const char* wire_path, const FileMetad void dir_metadata_list_apply(const DirTimeList* list, const char* root_directory, const Config* config); -/* A received delete-manifest frame: the keep-set (`keeps`, destination-relative - paths the sender transferred/keeps) plus `protected`, destination-relative - prefixes the sender asks the receiver never to delete (paths excluded on the - source, protected at any depth). When --delete-excluded is given the sender - transmits an empty protected list so excluded destination mirrors are treated - as ordinary extras. With --delete-missing-args a third section (`missing`) - carries the destination mirrors of explicitly-listed source entries that do - not exist: each is an exact deletion request, independent of the ordinary - extras walk (never blocked by the protected prefixes) and processed when the - manifest is committed. */ -typedef struct DeleteManifest { - ArrayList* keeps; - ArrayList* protected; - ArrayList* missing; - /* Destination-relative paths of the directories the sender synchronized for - this run. The extras walker only removes entries directly inside one of - these (the receive root is the "." sentinel); `--files-from` runs therefore - leave untransmitted directories and the unlisted parts of listed ones - alone, matching rsync's "delete only in synchronized directories". */ - ArrayList* dirs; -} DeleteManifest; - -void delete_manifest_free(DeleteManifest* manifest); -/* Read a delete-manifest frame (protocol 2.23.0): keep count + keeps, then - protected count + protected prefixes, then missing count + missing paths, - then synchronized-directory count + directory paths (self-delimiting; the - leading STATUS_MANIFEST code has been consumed). Returns an owned - DeleteManifest, or NULL after signalling STATUS_ERROR on a malformed frame. */ -DeleteManifest* receive_manifest_entries(int fd); -/* Remove destination entries under config->receive_root_directory that are not - in `manifest` (bounded, all-or-nothing walk; staging-dir, basis-dir and - protected-prefix skips). `--max-delete` and `--force` are honored here. The - caller decides WHEN to run it based on the negotiated delete timing. Returns - false (and the transfer fails) when the deletion cannot be committed. */ -bool manifest_delete_extras(const Config* config, DeleteManifest* manifest); -/* --delete-missing-args exact-path deletions: remove each destination mirror - in `manifest->missing` (never blocked by the protected prefixes, staging dir - and basis dirs excluded). A regular file/symlink is unlinked; an empty - directory is removed; a NON-empty directory is removed recursively only when - --delete or --force is in effect, otherwise it is left with a warning (rsync - parity). A missing path is a no-op. Returns false only on a genuine - confinement or I/O error (the run then fails); tolerated per-path cases are - reported and skipped. */ -bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest); -/* Budgeted form of manifest_delete_missing_args for the per-directory delete - session: each removed mirror draws from `max_delete` (SIZE_MAX = unlimited) - and the tallies are accumulated into `*deleted`/`*skipped`. `*limit_hit` is set - when the budget stopped the pass with entries left over. Returns false only - on a genuine deletion error. */ -bool manifest_delete_missing_args_limited(const Config* config, DeleteManifest* manifest, - size_t max_delete, size_t* deleted, size_t* skipped, - bool* limit_hit); -/* Observer-aware form of manifest_delete_missing_args_limited: `observer` (may - be NULL) is invoked for every destination-relative path truly removed. */ -bool manifest_delete_missing_args_limited_observed(const Config* config, DeleteManifest* manifest, - size_t max_delete, size_t* deleted, - size_t* skipped, bool* limit_hit, - DeletePathObserver observer, - void* observer_context); -/* Outcome of committing a delete manifest. LIMIT_REACHED reports rsync's - partial --max-delete result: the budget allowed some deletions and the rest - were skipped (the run still stores all file data but the client exits 25). */ -typedef enum { - DELETE_COMMIT_OK = 0, - DELETE_COMMIT_LIMIT_REACHED, - DELETE_COMMIT_ERROR -} DeleteCommitResult; - -/* Run every deletion family the manifest carries: the --delete-missing-args - exact-path deletions first (user requests are not blocked by exclusion - protection), then the ordinary extras walk when --delete is active. Both - share one --max-delete budget. Returns DELETE_COMMIT_OK when nothing was to - do or everything committed, DELETE_COMMIT_LIMIT_REACHED when the budget - stopped part of the work, or DELETE_COMMIT_ERROR on a genuine failure. */ -DeleteCommitResult manifest_delete_all(const Config* config, DeleteManifest* manifest); -/* Like manifest_delete_all, but reports how many destination entries the commit - removed (for the end-of-transfer wire stats). `deleted` may be NULL. */ -DeleteCommitResult manifest_delete_all_counted(const Config* config, DeleteManifest* manifest, - size_t* deleted); -/* Observer-aware form of manifest_delete_all_counted: `observer` (may be NULL) - is invoked for every destination-relative path truly removed. */ -DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteManifest* manifest, - size_t* deleted, DeletePathObserver observer, - void* observer_context); - -/* -n/--dry-run --delete would-delete reporting: walk the destination exactly as - the delete pass would and append (strdup'd) destination-relative paths that - WOULD be removed to `out`, without touching disk. Uses the same staging-dir, - basis-dir and protected-prefix skips as the real commit. Returns true on a - clean walk; `*count_out` receives the number of paths appended. */ -bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, ArrayList* out, - size_t* count_out); -/* Convert one basis-directory path to the receive-root-relative protection - prefix the delete walker uses (NULL when it lies outside the root). Exposed - for unit tests of the root-of-"/" and normalization edge cases. */ -char* file_receive_basis_delete_relative(const Config* config, const char* path); - -/* Outcome of a single file_save_to_disk operation. The receiver needs to - distinguish "written" from "skipped" so --remove-source-files can be told - which sources were actually stored. */ -typedef enum { FILE_SAVE_ERROR = 0, FILE_SAVE_WRITTEN = 1, FILE_SAVE_SKIPPED = 2 } FileSaveResult; - -FileSaveResult file_save_to_disk_full(const char* root_directory, const File* file, - const Config* config); -/* Protocol 2.28.0 variant: also reports through `created` (when non-NULL) - * whether the destination entry did not exist before this save, and through - * `created_dirs` how many parent directories the confined walk created, so the - * receiver can build rsync's `Number of created files` breakdown. The plain - * file_save_to_disk_full() is this with both out-params NULL. */ -FileSaveResult file_save_to_disk_full_ex(const char* root_directory, const File* file, - const Config* config, bool* created, - unsigned* created_dirs); -bool file_save_to_disk(const char* root_directory, const File* file, const Config* config); - -/* Protocol 2.28.0 receiver counter accumulator: fold one successfully saved - * entry into `stats`, adding its receiver-observed literal bytes and, when - * `created`, the matching created-by-type counter (regular file / symlink / - * special) plus `created_dirs` implicitly-created parent directories. - * Non-first hardlink siblings contribute no literal bytes. */ -void receiver_stats_note_saved(ReceiverStats* stats, const File* file, bool created, - unsigned created_dirs); - #endif diff --git a/src/shared/file_save.c b/src/shared/file_save.c new file mode 100644 index 0000000..98000ae --- /dev/null +++ b/src/shared/file_save.c @@ -0,0 +1,1164 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "array_list.h" +#include "charset.h" +#include "chmod.h" +#include "chunk.h" +#include "compression.h" +#include "config.h" +#include "data.h" +#include "delay_updates.h" +#include "delta.h" +#include "file.h" +#include "file_save.h" +#include "format.h" +#include "identity.h" +#include "log.h" +#include "metadata.h" +#include "protocol.h" +#include "utils.h" +#include "xattr.h" + +bool file_save_to_disk(const char* root_directory, const File* file, const Config* config) { + return file_save_to_disk_full_ex(root_directory, file, config, NULL, NULL) != FILE_SAVE_ERROR; +} + +FileSaveResult file_save_to_disk_full(const char* root_directory, const File* file, + const Config* config) { + return file_save_to_disk_full_ex(root_directory, file, config, NULL, NULL); +} + +/* --delay-updates receiver path: write the file into a private staging tree + below the receive root instead of its final destination, and remember it so + it can be atomically renamed into place only once the whole transfer has + succeeded. Existence/update policies (--existing/--ignore-existing/--update) + are decided against the FINAL destination path at stage time so the run + decides exactly what an immediate (non-delayed) run would decide; the staged + file is then never re-checked at publication. Backups are deferred to + publication so the final destination is untouched until the transfer ends. */ +static FileSaveResult file_stage_delayed_update(const char* root_directory, + const char* destination_path, const File* file, + Config* config) { + if (!config) + return FILE_SAVE_ERROR; + bool sparse = config->preserve_sparse; + FileAttrPolicy policy = file_attr_policy_from_config(config); + + if (config->existing && !file_path_exists_secure(destination_path)) + return FILE_SAVE_SKIPPED; + if (config->ignore_existing && file_path_exists_secure(destination_path)) + return FILE_SAVE_SKIPPED; + if (config->update && file_destination_is_newer_secure(destination_path, file->metadata)) + return FILE_SAVE_SKIPPED; + + FileMetadata adjusted_metadata; + const FileMetadata* metadata = file->metadata; + if (metadata && config->chmod_spec && *config->chmod_spec) { + adjusted_metadata = *metadata; + if (!chmod_apply(adjusted_metadata.mode, config->chmod_spec, &adjusted_metadata.mode)) + return FILE_SAVE_ERROR; + metadata = &adjusted_metadata; + } + + if (!config->delay_context) { + config->delay_context = delay_updates_context_create(root_directory); + if (!config->delay_context) + return FILE_SAVE_ERROR; + } + DelayUpdatesContext* context = config->delay_context; + if (!delay_updates_prepare(context)) + return FILE_SAVE_ERROR; + + char* staged_path = path_cat(context->staging_root, file->path); + if (!staged_path) + return FILE_SAVE_ERROR; + + /* The staged location is brand new (stale leftovers from a prior crash were + wiped by prepare), so the plain atomic temp+rename engine installs the + complete file there. --temp-dir scratch is deliberately not layered on + top of the delay-updates staging tree. A --link-dest basis file is hard + linked into the staging tree (so publication's rename keeps the link). */ + bool ok; + if (file->basis_link) { + ok = file_to_disk_secure_link(staged_path, file->basis_link, file->data->data, file->data->size, + config->preallocate, metadata, policy, config->use_fsync, NULL); + } else if (file->basis_copy) { + /* --copy-dest basis hit: stream the basis into the staging tree (bounded + buffers, so an over-limit basis still stages). */ + ok = file_copy_basis_stream_attrs(staged_path, file->basis_copy, file->data->size, + config->preallocate, metadata, policy, config->update, + config->use_fsync, file->xattrs, config->fake_super, NULL); + } else { + ok = + file_to_disk_secure_attrs(staged_path, file->data->data, file->data->size, false, sparse, + config->preallocate, metadata, policy, false, false, + config->use_fsync, file->xattrs, config->fake_super, false, NULL); + } + if (!ok) { + free(staged_path); + return FILE_SAVE_ERROR; + } + + if (!delay_updates_record(context, staged_path, destination_path, file->path)) { + unlink(staged_path); + free(staged_path); + return FILE_SAVE_ERROR; + } + free(staged_path); + return FILE_SAVE_WRITTEN; +} + +/* Read the whole content of a confined regular file (used to fall back to a + byte-identical copy when a hard-link sibling's link() fails). Symlink-safe + (parent resolved via file_open_secure_parent + O_NOFOLLOW). A zero-length + file yields *out_size 0 and *out_buf NULL as a SUCCESS. Returns false only + on a real error/read failure, setting *source_absent to true when the reason + was that the path does not exist (ENOENT/ENOTDIR), so the caller can decide + between an abort and a graceful skip. */ +static bool hardlink_read_source(const char* path, void** out_buf, unsigned long long* out_size, + bool* source_absent) { + *out_buf = NULL; + *out_size = 0; + *source_absent = false; + if (!path) + return false; + char* leaf = NULL; + int parent_fd = file_open_secure_parent(path, &leaf, false); + if (parent_fd < 0) { + *source_absent = errno == ENOENT || errno == ENOTDIR; + return false; + } + /* O_NONBLOCK is a no-op for a regular file but makes openat() fail/succeed + immediately for a client-planted FIFO instead of blocking the receive + thread forever; the post-open S_ISREG gate below is the actual type check. */ + int fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); + int saved_errno = errno; + free(leaf); + close(parent_fd); + if (fd < 0) { + *source_absent = saved_errno == ENOENT || saved_errno == ENOTDIR; + return false; + } + struct stat st; + if (fstat(fd, &st) != 0 || !S_ISREG(st.st_mode)) { + close(fd); + return false; + } + unsigned long long size = (unsigned long long)st.st_size; + if (size > MAX_RECEIVE_WHOLE_FILE_SIZE || size > SIZE_MAX) { + close(fd); + return false; + } + if (size == 0) { + close(fd); + return true; + } + void* buf = protocol_alloc((size_t)size); + if (!buf) { + close(fd); + return false; + } + size_t got = 0; + while (got < (size_t)size) { + ssize_t n = read(fd, (char*)buf + got, (size_t)size - got); + if (n <= 0) { + free(buf); + close(fd); + return false; + } + got += (size_t)n; + } + close(fd); + *out_buf = buf; + *out_size = size; + return true; +} + +/* The group's first member's installed file is absent, but its destination + path was validated (a sibling is only ever processed after its group's first + member). When the sibling's OWN destination already exists it should be + left alone -- a clean skip -- rather than aborting the whole transfer (the + asymmetric --existing case: the first member was skipped because its + destination was missing, while the sibling already has one). Only when the + sibling's destination is missing too is this a genuine failure to + link/copy, which aborts. */ +static FileSaveResult hardlink_sibling_absent_first(const char* destination_path) { + if (destination_path && file_path_exists_secure(destination_path)) + return FILE_SAVE_SKIPPED; + return FILE_SAVE_ERROR; +} + +/* Install a --hard-links/-H sibling: the destination entry is atomically + replaced (temp + rename) with a hard link to the group's first member. The + first member is guaranteed already installed at `hardlink_target` under the + root because -H relies on the receiver's single-FIFO-writer pipeline (one + receive thread, one write thread, FIFO queue => wire order == write order) + plus the sender's forced sequential scan, so a sibling is always processed + after its group's first member. When link() fails (different filesystem, + filesystem refuses links) a byte-identical copy of the first member is + written instead, so the result is never partial or corrupt. With + --delay-updates the sibling is staged as a hard link to the first member's + STAGED file (publication's renames preserve the shared inode). The final + --existing/--ignore-existing/--update policies are decided against the final + destination like every normal write. */ +static FileSaveResult file_save_hardlink_sibling(const char* root_directory, const File* file, + const Config* config, bool* created) { + Config* cfg = (Config*)config; + if (!root_directory || !file || !file->path || !file->hardlink_target) + return FILE_SAVE_ERROR; + char* destination_path = path_cat(root_directory, file->path); + if (!destination_path) + return FILE_SAVE_ERROR; + bool existed = file_path_exists_secure(destination_path); + + if (cfg->existing && !file_path_exists_secure(destination_path)) { + free(destination_path); + return FILE_SAVE_SKIPPED; + } + if (cfg->ignore_existing && file_path_exists_secure(destination_path)) { + free(destination_path); + return FILE_SAVE_SKIPPED; + } + if (cfg->update && file_destination_is_newer_secure(destination_path, file->metadata)) { + free(destination_path); + return FILE_SAVE_SKIPPED; + } + + bool preallocate = cfg && cfg->preallocate; + FileAttrPolicy policy = file_attr_policy_from_config(cfg); + bool use_fsync = cfg && cfg->use_fsync; + + if (cfg->delay_updates) { + if (!cfg->delay_context) { + cfg->delay_context = delay_updates_context_create(root_directory); + if (!cfg->delay_context) { + free(destination_path); + return FILE_SAVE_ERROR; + } + } + if (!delay_updates_prepare(cfg->delay_context)) { + free(destination_path); + return FILE_SAVE_ERROR; + } + char* staged_first = path_cat(cfg->delay_context->staging_root, file->hardlink_target); + char* staged_sibling = path_cat(cfg->delay_context->staging_root, file->path); + if (!staged_first || !staged_sibling) { + free(staged_first); + free(staged_sibling); + free(destination_path); + return FILE_SAVE_ERROR; + } + void* content = NULL; + unsigned long long content_size = 0; + bool source_absent = false; + if (!hardlink_read_source(staged_first, &content, &content_size, &source_absent)) { + FileSaveResult absent_result = + source_absent ? hardlink_sibling_absent_first(destination_path) : FILE_SAVE_ERROR; + free(staged_first); + free(staged_sibling); + free(destination_path); + return absent_result; + } + FileXattrList* sibling_xattrs = + cfg->use_xattrs ? xattr_capture_path(staged_first, cfg->preserve_acls) : NULL; + bool ok = file_to_disk_secure_link_attrs(staged_sibling, staged_first, content, content_size, + preallocate, file->metadata, policy, use_fsync, + sibling_xattrs, cfg ? cfg->fake_super : false, NULL); + xattr_list_free(sibling_xattrs); + free(content); + if (ok) + ok = delay_updates_record(cfg->delay_context, staged_sibling, destination_path, file->path); + if (!ok) + unlink(staged_sibling); + free(staged_first); + free(staged_sibling); + free(destination_path); + if (ok && created && !existed) + *created = true; + return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; + } + + char* first_disk = path_cat(root_directory, file->hardlink_target); + if (!first_disk) { + free(destination_path); + return FILE_SAVE_ERROR; + } + void* content = NULL; + unsigned long long content_size = 0; + bool source_absent = false; + if (!hardlink_read_source(first_disk, &content, &content_size, &source_absent)) { + FileSaveResult absent_result = + source_absent ? hardlink_sibling_absent_first(destination_path) : FILE_SAVE_ERROR; + free(first_disk); + free(destination_path); + return absent_result; + } + /* Resolve a relative --temp-dir under the destination root, exactly as the + * primary save path does; an absolute or `..`-escaping value is rejected. */ + char* resolved_temp = NULL; + if (cfg->temp_dir) { + if (cfg->temp_dir[0] == '/' || has_path_traversal(cfg->temp_dir)) { + free(content); + free(first_disk); + free(destination_path); + return FILE_SAVE_ERROR; + } + resolved_temp = path_cat(root_directory, cfg->temp_dir); + if (!resolved_temp) { + free(content); + free(first_disk); + free(destination_path); + return FILE_SAVE_ERROR; + } + } + FileXattrList* sibling_xattrs = + cfg->use_xattrs ? xattr_capture_path(first_disk, cfg->preserve_acls) : NULL; + bool ok = file_to_disk_secure_link_attrs( + destination_path, first_disk, content, content_size, preallocate, file->metadata, policy, + use_fsync, sibling_xattrs, cfg ? cfg->fake_super : false, resolved_temp); + xattr_list_free(sibling_xattrs); + free(resolved_temp); + free(content); + free(first_disk); + free(destination_path); + if (ok && created && !existed) + *created = true; + return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; +} + +/* Validate a transmitted special rdev against the node kind implied by `mode`'s + * S_IFMT bits. Char/block devices require a legal major/minor pair (non-negative, + * range-checked); a non-device special (FIFO/socket) must carry an empty rdev. + * Used identically on the wire path and at the secure recreation site so a + * malicious/bogus rdev can never drive a dangerous node. */ +bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode) { + bool is_device = S_ISCHR(mode) || S_ISBLK(mode); + if (is_device) + return major >= 0 && minor >= 0 && major <= 0xffff && minor <= 0x00ffffff; + /* A non-device entry must actually be a special (FIFO/socket) and carry no + rdev; a regular/dir mode is never a valid special node. */ + return (S_ISFIFO(mode) || S_ISSOCK(mode)) && major == 0 && minor == 0; +} + +/* ---- Device/special node RECREATION (--devices/--specials), receiver side ---- + * + * Privilege gating: making a real device node requires CAP_MKNOD (root); making + * a FIFO works unprivileged (mkfifo). When the receiver lacks the capability, + * mknodat() fails with EPERM and the entry is SKIPPED with a warning -- the + * whole transfer must NOT abort just because the environment cannot make the + * node. CI runs non-root, so device creation is expected to skip there and + * only a FIFO is honestly assertable unprivileged. + * + * Confinement: the parent directory is opened fd-relative below the receive + * root (file_open_secure_parent: O_NOFOLLOW, no "..", root-checked) and the + * node is created with mknodat()/mkfifoat(), so it can never be placed outside + * the confined root and never follows a symlink. + * + * rdev validation: a malicious/bogus rdev (negative, out-of-range) is rejected + * here as well as on the wire (file_receive_special / chunk_deserialize), and a + * non-device entry must carry an empty rdev. + */ +static FileSaveResult file_save_special_to_disk(const char* root_directory, const File* file, + const Config* config, bool* created) { + /* The empty-path and structural checks stay unconditional; the redundant + ".." list-path re-check is skipped under --trust-sender exactly like the + receive layer (confinement is deferred to the secure parent walk below, + which is never disabled). */ + if (!root_directory || !file || !file->path || file->path[0] == '\0' || + (!file_get_trust_sender() && has_path_traversal(file->path)) || !file->metadata) + return FILE_SAVE_ERROR; + + mode_t mode = file->metadata->mode; + bool is_char = S_ISCHR(mode); + bool is_blk = S_ISBLK(mode); + bool is_fifo = S_ISFIFO(mode); + bool is_sock = S_ISSOCK(mode); + if (!is_char && !is_blk && !is_fifo && !is_sock) { + log_message(LOG_LEVEL_ERROR, "Special node has no device/FIFO/socket mode"); + return FILE_SAVE_ERROR; + } + if (is_char || is_blk) { + if (!config || !config->preserve_devices) + return FILE_SAVE_SKIPPED; + /* --super / --no-super (P7 Wave E): char/block device-node creation is a + super-user activity. --no-super forbids it even for a root receiver; + AUTO and --super attempt it (an unprivileged attempt is refused by the + kernel and skipped). The helper is evaluated against THIS config's mode + so the policy does not depend on a prior identity_set_active(). Pure + FIFO creation is unprivileged and deliberately NOT gated here. */ + if (!privilege_super_mode_permitted(config->super_mode)) { + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, + "skipping %s: super-user device-node creation is not permitted on this receiver", + escaped_path ? escaped_path : ""); + free(escaped_path); + return FILE_SAVE_SKIPPED; + } + } else if (is_fifo || is_sock) { + /* FIFOs and unix sockets are recreated by --specials. mknod(S_IFSOCK) + works unprivileged on Linux (the node carries no live socket), so unlike + a socket bound to a live fd it can be materialized. */ + if (!config || !config->preserve_specials) + return FILE_SAVE_SKIPPED; + } + /* Defense-in-depth rdev/type validation (also done on the wire path). */ + if (!file_special_rdev_valid(file->rdev_major, file->rdev_minor, mode)) { + log_message(LOG_LEVEL_ERROR, "Rejected out-of-range device rdev %d:%d", file->rdev_major, + file->rdev_minor); + return FILE_SAVE_ERROR; + } + + char* destination = path_cat(root_directory, file->path); + if (!destination) + return FILE_SAVE_ERROR; + bool existed = file_path_exists_secure(destination); + char* leaf = NULL; + int parent_fd = file_open_secure_parent(destination, &leaf, true); + if (parent_fd < 0) { + free(destination); + return FILE_SAVE_ERROR; + } + + /* --existing / --ignore-existing / --update decide against the node that + would be replaced, mirroring the regular-file path. */ + if (config->existing && !file_path_exists_secure(destination)) { + close(parent_fd); + free(leaf); + free(destination); + return FILE_SAVE_SKIPPED; + } + if (config->ignore_existing && file_path_exists_secure(destination)) { + close(parent_fd); + free(leaf); + free(destination); + return FILE_SAVE_SKIPPED; + } + if (config->update && file_destination_is_newer_secure(destination, file->metadata)) { + close(parent_fd); + free(leaf); + free(destination); + return FILE_SAVE_SKIPPED; + } + + dev_t rdev = 0; + mode_t create_mode; + if (is_char) { + create_mode = S_IFCHR; + rdev = makedev((unsigned)file->rdev_major, (unsigned)file->rdev_minor); + } else if (is_blk) { + create_mode = S_IFBLK; + rdev = makedev((unsigned)file->rdev_major, (unsigned)file->rdev_minor); + } else if (is_sock) { + create_mode = S_IFSOCK; + } else { + create_mode = S_IFIFO; + } + const char* node_kind = (is_char || is_blk) ? "device" : (is_fifo ? "FIFO" : "socket"); + /* Under -p/--perms rsync copies the source's permission and special bits; a + * kernel that denies setuid/setgid/sticky reports the failure rather than + * having them masked here. Without -p the node is created like any other new + * entry: source_mode & 0777 & ~umask. When super-user activities are + * forbidden, the special bits are stripped even under -p (they are + * super-user activities just like device-node creation). */ + mode_t perms = config->preserve_perms ? (mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777)) + : (mode & 0777 & ~(mode_t)file_process_umask()); + if (!privilege_super_mode_permitted(config->super_mode)) + perms &= ~(mode_t)(S_ISUID | S_ISGID | S_ISVTX); + + int rc = is_fifo ? mkfifoat(parent_fd, leaf, perms) + : mknodat(parent_fd, leaf, create_mode | perms, rdev); + if (rc != 0) { + if (errno == EEXIST) { + /* An entry already exists: only skip when it already is a matching node; + never replace an existing directory or unrelated entry with the node. */ + struct stat st; + if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) == 0 && + ((is_char && S_ISCHR(st.st_mode)) || (is_blk && S_ISBLK(st.st_mode)) || + (is_fifo && S_ISFIFO(st.st_mode)) || (is_sock && S_ISSOCK(st.st_mode)))) { + close(parent_fd); + free(leaf); + free(destination); + return FILE_SAVE_SKIPPED; + } + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "refusing to replace existing entry with %s: %s (skipped)", + node_kind, escaped_path ? escaped_path : ""); + free(escaped_path); + } else if (errno == EPERM || errno == EACCES) { + /* Missing CAP_MKNOD / parent write permission: the environment cannot + create the node, so skip instead of failing the whole run. */ + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, + "skipping %s: cannot create %s node (%s)\n" + " --devices/--specials node creation needs privilege (CAP_MKNOD)", + escaped_path ? escaped_path : "", node_kind, strerror(errno)); + free(escaped_path); + } else { + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "failed to create %s %s: %s (skipped)", node_kind, + escaped_path ? escaped_path : "", strerror(errno)); + free(escaped_path); + } + close(parent_fd); + free(leaf); + free(destination); + return FILE_SAVE_SKIPPED; + } + + /* Apply times on the fresh node (utimensat, no-follow) per the negotiated + * per-attribute policy: mtime only under -t, atime only under -U. The slot + * not requested stays UTIME_OMIT so it is left untouched. */ + FileAttrPolicy policy = file_attr_policy_from_config(config); + if (policy.times || (policy.atimes && file->metadata->atime_valid)) { + struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT}, + {.tv_sec = 0, .tv_nsec = UTIME_OMIT}}; + if (policy.times) { + times[1].tv_sec = file->metadata->mtime_sec; + times[1].tv_nsec = file->metadata->mtime_nsec; + } + if (policy.atimes && file->metadata->atime_valid) { + times[0].tv_sec = file->metadata->atime_sec; + times[0].tv_nsec = file->metadata->atime_nsec; + } + utimensat(parent_fd, leaf, times, AT_SYMLINK_NOFOLLOW); + } + /* P7 Wave E: apply the negotiated ownership to the node ITSELF. A FIFO is + created unprivileged, but --copy-as and explicit identity policies own + every entry (a char/block node path is already privilege-gated above). The + no-follow helper changes the node's own ownership without dereferencing it; + it is a no-op unless an identity policy is active. */ + bool owner_ok = true; + if (identity_active_enabled()) + owner_ok = identity_apply_ownership_link(parent_fd, leaf, (int32_t)file->metadata->uid, + (int32_t)file->metadata->gid); + close(parent_fd); + free(leaf); + free(destination); + /* A failed required --copy-as ownership marks the node as failed; every other + * identity policy stays best-effort. */ + if (owner_ok && created && !existed) + *created = true; + return owner_ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; +} + +/* --write-devices (receiver): write the received data directly into an EXISTING + * device node on the destination instead of creating a regular file. The node + * must already exist and be a char/block device (the device itself is opened and + * followed); it is confined to the receive root via file_open_secure_parent. + * Dangerous by nature, so deliberately restricted: a missing/non-device + * destination, or a write failure, is SKIPPED with a warning rather than + * allowed. On environments without device access the run still succeeds (the + * entry is skipped), never aborts. */ +static FileSaveResult file_save_write_device(const char* root_directory, const File* file) { + if (!root_directory || !file || !file->path || file->path[0] == '\0' || + (!file_get_trust_sender() && has_path_traversal(file->path))) + return FILE_SAVE_ERROR; + if (!file->data) + return FILE_SAVE_ERROR; + char* destination = path_cat(root_directory, file->path); + if (!destination) + return FILE_SAVE_ERROR; + char* leaf = NULL; + int parent_fd = file_open_secure_parent(destination, &leaf, false); + if (parent_fd < 0) { + free(destination); + return FILE_SAVE_SKIPPED; + } + /* O_NONBLOCK: a pre-existing FIFO at the target would otherwise block the + receive thread forever on open(2). With it the open only succeeds for a + readerless FIFO with O_RDWR (which the device fstat gate rejects anyway) + or fails with ENXIO/EAGAIN, both treated as a normal skip below. */ + int fd = openat(parent_fd, leaf, O_WRONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); + int saved_errno = errno; + free(leaf); + close(parent_fd); + if (fd < 0) { + free(destination); + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + const char* shown_path = escaped_path ? escaped_path : ""; + if (saved_errno == ENXIO || saved_errno == EAGAIN) { + /* A FIFO with no reader / an unreadable special: skip like every other + unusable write-devices target instead of blocking or failing. */ + log_message(LOG_LEVEL_WARNING, "write-devices: %s not writable (%s); skipped", shown_path, + strerror(saved_errno)); + } else { + log_message(LOG_LEVEL_WARNING, "write-devices: cannot open %s (%s); skipped", shown_path, + strerror(saved_errno)); + } + free(escaped_path); + return FILE_SAVE_SKIPPED; + } + struct stat st; + if (fstat(fd, &st) != 0 || !(S_ISCHR(st.st_mode) || S_ISBLK(st.st_mode))) { + close(fd); + free(destination); + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "write-devices: %s is not a device node; skipped", + escaped_path ? escaped_path : ""); + free(escaped_path); + return FILE_SAVE_SKIPPED; + } + bool ok = true; + if (file->data->size > 0) { + size_t total = (size_t)file->data->size; + size_t written = 0; + while (written < total) { + ssize_t n = write(fd, (char*)file->data->data + written, total - written); + if (n <= 0) { + ok = false; + break; + } + written += (size_t)n; + } + } + close(fd); + free(destination); + return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_SKIPPED; +} + +/* ---- file_save_to_disk_full_ex decomposition ---- + * + * The regular-file install path is split into small static helpers that share + * one FileSavePlan (the owned path-set) and route every exit through a single + * cleanup epilogue so the path-set is released exactly once. Each helper owns + * one decision: request validation, special/device dispatch, directory and + * symlink creation, path resolution, the --existing/--ignore-existing/--update/ + * --force/--backup pre-write policies, and the data install. The ordering of + * every check and every protocol/write operation is unchanged. + */ + +/* Owned state for one regular-file install. The path-set pointers are owned by + * the plan and freed together by file_save_plan_dispose(). */ +typedef struct { + const char* root_directory; + const File* file; + const Config* config; + bool backup_enabled; + bool inplace; + bool sparse; + FileAttrPolicy policy; + const char* backup_suffix; + const char* backup_dir; + const char* partial_dir; + const char* temp_dir; + bool use_partial_root; + bool dest_existed; + char* confined_backup; + char* confined_partial; + char* disk_path; + char* destination_path; + char* backup_path; + char* parent_copy; + char* confined_temp; +} FileSavePlan; + +static void file_save_plan_init(FileSavePlan* plan, const char* root_directory, const File* file, + const Config* config) { + memset(plan, 0, sizeof(*plan)); + plan->root_directory = root_directory; + plan->file = file; + plan->config = config; + plan->backup_enabled = config && config->backup && !config->ignore_existing; + plan->inplace = config && config->inplace; + plan->sparse = config && config->preserve_sparse; + plan->policy = file_attr_policy_from_config(config); + plan->backup_suffix = (config && config->suffix) ? config->suffix : "~"; + plan->backup_dir = (config && config->backup_dir) ? config->backup_dir : NULL; + plan->partial_dir = (config && config->partial_dir) ? config->partial_dir : NULL; + plan->temp_dir = (config && config->temp_dir) ? config->temp_dir : NULL; + plan->use_partial_root = plan->partial_dir && config && config->partial; +} + +/* Single cleanup epilogue: release the whole owned path-set exactly once. */ +static void file_save_plan_dispose(FileSavePlan* plan) { + free(plan->parent_copy); + free(plan->backup_path); + free(plan->confined_backup); + free(plan->confined_partial); + free(plan->destination_path); + free(plan->disk_path); + free(plan->confined_temp); +} + +/* Structural validation of the received entry. */ +static bool file_save_validate(const FileSavePlan* plan) { + const File* file = plan->file; + const char* backup_suffix = plan->backup_suffix; + return file && file->path && file->data && + (file->data->size == 0 || file->data->data || file->basis_link || file->basis_copy) && + (file_get_trust_sender() || !has_path_traversal(file->path)) && + (!plan->backup_enabled || + (backup_suffix && backup_suffix[0] != '\0' && strchr(backup_suffix, '/') == NULL && + strcmp(backup_suffix, ".") != 0 && strcmp(backup_suffix, "..") != 0)); +} + +/* Explicit directory entry (--dirs). */ +static FileSaveResult file_save_directory_to_disk(const FileSavePlan* plan, bool* created) { + const File* file = plan->file; + if (file->path[0] == '\0' || (!file_get_trust_sender() && has_path_traversal(file->path))) { + log_message(LOG_LEVEL_ERROR, "Invalid directory path received"); + return FILE_SAVE_ERROR; + } + char* dir_path = path_cat(plan->root_directory, file->path); + if (!dir_path) + return FILE_SAVE_ERROR; + bool dir_existed = file_path_exists_secure(dir_path); + bool ok = file_ensure_directory_secure(dir_path); + /* P7 Wave E: apply the negotiated ownership to the directory ITSELF (not + just the files inside it). --copy-as and every explicit identity policy + own every entry, so a directory must not keep the receiver's owner while + its children get the policy owner. Applied no-follow on the confined + parent fd after the mkdir; identity_apply_ownership_link() is itself a + no-op unless an identity policy is active. */ + if (ok && file->metadata && identity_active_enabled()) { + char* leaf = NULL; + int parent_fd = file_open_secure_parent(dir_path, &leaf, false); + if (parent_fd >= 0) { + if (!identity_apply_ownership_link(parent_fd, leaf, (int32_t)file->metadata->uid, + (int32_t)file->metadata->gid)) + ok = false; + close(parent_fd); + } else if (identity_copy_as_active()) { + /* The directory exists (ok) but its required --copy-as ownership could + not be applied because the confined parent could not be opened. */ + ok = false; + } + free(leaf); + } + free(dir_path); + if (ok && created && !dir_existed) + *created = true; + return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; +} + +/* Symlink entry. (The process-wide --keep-dirlinks policy is set once by the + connection handler from the negotiated config, before any receiver/writer + threads start, so it is stable throughout this walk.) */ +static FileSaveResult file_save_symlink_to_disk(const FileSavePlan* plan, bool* created) { + const File* file = plan->file; + const Config* config = plan->config; + if (!file->symlink_target || file->path[0] == '\0' || + (!file_get_trust_sender() && has_path_traversal(file->path))) { + log_message(LOG_LEVEL_ERROR, "Invalid symlink entry received"); + return FILE_SAVE_ERROR; + } + char* link_path = path_cat(plan->root_directory, file->path); + if (!link_path) + return FILE_SAVE_ERROR; + bool link_existed = file_path_exists_secure(link_path); + /* The link value is stored verbatim (rsync -l parity: absolute and + ".."-bearing targets are preserved; the scanner's --safe-links / + --copy-unsafe-links decide which links are sent at all). --munge-links + is a RECEIVER-side rewrite: the stored target is prefixed with + /rsyncd-munged/, making the link unusable while the referenced directory + does not exist -- exactly as rsync's receiver munges. Only the link's + own placement path is confined below the receive root. */ + bool munge = config && config->munge_links; + char* target = str_dup(file->symlink_target); + bool ok = target != NULL; + if (ok && munge) { + char* munged = file_symlink_munge(target); + free(target); + target = munged; + ok = target != NULL; + } + if (!ok) { + free(target); + free(link_path); + return FILE_SAVE_SKIPPED; + } + char* parent = str_dup(link_path); + if (parent) { + /* Propagate a failed --copy-as ownership of the parent directory this + creates; every other failure mode stays best-effort as before. */ + ok = file_ensure_directory_secure(dirname(parent)); + free(parent); + } + if (ok) + ok = file_symlink_at_secure(link_path, target); + free(target); + /* P7 Wave D: apply the symlink's own metadata with no-follow primitives + (utimensat/lchown/fchmodat AT_SYMLINK_NOFOLLOW). -J/--omit-link-times + suppresses the timestamps; ownership stays gated by the identity policy. + A symlink has no children, so this can be applied immediately. */ + if (ok && config && config->use_metadata) { + FileAttrPolicy link_policy = file_attr_policy_from_config(config); + ok = file_restore_symlink_metadata(link_path, file->metadata, link_policy, + config->omit_link_times); + } + if (ok && created && !link_existed) + *created = true; + free(link_path); + return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; +} + +/* Device/special node (--devices/--specials) and --write-devices dispatch. */ +static bool file_save_try_special_dispatch(const FileSavePlan* plan, bool* created, + FileSaveResult* out) { + const File* file = plan->file; + const Config* config = plan->config; + /* Device/special node (--devices/--specials): recreate the node instead of + writing content (privilege-gated, confined, rdev-validated). */ + if (file->is_special) { + *out = file_save_special_to_disk(plan->root_directory, file, config, created); + return true; + } + /* --write-devices: write straight into an existing device node. Writing + into a device is a super-user activity, so --no-super must suppress it just + like device-node creation; the default AUTO/--super attempt it (the wide + open below keeps its own confinement and best-effort skip semantics). */ + if (config && config->write_devices) { + if (!privilege_super_mode_permitted(config->super_mode)) { + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, + "write-devices: %s skipped: super-user activities are not permitted on this " + "receiver", + escaped_path ? escaped_path : "(null)"); + free(escaped_path); + *out = FILE_SAVE_SKIPPED; + return true; + } + *out = file_save_write_device(plan->root_directory, file); + return true; + } + return false; +} + +/* Resolve and confine the backup/partial/temp directories and the destination + and disk paths. Returns false on an invalid/escaping option or an + allocation failure (the caller routes to the cleanup epilogue). */ +static bool file_save_resolve_paths(FileSavePlan* plan) { + /* These options arrive from the client. --backup-dir, --partial-dir and + --temp-dir are names below the server root, never independent filesystem + roots: an absolute or `..`-escaping value is rejected outright (rsync's + daemon confines temp-dir to the module the same way). A relative temp dir + is resolved under the receive root below; if that resolution still lands on + a different filesystem than the destination the install falls back to a + non-atomic copy (see file_to_disk_secure_impl), never an abort. */ + if ((plan->backup_dir && (plan->backup_dir[0] == '/' || has_path_traversal(plan->backup_dir))) || + (plan->partial_dir && + (plan->partial_dir[0] == '/' || has_path_traversal(plan->partial_dir))) || + (plan->temp_dir && (plan->temp_dir[0] == '/' || has_path_traversal(plan->temp_dir)))) + return false; + if (plan->backup_dir && + !(plan->confined_backup = path_cat(plan->root_directory, plan->backup_dir))) + return false; + if (plan->partial_dir && + !(plan->confined_partial = path_cat(plan->root_directory, plan->partial_dir))) + return false; + + const char* actual_root = plan->use_partial_root ? plan->confined_partial : plan->root_directory; + plan->destination_path = path_cat(plan->root_directory, plan->file->path); + plan->disk_path = path_cat(actual_root, plan->file->path); + if (plan->destination_path == NULL || plan->disk_path == NULL) + return false; + /* Snapshot the final destination's existence BEFORE any backup/force/partial + step can move or remove it, so the receiver can report rsync's + `Number of created files` (protocol 2.28.0). */ + plan->dest_existed = file_path_exists_secure(plan->destination_path); + return true; +} + +typedef enum { + FILE_SAVE_POLICY_CONTINUE, /* proceed to the install */ + FILE_SAVE_POLICY_SKIP, /* --existing/--ignore-existing/--update skip */ + FILE_SAVE_POLICY_ERROR, /* --force/--backup failure */ +} FileSavePolicyOutcome; + +/* The immediate-install pre-write policies: --existing, --ignore-existing, + --update, --force and --backup, in that order. */ +static FileSavePolicyOutcome file_save_apply_prewrite_policies(FileSavePlan* plan) { + const Config* config = plan->config; + const File* file = plan->file; + + /* --existing checks the final destination, not a temporary partial path. */ + if (config && config->existing && !file_path_exists_secure(plan->destination_path)) + return FILE_SAVE_POLICY_SKIP; + + /* --ignore-existing checks the final destination before partial files or + overwrite policies can modify it. */ + if (config && config->ignore_existing) { + bool exists = file_path_exists_secure(plan->destination_path); + if (exists) + return FILE_SAVE_POLICY_SKIP; + } + + /* --update is receiver-side policy: never replace a newer destination. + In partial-dir mode the entry that would be replaced is the real + destination, not the temporary partial file. The secure stat does not + require read permission on the destination. */ + const char* update_target = plan->use_partial_root ? plan->destination_path : plan->disk_path; + if (config && config->update && file_destination_is_newer_secure(update_target, file->metadata)) + return FILE_SAVE_POLICY_SKIP; + + /* --force (rsync semantics): an incoming regular file may replace a + destination DIRECTORY by removing that (possibly non-empty, symlink-safe) + tree first, so the atomic temp+rename below can install the file. Only the + immediate-install path does this: a --delay-updates run stages into its own + tree and is unaffected here (its publication renames over regular files + only). The blocking directory is removed only after the --update / + --existing / --ignore-existing decisions above, which see it as an existing + destination entry. */ + if (config && config->force_delete && !file->is_dir && + file_directory_exists_secure(plan->destination_path)) { + if (!file_remove_tree_secure(plan->destination_path)) + return FILE_SAVE_POLICY_ERROR; + } + + if (plan->backup_enabled) { + /* Back up the entry that the incoming write will replace. When writing + through a partial dir the pre-existing destination file is the one to + preserve; any stale partial file is overwritten without a backup. */ + const char* replace_target = plan->use_partial_root ? plan->destination_path : plan->disk_path; + struct stat backup_stat; + if (file_stat_secure(replace_target, &backup_stat)) { + if (plan->backup_dir) { + plan->backup_path = path_cat(plan->confined_backup, file->path); + } else { + size_t path_len = strlen(replace_target); + size_t suffix_len = strlen(plan->backup_suffix); + if (path_len > SIZE_MAX - suffix_len - 1) + return FILE_SAVE_POLICY_ERROR; + plan->backup_path = malloc(path_len + suffix_len + 1); + if (plan->backup_path) { + memcpy(plan->backup_path, replace_target, path_len); + memcpy(plan->backup_path + path_len, plan->backup_suffix, suffix_len + 1); + } + } + if (!plan->backup_path) + return FILE_SAVE_POLICY_ERROR; + plan->parent_copy = str_dup(plan->backup_path); + if (!plan->parent_copy || !file_ensure_directory_secure(dirname(plan->parent_copy))) + return FILE_SAVE_POLICY_ERROR; + free(plan->parent_copy); + plan->parent_copy = NULL; + if (!file_rename_secure(replace_target, plan->backup_path)) + return FILE_SAVE_POLICY_ERROR; + free(plan->backup_path); + plan->backup_path = NULL; + } + } + return FILE_SAVE_POLICY_CONTINUE; +} + +/* Install the file data into the destination (or staging/partial path): + --link-dest hard link, --copy-dest streamed copy, or the plain atomic + temp+rename engine with per-file xattr/--fake-super application. */ +static bool file_save_install_data(FileSavePlan* plan, const FileMetadata* metadata, + unsigned* created_dirs) { + const File* file = plan->file; + const Config* config = plan->config; + char* count_floor = file_transfer_root_floor(config); + bool ok; + if (config && file->basis_link) { + ok = file_to_disk_secure_link_attrs_counted( + plan->disk_path, file->basis_link, file->data->data, file->data->size, config->preallocate, + metadata, plan->policy, config->use_fsync, file->xattrs, config->fake_super, + plan->confined_temp, created_dirs, count_floor); + } else if (config && file->basis_copy) { + /* --copy-dest: stream the basis bytes through a bounded buffer so a basis + larger than any whole-file bound still materializes. The source + metadata was transmitted with the check frame. */ + ok = file_copy_basis_stream_attrs(plan->disk_path, file->basis_copy, file->data->size, + config->preallocate, metadata, plan->policy, config->update, + config->use_fsync, file->xattrs, config->fake_super, + plan->confined_temp); + } else { + /* The plain no-replace / update / with-fsync engines, plus per-file xattr + (-X/-A) and --fake-super application on the written fd. */ + ok = file_to_disk_secure_attrs_counted( + plan->disk_path, file->data->data, file->data->size, plan->inplace, plan->sparse, + config && config->preallocate, metadata, plan->policy, config && config->update, + config && config->ignore_existing, config && config->use_fsync, file->xattrs, + config ? config->fake_super : false, config ? config->partial : false, plan->confined_temp, + created_dirs, count_floor); + } + free(count_floor); + return ok; +} + +FileSaveResult file_save_to_disk_full_ex(const char* root_directory, const File* file, + const Config* config, bool* created, + unsigned* created_dirs) { + if (created) + *created = false; + if (created_dirs) + *created_dirs = 0; + /* Central no-mutation guard: a server-contacting --dry-run (or a local batch + apply that somehow carries dry_run) must never touch the destination, no + matter which caller reached this primitive. The per-caller guards remain, + but this is the last line of defense for every save path. Report SKIPPED + so a --remove-source-files sender correctly keeps its source. */ + if (config && config->dry_run) + return FILE_SAVE_SKIPPED; + + FileSavePlan plan; + file_save_plan_init(&plan, root_directory, file, config); + FileSaveResult result = FILE_SAVE_ERROR; + + if (!file_save_validate(&plan)) { + log_message(LOG_LEVEL_ERROR, "Invalid file or path received"); + goto out; + } + + /* P7 Wave D #1: a STATUS_DIR_TIMES entry is RECORD-ONLY. The scanner + captures every traversed directory -- including empty ones whose parents + were never created by a child write and directories pruned by + -m/--prune-empty-dirs. Creating them here would resurrect empty + directories (an -a behavior change) and could abort the whole transfer on a + pre-existing regular file/symlink at the mirror path. Short-circuit before + any device/write-devices/directory branch and report it as skipped so the + sink still accumulates its metadata for the deferred DirTimeList + application, but create nothing. */ + if (file->dir_time_only) { + result = FILE_SAVE_SKIPPED; + goto out; + } + + FileSaveResult dispatched; + if (file_save_try_special_dispatch(&plan, created, &dispatched)) { + result = dispatched; + goto out; + } + + /* Explicit directory entries (--dirs) carry an empty payload; the entry is + created as a directory under the receive root, applying the same secure + mkdir-parent semantics as regular writes. Directories are created + immediately (they are never staged by --delay-updates, matching rsync, + where directory creation is not delayed). */ + if (file->is_dir) { + result = file_save_directory_to_disk(&plan, created); + goto out; + } + + if (file->is_symlink) { + result = file_save_symlink_to_disk(&plan, created); + goto out; + } + + /* --hard-links/-H sibling: a later member of a link group arrives with no + payload and is installed as a hard link to (or, on link() failure, a + byte-identical copy of) the group's first member. Handled entirely here, + before the normal data-write paths (which would create an empty file). */ + if (file->link_group != 0 && !file->link_first && file->hardlink_target != NULL) { + result = file_save_hardlink_sibling(root_directory, file, config, created); + goto out; + } + + if (!file_save_resolve_paths(&plan)) + goto out; + + /* --delay-updates diverts the whole write into the staging tree; the rest of + this function is the immediate-install path. */ + if (config && config->delay_updates) { + FileSaveResult staged = + file_stage_delayed_update(root_directory, plan.destination_path, file, (Config*)config); + if (staged == FILE_SAVE_WRITTEN && created && !plan.dest_existed) + *created = true; + result = staged; + goto out; + } + + FileSavePolicyOutcome policy_outcome = file_save_apply_prewrite_policies(&plan); + if (policy_outcome == FILE_SAVE_POLICY_SKIP) { + result = FILE_SAVE_SKIPPED; + goto out; + } + if (policy_outcome == FILE_SAVE_POLICY_ERROR) + goto out; + + FileMetadata adjusted_metadata; + const FileMetadata* metadata = file->metadata; + if (metadata && config && config->chmod_spec && *config->chmod_spec) { + adjusted_metadata = *metadata; + if (!chmod_apply(adjusted_metadata.mode, config->chmod_spec, &adjusted_metadata.mode)) + goto out; + metadata = &adjusted_metadata; + } + + /* A configured --temp-dir sends the temporary working copy to a scratch + directory; the engine then atomically renames the completed file into the + final destination directory. A relative temp dir is resolved under the + receive root and must already exist (an absolute or `..`-escaping value was + rejected above); the engine falls back to a non-atomic copy on EXDEV. The + partial-dir flow already keeps its working copy in a separate directory and + --inplace writes directly, so neither diverts through the scratch dir + (matching rsync, where --inplace/--partial-dir supersede --temp-dir). */ + bool use_temp_dir = plan.temp_dir != NULL && !plan.inplace && !plan.use_partial_root; + if (use_temp_dir) { + plan.confined_temp = path_cat(root_directory, plan.temp_dir); + if (!plan.confined_temp) + goto out; + /* A user-supplied trailing slash would leave the scratch path ending in + "/", which has no final component to create/open. Normalize it away. */ + size_t temp_len = strlen(plan.confined_temp); + while (temp_len > 1 && plan.confined_temp[temp_len - 1] == '/') + plan.confined_temp[--temp_len] = '\0'; + } + + /* A --link-dest basis hit installs an atomic hard link (with a byte-copy + fallback); --inplace and the update/no-replace write variants do not + apply to a fresh hard link, whose inode attributes already match. The + existing/ignore-existing/update/backup preamble above has already made the + policy decision. */ + if (!file_save_install_data(&plan, metadata, created_dirs)) + goto out; + + /* --partial --partial-dir writes the complete file under the partial dir so + interrupted transfers leave a resumable copy there. Once the file is + fully written it must be atomically installed at the real destination; + otherwise completed transfers would linger under the partial dir. */ + if (plan.use_partial_root) { + if (!file_rename_secure(plan.disk_path, plan.destination_path)) + goto out; + } + + if (created && !plan.dest_existed) + *created = true; + result = FILE_SAVE_WRITTEN; + +out: + file_save_plan_dispose(&plan); + return result; +} + +void receiver_stats_note_saved(ReceiverStats* stats, const File* file, bool created, + unsigned created_dirs) { + if (!stats || !file) + return; + /* A basis-dir hit (--link-dest/--copy-dest) materializes bytes the sender + * never transferred. rsync reports no literal data and no created entry for + * such a file, and does not count the parent directories it creates only to + * hold it, so exclude the whole entry from the receiver tallies. */ + bool basis_sourced = file->basis_link != NULL || file->basis_copy != NULL; + if (basis_sourced) + return; + bool is_sibling = file->link_group != 0 && !file->link_first; + if (!file->is_dir && !file->is_symlink && !file->is_special && !is_sibling) { + unsigned long long literal = file->literal_bytes; + if (literal == 0 && file->matched_bytes == 0) + literal = file->data ? file->data->size : 0; + stats->literal_bytes += literal; + } + stats->created_dir += created_dirs; + if (!created) + return; + if (file->is_dir) + stats->created_dir++; + else if (file->is_symlink) + stats->created_link++; + else if (file->is_special) + stats->created_special++; + else + stats->created_reg++; +} diff --git a/src/shared/file_save.h b/src/shared/file_save.h new file mode 100644 index 0000000..d995e5f --- /dev/null +++ b/src/shared/file_save.h @@ -0,0 +1,40 @@ +#ifndef FILE_SAVE_H +#define FILE_SAVE_H + +#include "config.h" +#include "file_types.h" +#include "format.h" +#include + +/* Save-to-disk module: regular-file/symlink/hardlink/special install, xattr + * application, --fake-super and the --delay-updates staging path. These + * declarations are re-exported by the file_receive.h facade. */ + +/* Outcome of a single file_save_to_disk operation. The receiver needs to + distinguish "written" from "skipped" so --remove-source-files can be told + which sources were actually stored. */ +typedef enum { FILE_SAVE_ERROR = 0, FILE_SAVE_WRITTEN = 1, FILE_SAVE_SKIPPED = 2 } FileSaveResult; + +bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode); + +FileSaveResult file_save_to_disk_full(const char* root_directory, const File* file, + const Config* config); +/* Protocol 2.28.0 variant: also reports through `created` (when non-NULL) + * whether the destination entry did not exist before this save, and through + * `created_dirs` how many parent directories the confined walk created, so the + * receiver can build rsync's `Number of created files` breakdown. The plain + * file_save_to_disk_full() is this with both out-params NULL. */ +FileSaveResult file_save_to_disk_full_ex(const char* root_directory, const File* file, + const Config* config, bool* created, + unsigned* created_dirs); +bool file_save_to_disk(const char* root_directory, const File* file, const Config* config); + +/* Protocol 2.28.0 receiver counter accumulator: fold one successfully saved + * entry into `stats`, adding its receiver-observed literal bytes and, when + * `created`, the matching created-by-type counter (regular file / symlink / + * special) plus `created_dirs` implicitly-created parent directories. + * Non-first hardlink siblings contribute no literal bytes. */ +void receiver_stats_note_saved(ReceiverStats* stats, const File* file, bool created, + unsigned created_dirs); + +#endif diff --git a/src/shared/incremental_check.c b/src/shared/incremental_check.c new file mode 100644 index 0000000..98d4eb0 --- /dev/null +++ b/src/shared/incremental_check.c @@ -0,0 +1,1676 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "array_list.h" +#include "charset.h" +#include "chmod.h" +#include "chunk.h" +#include "compression.h" +#include "config.h" +#include "data.h" +#include "delay_updates.h" +#include "delta.h" +#include "file.h" +#include "format.h" +#include "identity.h" +#include "incremental_check.h" +#include "log.h" +#include "metadata.h" +#include "protocol.h" +#include "utils.h" +#include "xattr.h" + +/* Receive a file's xattr block (when the config enables xattr transport) and + * attach it to `file`. Returns false on a malformed/oversized frame. */ +bool receive_file_xattrs(File* file, int fd, const Config* config) { + if (!config->use_xattrs) + return true; + int xok = 0; + FileXattrList* list = xattr_receive(fd, &xok, config->preserve_acls); + if (!xok) { + xattr_list_free(list); + return false; + } + file->xattrs = list; + return true; +} + +static File* receive_delta_file(int fd, const Config* config, const char* check_path, + void* old_data, unsigned long long old_size, bool* failed) { + if (!old_data) { + free(old_data); /* defensive: old_data is always non-NULL today */ + *failed = true; + return NULL; + } + + DeltaSignature* sig = delta_signature_create_seeded(old_data, old_size, config->delta_block_size, + (uint32_t)config->checksum_seed); + if (!sig) { + free(old_data); + *failed = true; + return NULL; + } + + Data* sig_data = delta_signature_serialize(sig); + if (!sig_data) { + delta_signature_destroy(sig); + free(old_data); + *failed = true; + return NULL; + } + + bool sig_sent = send_status(fd, STATUS_DELTA_SIGNATURE) && send_data(fd, sig_data); + data_destroy(sig_data); + + if (!sig_sent) { + delta_signature_destroy(sig); + free(old_data); + *failed = true; + return NULL; + } + + Status resp; + if (!receive_status(fd, &resp)) { + delta_signature_destroy(sig); + free(old_data); + *failed = true; + return NULL; + } + + if (resp == STATUS_DELTA_DATA) { + Data* delta_data = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); + if (!delta_data) { + delta_signature_destroy(sig); + free(old_data); + *failed = true; + return NULL; + } + + Data* raw_delta = delta_data; + if (config->use_compression && + !compression_should_skip_with_suffixes( + check_path, config->skip_compress_suffixes, + config->skip_compress_set ? config->skip_compress_count : -1)) { + ProtocolSession* owner = delta_data->owner; + raw_delta = data_decompress_limited(delta_data, MAX_RECEIVE_WHOLE_FILE_SIZE); + data_destroy(delta_data); + if (!raw_delta) { + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + /* Charge the decompressed delta to the connection budget (the paired + wire buffer's charge was just released). */ + if (!data_charge_session(raw_delta, owner, raw_delta->size)) { + data_destroy(raw_delta); + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + } + + Delta* delta = delta_deserialize(raw_delta); + data_destroy(raw_delta); + if (!delta) { + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + + uint64_t new_size = delta->new_file_size; + if (new_size > MAX_RECEIVE_WHOLE_FILE_SIZE || new_size > SIZE_MAX) { + delta_destroy(delta); + free(old_data); + delta_signature_destroy(sig); + send_status(fd, STATUS_ERROR); + *failed = true; + return NULL; + } + /* Wire-stats tally: bytes taken straight from the basis file (matched + delta blocks) and bytes shipped literally (protocol 2.28.0). Computed + before the delta is destroyed. */ + unsigned long long matched = 0; + unsigned long long literal = 0; + for (uint32_t k = 0; k < delta->instruction_count; k++) { + if (delta->instructions[k].type == DELTA_INSTR_BLOCK_MATCH) + matched += delta->instructions[k].match.length; + else if (delta->instructions[k].type == DELTA_INSTR_LITERAL) + literal += delta->instructions[k].literal.length; + } + void* new_data = delta_apply(old_data, old_size, delta, config->delta_block_size); + delta_destroy(delta); + + if (!new_data) { + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + + File* file = file_create(check_path); + if (!file) { + free(new_data); + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + file->matched_bytes = matched; + file->literal_bytes = literal; + + if (config->use_metadata) { + int meta_ok = 1; + file->metadata = metadata_receive(fd, &meta_ok); + if (!meta_ok) { + file_destroy(file); + free(new_data); + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + } + if (!receive_file_xattrs(file, fd, config)) { + file_destroy(file); + free(new_data); + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + + Data* replacement = data_create(new_data, (size_t)new_size); + if (replacement == NULL) { + file_destroy(file); + free(old_data); + delta_signature_destroy(sig); + send_status(fd, STATUS_ERROR); + *failed = true; + return NULL; + } + data_destroy(file->data); + file->data = replacement; + + free(old_data); + delta_signature_destroy(sig); + return file; + } + + if (resp == STATUS_NEXT) { + delta_signature_destroy(sig); + free(old_data); + + File* file = file_create(check_path); + if (!file) { + *failed = true; + return NULL; + } + + if (config->use_metadata) { + int meta_ok = 1; + file->metadata = metadata_receive(fd, &meta_ok); + if (!meta_ok) { + file_destroy(file); + *failed = true; + return NULL; + } + } + if (!receive_file_xattrs(file, fd, config)) { + file_destroy(file); + *failed = true; + return NULL; + } + + Data* file_data = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); + if (file_data == NULL) { + file_destroy(file); + *failed = true; + return NULL; + } + + if (config->use_compression && + !compression_should_skip_with_suffixes( + file->path, config->skip_compress_suffixes, + config->skip_compress_set ? config->skip_compress_count : -1)) { + Data* uncompressed = data_decompress_limited(file_data, MAX_RECEIVE_WHOLE_FILE_SIZE); + ProtocolSession* owner = file_data->owner; + data_destroy(file_data); + if (uncompressed == NULL) { + file_destroy(file); + *failed = true; + return NULL; + } + if (!data_charge_session(uncompressed, owner, uncompressed->size)) { + data_destroy(uncompressed); + file_destroy(file); + send_status(fd, STATUS_ERROR); + *failed = true; + return NULL; + } + if (uncompressed->size > MAX_FILE_DATA_SIZE) { + data_destroy(uncompressed); + file_destroy(file); + send_status(fd, STATUS_ERROR); + *failed = true; + return NULL; + } + file_data = uncompressed; + } + + data_destroy(file->data); + file->data = file_data; + return file; + } + + delta_signature_destroy(sig); + free(old_data); + send_status(fd, STATUS_ERROR); + *failed = true; + return NULL; +} + +/* ---- Alternate basis directories (--compare-dest / --copy-dest / --link-dest) ---- + * The receiver consults the ordered basis-dir list only when the destination + * entry is NOT already up to date. By default an "exact match" is rsync's + * metadata quick-check: an equal size and an equal mtime (unless --size-only). + * The FastSync-only --verify-basis additionally requires an equal whole-file + * content digest, so a hard link / local copy is only then made from + * byte-verified content. */ + +typedef struct BasisMatch { + bool hit; + BasisDestType type; + char* basis_path; /* owned absolute path of the matched basis file */ + struct stat st; /* fstat() of the matched basis file */ +} BasisMatch; + +static void basis_match_free(BasisMatch* match) { + if (!match) + return; + free(match->basis_path); + match->basis_path = NULL; + match->hit = false; + match->type = BASIS_DEST_NONE; +} + +/* Open `path` (via the secure, root-confined primitives) and require it to be + a regular file of exactly `expected_size` bytes. Returns an open read-only + descriptor and its fstat on success. */ +static bool basis_open_regular(const char* path, unsigned long long expected_size, int* out_fd, + struct stat* out_st) { + char* leaf = NULL; + int parent_fd = file_open_secure_parent(path, &leaf, false); + if (parent_fd < 0) + return false; + /* O_NONBLOCK: a client-planted FIFO must not block the receiver's openat() + forever; the fstat()/S_ISREG gate below rejects it immediately. */ + int fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); + free(leaf); + close(parent_fd); + if (fd < 0) + return false; + struct stat st; + if (fstat(fd, &st) != 0 || !S_ISREG(st.st_mode) || + (unsigned long long)st.st_size != expected_size) { + close(fd); + return false; + } + *out_fd = fd; + *out_st = st; + return true; +} + +/* --ignore-times forces every file to be updated, so no basis hit is ever + declared (matching rsync, where -I prevents link-dest from linking). */ +bool file_basis_quick_match(const Config* config, const struct stat* st, time_t check_mtime, + long check_mtime_nsec) { + if (config->size_only) + return true; + long mtime_nsec = 0; +#ifdef __linux__ + mtime_nsec = st->st_mtim.tv_nsec; +#endif + return metadata_mtime_matches(st->st_mtime, mtime_nsec, check_mtime, check_mtime_nsec, + config->modify_window); +} + +/* True when a basis hit must be confirmed by a whole-file content digest + (--verify-basis). False is the rsync-parity default: the metadata + quick-check alone decides a hit. */ +bool file_basis_content_required(const Config* config) { + return config != NULL && config->verify_basis; +} + +/* Probe one candidate basis file: open it (confined, O_NOFOLLOW) and apply + rsync's metadata quick-check; under --verify-basis also hash its bytes and + require the sender's digest. On a hit record `candidate` in `out` and return + true. The caller retains ownership of `candidate`. */ +static bool basis_match_probe(const Config* config, const char* candidate, + unsigned long long check_size, time_t check_mtime, + long check_mtime_nsec, const uint8_t* check_digest, + size_t check_digest_len, BasisDestType type, BasisMatch* out) { + int fd; + struct stat st; + if (!basis_open_regular(candidate, check_size, &fd, &st)) + return false; + bool hit = false; + if (file_basis_quick_match(config, &st, check_mtime, check_mtime_nsec)) { + hit = true; + if (file_basis_content_required(config)) { + uint8_t basis_digest[CHECKSUM_MAX_DIGEST_LEN]; + size_t basis_len = 0; + bool hashed = checksum_digest_fd((ChecksumAlgo)config->checksum_algo, config->checksum_seed, + fd, basis_digest, sizeof(basis_digest), &basis_len); + hit = hashed && basis_len == check_digest_len && check_digest_len > 0 && + memcmp(basis_digest, check_digest, check_digest_len) == 0; + } + } + close(fd); + if (!hit) + return false; + char* owned = str_dup(candidate); + if (!owned) + return false; + out->hit = true; + out->type = type; + out->basis_path = owned; + out->st = st; + return true; +} + +/* Search the basis-dir list in command-line order and return the first match. + By default (no --verify-basis) rsync's metadata quick-check is sufficient: + basis_open_regular has already required an equal size, and + file_basis_quick_match applies rsync's mtime (or --size-only) rule. + --verify-basis additionally requires the basis bytes' whole-file digest to + equal the sender's, restoring FastSync's historical content equality; that + digest is computed by streaming the open basis descriptor, so an arbitrarily + large basis is verified without buffering it. A copy/link install re-reads + the basis from its path in bounded buffers, so no content buffer is kept. + + `hash_content` gates content READS under --verify-basis: a server-contacting + --dry-run passes false because hashing a basis against a client-supplied + digest would be a 1-bit content oracle. Without --verify-basis a dry-run can + still confirm the metadata-only hit without reading any basis bytes, matching + rsync's read-only quick-check. + + Path resolution (rsync 3.4.1 parity): rsync resolves a relative + --compare-dest/--copy-dest/--link-dest DIR against the destination directory + (the receiver's cwd) and appends the file's TRANSFER-RELATIVE name, e.g. + `--compare-dest=basis` with `rsync src/ dst/` probes `dst/basis/`. + FastSync's receive root IS the destination directory, but its default transfer + mirrors the absolute source path below that root, so check_path carries the + source-root scaffolding rsync would not append. Recover rsync's spelling with + utils_strip_transfer_root for a relative DIR; under -R/--files-from the wire + path is already transfer-relative, so it is used as-is. A relative DIR also + probes the historical mirror-appended spelling as a fallback, so existing + FastSync-laid-out snapshot trees keep resolving. An absolute DIR is used + verbatim and keeps appending the destination-relative check_path (FastSync's + mirrored layout). Every candidate stays confined to the authorized root by + file_open_secure_parent. */ +static bool basis_match_find(const Config* config, const char* check_path, + unsigned long long check_size, time_t check_mtime, + long check_mtime_nsec, const uint8_t* check_digest, + size_t check_digest_len, bool hash_content, BasisMatch* out) { + memset(out, 0, sizeof(*out)); + if (!config || !config_has_basis(config) || config->ignore_times) + return false; + /* --verify-basis needs the basis content; a content-blind (dry-run) pass can + never confirm it and must not read the file, so decline without touching + the basis bytes. */ + if (file_basis_content_required(config) && !hash_content) + return false; + const char* transfer_rel = check_path; + if (!config->relative && config->files_from_set == NULL) + transfer_rel = utils_strip_transfer_root(check_path, config->send_directory); + for (int i = 0; i < config->basis_count; i++) { + const BasisDest* entry = &config->basis_dirs[i]; + /* An absolute basis path is used verbatim (rsync semantics); a relative one + is resolved below the receive root. Both remain subject to the receiver's + authorized-root confinement inside file_open_secure_parent. */ + bool absolute = entry->path[0] == '/'; + char* basis_dir = + absolute ? str_dup(entry->path) : path_cat(config->receive_root_directory, entry->path); + if (!basis_dir) + continue; + const char* names[2]; + int name_count = 0; + if (absolute) + names[name_count++] = check_path; + else + names[name_count++] = transfer_rel; + if (!absolute && strcmp(transfer_rel, check_path) != 0) + names[name_count++] = check_path; /* historical mirror-appended spelling */ + bool found = false; + for (int n = 0; n < name_count && !found; n++) { + char* candidate = path_cat(basis_dir, names[n]); + if (!candidate) + continue; + found = basis_match_probe(config, candidate, check_size, check_mtime, check_mtime_nsec, + check_digest, check_digest_len, entry->type, out); + free(candidate); + } + free(basis_dir); + if (found) + return true; + } + return false; +} + +/* --------------------------------------------------------------------------- + * -y/--fuzzy similar-file delta basis. + * + * When a file must be transferred and the destination holds no usable content + * at the exact path (the destination file is absent, or is outside the delta + * engine's size bounds), --fuzzy lets the receiver reuse an EXISTING regular + * file in the SAME destination directory as the delta basis, so the sender + * transmits only the differences instead of the whole file. This is the + * rsync "find a similar file to use as a basis for a transfer" case (e.g. a + * file recreated under a new name whose old-named sibling is still present). + * + * The delta handshake is unchanged and receiver-driven, so the sender never + * learns the basis was a different file and needs no new protocol. Byte + * exactness never depends on which bytes the basis holds: the delta protocol + * only references basis blocks whose Adler-32 + xxHash32 checksums match the + * source, delta_apply validates every reference against the basis size, and a + * basis that shares nothing simply makes the sender reply STATUS_NEXT (full + * transfer). A fuzzy basis can therefore waste bandwidth but never corrupt a + * file. + * + * Similarity heuristic (rsync 3.4.1 parity, util1.c fuzzy_distance / + * find_filename_suffix + generator.c find_fuzzy): + * * candidates are the target's sibling entries in its destination + * directory, opened through the confined root (file_open_secure_parent + + * openat O_NOFOLLOW, fstatat AT_SYMLINK_NOFOLLOW) -- symlinks are never + * followed and nothing outside the destination root is ever read; + * * dotfiles, directories, the target's own name, and the .fastsync-stage / + * temp scratch names are never candidates; + * * size gate = rsync's, NOT the ordinary delta engine's bounds: any + * non-empty regular sibling up to the receiver's whole-file buffer cap is + * eligible, regardless of the 16 KiB delta minimum or the 10x delta size + * ratio (rsync's find_fuzzy has no delta-size gate at all). The delta + * engine consumes the fuzzy basis through the same signature handshake + * whether or not it is inside delta_should_attempt's window; + * * first pass = an exact size+mtime match wins regardless of name (rsync's + * "fuzzy size/modtime match"); + * * otherwise the winner minimizes rsync's weighted Levenshtein distance + * (substitution ± byte difference, insertion UNIT+byte, 16.16 fixed point) + * plus ten times the suffix distance, accepted only when <= 25*UNIT; the + * tie-break (smallest size gap, then lexical name) keeps the result + * deterministic across filesystem readdir order (rsync leaves equal + * distances to its file-list order). + * ------------------------------------------------------------------------- */ + +/* A directory scan is linear in the number of entries; the fuzzy search stops + * after this many so a pathological huge directory cannot stall a transfer. + * The cap bounds the readdir() ITERATIONS, not the per-entry work: every + * entry that survives the (cheap) size and pre-name gates still runs an + * edit-distance DP, so the per-entry DP cost is separately bounded below by + * pre-pruning on the name length gap and the absent-character bound, and by + * trimming the common prefix/suffix before the DP runs on the middles only. */ +#define FUZZY_MAX_DIRECTORY_SCAN 4096 +/* Names longer than this never take part in fuzzy matching: the edit-distance + * DP below is O(len^2), so over-long names are bounded out of the search. */ +#define FUZZY_NAME_LIMIT 192 + +typedef struct { + char name[FUZZY_NAME_LIMIT + 1]; + unsigned long long size; + uint32_t distance; + unsigned long long size_gap; +} FuzzyCandidate; + +/* rsync's fuzzy distance is a weighted Levenshtein variant in 16.16 fixed point + * (util1.c fuzzy_distance): a substitution costs UNIT +/- the byte difference + * and an insertion costs UNIT + the inserted byte, so similar names score low. + * The search keeps only distances <= 25*UNIT. Ported verbatim for parity. */ +#define FUZZY_DIST_UNIT (1u << 16) +#define FUZZY_DIST_REJECT (0xFFFFu * FUZZY_DIST_UNIT + 1) +#define FUZZY_DIST_LIMIT (25u * FUZZY_DIST_UNIT) + +static uint32_t fuzzy_distance(const char* s1, unsigned len1, const char* s2, unsigned len2, + uint32_t upperlimit, uint32_t* scratch) { + if ((len1 > len2 ? len1 - len2 : len2 - len1) * FUZZY_DIST_UNIT > upperlimit) + return FUZZY_DIST_REJECT; + if (!len1 || !len2) { + if (!len1) { + s1 = s2; + len1 = len2; + } + uint32_t cost = 0; + for (unsigned i = 0; i < len1; i++) + cost += (uint8_t)s1[i]; + return (uint32_t)len1 * FUZZY_DIST_UNIT + cost; + } + uint32_t* a = scratch; + for (unsigned i2 = 0; i2 < len2; i2++) + a[i2] = (i2 + 1) * FUZZY_DIST_UNIT; + for (unsigned i1 = 0; i1 < len1; i1++) { + uint32_t diag = i1 * FUZZY_DIST_UNIT; + uint32_t above = (i1 + 1) * FUZZY_DIST_UNIT; + for (unsigned i2 = 0; i2 < len2; i2++) { + uint32_t left = a[i2]; + int32_t cost = (int32_t)(uint8_t)s1[i1] - (int32_t)(uint8_t)s2[i2]; + if (cost != 0) + cost = cost < 0 ? (int32_t)(FUZZY_DIST_UNIT - (uint32_t)(-cost)) + : (int32_t)(FUZZY_DIST_UNIT + (uint32_t)cost); + uint32_t diag_inc = diag + (uint32_t)cost; + uint32_t left_inc = left + FUZZY_DIST_UNIT + (uint8_t)s1[i1]; + uint32_t above_inc = above + FUZZY_DIST_UNIT + (uint8_t)s2[i2]; + a[i2] = above = left < above ? (left_inc < diag_inc ? left_inc : diag_inc) + : (above_inc < diag_inc ? above_inc : diag_inc); + diag = left; + } + } + return a[len2 - 1]; +} + +/* rsync's find_filename_suffix (util1.c): return the last significant filename + * suffix (its dot included). Leading dots are not a suffix; a trailing "~" is + * ignored; .bak/.old/.orig and a "~/" backup marker are skipped. */ +static const char* fuzzy_find_suffix(const char* fn, int fn_len, int* len_ptr) { + const char* suf; + const char* s; + bool had_tilde; + + while (fn_len && *fn == '.') { + fn++; + fn_len--; + } + if (fn_len > 1 && fn[fn_len - 1] == '~') { + fn_len--; + had_tilde = true; + } else { + had_tilde = false; + } + suf = ""; + *len_ptr = 0; + for (s = fn + fn_len; fn_len > 1;) { + int s_len; + while (--s != fn && *s != '.') { + } + if (s == fn) + break; + s_len = fn_len - (int)(s - fn); + fn_len = (int)(s - fn); + if (s_len == 4) { + if (strcmp(s + 1, "bak") == 0 || strcmp(s + 1, "old") == 0) + continue; + } else if (s_len == 5) { + if (strcmp(s + 1, "orig") == 0) + continue; + } else if (s_len > 2 && had_tilde && s[1] == '~' && isdigit((unsigned char)s[2])) { + continue; + } + *len_ptr = s_len; + suf = s; + if (s_len == 1) + break; + for (s++, s_len--; s_len > 0; s++, s_len--) { + if (!isdigit((unsigned char)*s)) + return suf; + } + s = suf; + } + return suf; +} + +/* Deterministic ordering of two fuzzy candidates with equal rsync distance: + * smallest size gap, then the lexical basename (rsync itself takes the last + * equal-distance candidate in file-list order). */ +static bool fuzzy_candidate_better(const FuzzyCandidate* cand, const FuzzyCandidate* best) { + if (!best->name[0]) + return true; + if (cand->distance != best->distance) + return cand->distance < best->distance; + if (cand->size_gap != best->size_gap) + return cand->size_gap < best->size_gap; + return strcmp(cand->name, best->name) < 0; +} + +/* Search the destination directory that will contain `check_path` for a + * similar regular file usable as a --fuzzy delta basis and return its full + * content in a malloc'd (protocol_alloc) buffer. Returns NULL (with *out_size + * = 0) when no candidate qualifies, which means the caller performs the normal + * whole-file transfer. */ +static void* fuzzy_basis_find_and_load(const Config* config, const char* check_path, + unsigned long long check_size, time_t check_mtime, + long check_mtime_nsec, unsigned long long* out_size) { + *out_size = 0; + if (!config || !config->receive_root_directory || !config->fuzzy || !config->use_delta || + !check_path || check_size > MAX_RECEIVE_WHOLE_FILE_SIZE) + return NULL; + + char* full_path = path_cat(config->receive_root_directory, check_path); + if (!full_path) + return NULL; + char* leaf = NULL; + int dir_fd = file_open_secure_parent(full_path, &leaf, false); + if (dir_fd < 0 || !leaf) { + free(leaf); + free(full_path); + return NULL; + } + size_t target_len = strlen(leaf); + /* A target basename longer than FUZZY_NAME_LIMIT can never pass the name gate + (every candidate name is bounded by the same limit), so skip the scan. */ + if (target_len > FUZZY_NAME_LIMIT) { + close(dir_fd); + free(leaf); + free(full_path); + return NULL; + } + + int scanfd = dup(dir_fd); + if (scanfd < 0) { + close(dir_fd); + free(leaf); + free(full_path); + return NULL; + } + DIR* dir = fdopendir(scanfd); + if (!dir) { + close(scanfd); + close(dir_fd); + free(leaf); + free(full_path); + return NULL; + } + + /* The weighted-distance scratch row is allocated once per scan (not once per + candidate). */ + uint32_t* dist_scratch = malloc((FUZZY_NAME_LIMIT + 1) * sizeof(uint32_t)); + if (!dist_scratch) { + closedir(dir); + close(dir_fd); + free(leaf); + free(full_path); + return NULL; + } + int fname_suf_len = 0; + const char* fname_suf = fuzzy_find_suffix(leaf, (int)target_len, &fname_suf_len); + + FuzzyCandidate best; + memset(&best, 0, sizeof(best)); + uint32_t lowest_dist = FUZZY_DIST_LIMIT; + /* rsync's fuzzy search runs an exact size+mtime pass before the name-distance + pass; such a candidate is almost certainly the same content and wins + regardless of how dissimilar its name is. The first one (directory order, + deterministic) is kept. */ + FuzzyCandidate exact; + memset(&exact, 0, sizeof(exact)); + const struct dirent* entry; + size_t scanned = 0; + /* readdir() yields entries in filesystem-dependent order, so the SET of + candidates seen is order-dependent; the winner is still deterministic + because every candidate is compared with the total ordering in + fuzzy_candidate_better (acceptable for a heuristic). */ + while (scanned < FUZZY_MAX_DIRECTORY_SCAN && (entry = readdir(dir)) != NULL) { + scanned++; + const char* name = entry->d_name; + size_t name_len = strlen(name); + if (name[0] == '.' || name_len == 0 || name_len > FUZZY_NAME_LIMIT || strcmp(name, leaf) == 0) + continue; + struct stat st; + if (fstatat(dir_fd, name, &st, AT_SYMLINK_NOFOLLOW) != 0 || !S_ISREG(st.st_mode)) + continue; + unsigned long long cand_size = (unsigned long long)st.st_size; + if (cand_size == 0 || cand_size > MAX_RECEIVE_WHOLE_FILE_SIZE) + continue; + long cand_nsec = 0; +#ifdef __linux__ + cand_nsec = st.st_mtim.tv_nsec; +#endif + if (!exact.name[0] && cand_size == check_size && + metadata_mtime_matches(st.st_mtime, cand_nsec, check_mtime, check_mtime_nsec, + config->modify_window)) { + memcpy(exact.name, name, name_len + 1); + exact.size = cand_size; + exact.size_gap = 0; + continue; + } + /* rsync's name-distance pass: a weighted Levenshtein distance over the full + basenames, plus ten times the same distance over the filename suffixes, + accepted only when it does not exceed the running lowest distance. */ + int name_suf_len = 0; + const char* name_suf = fuzzy_find_suffix(name, (int)name_len, &name_suf_len); + uint32_t distance = fuzzy_distance(name, (unsigned)name_len, leaf, (unsigned)target_len, + lowest_dist, dist_scratch); + if (distance < 0xFFFF0000U) + distance += fuzzy_distance(name_suf, (unsigned)name_suf_len, fname_suf, + (unsigned)fname_suf_len, 0xFFFF0000U, dist_scratch) * + 10; + if (distance > lowest_dist) + continue; + lowest_dist = distance; + FuzzyCandidate cand; + memcpy(cand.name, name, name_len + 1); + cand.size = cand_size; + cand.distance = distance; + cand.size_gap = cand_size > check_size ? cand_size - check_size : check_size - cand_size; + if (fuzzy_candidate_better(&cand, &best)) + best = cand; + } + closedir(dir); + free(leaf); + free(dist_scratch); + + /* Prefer the exact size+mtime candidate over any name-distance winner. */ + if (exact.name[0]) + best = exact; + + void* basis = NULL; + if (best.name[0]) { + /* O_NONBLOCK: a name raced to a FIFO between the fstatat gate and this open + would otherwise block the receive thread forever on open(2); with it the + open fails (ENXIO) and the fstat/S_ISREG gate below would reject it too. + A regular file opened with O_NONBLOCK is unaffected. */ + int fd = openat(dir_fd, best.name, O_RDONLY | O_NOFOLLOW | O_NONBLOCK | O_CLOEXEC); + if (fd >= 0) { + struct stat st; + if (fstat(fd, &st) == 0 && S_ISREG(st.st_mode) && + (unsigned long long)st.st_size == best.size && best.size <= SIZE_MAX) { + basis = protocol_alloc((size_t)best.size); + if (basis) { + size_t got = 0; + while (got < (size_t)best.size) { + ssize_t n = read(fd, (char*)basis + got, (size_t)best.size - got); + if (n <= 0) { + free(basis); + basis = NULL; + break; + } + got += (size_t)n; + } + } + } + close(fd); + } + } + close(dir_fd); + free(full_path); + if (basis) + *out_size = best.size; + return basis; +} + +/* Read the remainder of a full-file transfer after the receiver has already + * sent STATUS_NEXT: receive the metadata frame (when enabled) followed by the + * data frame, and return an owned File. Shared by the plain full-transfer path + * and the --append-verify prefix-mismatch fallback (a clean full transfer + * instead of a corrupt prefix+tail blend). */ +static File* receive_full_file(int fd, const Config* config, const char* path) { + File* file = file_create(path); + if (!file) + return NULL; + if (config->use_metadata) { + int meta_ok = 1; + file->metadata = metadata_receive(fd, &meta_ok); + if (!meta_ok) { + file_destroy(file); + return NULL; + } + } + if (!receive_file_xattrs(file, fd, config)) { + file_destroy(file); + return NULL; + } + Data* file_data = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); + if (file_data == NULL) { + file_destroy(file); + return NULL; + } + if (config->use_compression && + !compression_should_skip_with_suffixes(file->path, config->skip_compress_suffixes, + config->skip_compress_set ? config->skip_compress_count + : -1)) { + Data* uncompressed = data_decompress_limited(file_data, MAX_RECEIVE_WHOLE_FILE_SIZE); + ProtocolSession* owner = file_data->owner; + data_destroy(file_data); + if (uncompressed == NULL) { + file_destroy(file); + return NULL; + } + if (!data_charge_session(uncompressed, owner, uncompressed->size)) { + data_destroy(uncompressed); + file_destroy(file); + return NULL; + } + if (uncompressed->size > MAX_FILE_DATA_SIZE) { + data_destroy(uncompressed); + file_destroy(file); + return NULL; + } + file_data = uncompressed; + } + data_destroy(file->data); + file->data = file_data; + return file; +} + +/* --------------------------------------------------------------------------- + * receive_incremental_check() decomposition. + * + * The per-file STATUS_CHECK fast path is split into the small helpers below, + * called in order by a short linear orchestrator (receive_incremental_check_ex). + * Each helper owns one decision: request validation, secure destination open, + * metadata-only skip, server-contacting --dry-run no-mutation short-circuit, + * alternate-basis match, --append tail resume, block delta, --fuzzy basis, and + * the final "send the whole file" fallback. Every protocol send/receive and + * every resource cleanup is preserved exactly; the non-dry-run wire is + * byte-for-byte unchanged. receive_incremental_check_ex additionally exposes a + * `would_transfer` out-param for the dry-run caller; the 3-arg + * receive_incremental_check wrapper passes NULL. + * ------------------------------------------------------------------------- */ + +/* Owned state threaded through the helpers below. */ +typedef struct { + int fd; + const Config* config; + char* check_path; /* received destination-relative path */ + char* full_path; /* receive-root-prefixed destination path */ + unsigned long long check_size; + long long check_mtime; + long long check_mtime_nsec; + uint8_t check_digest[CHECKSUM_MAX_DIGEST_LEN]; + size_t check_digest_len; + /* Source metadata carried alongside the check frame whenever a basis dir is + configured (rsync keeps the whole file list; FastSync's sender-driven + incremental path otherwise never transmits metadata for a SKIPPED file). + A basis materialization applies these SOURCE attributes instead of the + basis inode's, matching rsync's "copy then fix attributes". */ + FileMetadata* source_metadata; + bool dest_exists; /* any destination entry exists (lstat succeeded) */ + bool has_old_file; + int old_fd; + struct stat old_st; + unsigned long long old_size; + void* old_data; /* snapshot of the existing destination, or NULL */ +} IncrementalCheckState; + +typedef enum { + INCREMENTAL_CONTINUE, /* proceed to the next helper */ + INCREMENTAL_ERROR, /* protocol/validation failure: return NULL */ + INCREMENTAL_SKIP, /* up to date: *skipped = true, return NULL */ + INCREMENTAL_DRY_RUN, /* --dry-run resolved: flags set, return NULL */ + INCREMENTAL_FILE, /* a File* was produced (out_file) */ +} IncrementalCheckOutcome; + +static void incremental_check_state_init(IncrementalCheckState* state, int fd, + const Config* config) { + memset(state, 0, sizeof(*state)); + state->fd = fd; + state->config = config; + state->old_fd = -1; +} + +/* Release every resource the helpers may have acquired. Idempotent, so it is + safe on every exit path exactly the way the original inline cleanup was. */ +static void incremental_check_state_cleanup(IncrementalCheckState* state) { + free(state->old_data); + state->old_data = NULL; + if (state->old_fd >= 0) + close(state->old_fd); + state->old_fd = -1; + file_metadata_destroy(state->source_metadata); + state->source_metadata = NULL; + free(state->full_path); + state->full_path = NULL; + free(state->check_path); + state->check_path = NULL; +} + +/* Receive and validate the STATUS_CHECK request frame: path, size, mtime, + nanosecond mtime, and (when negotiated) the source digest. */ +static IncrementalCheckOutcome incremental_check_receive_request(IncrementalCheckState* state) { + int fd = state->fd; + const Config* config = state->config; + char* check_path = receive_wire_str(fd); + if (check_path == NULL) + return INCREMENTAL_ERROR; + state->check_path = check_path; + + if (!receive_n_data(fd, &state->check_size, sizeof(state->check_size)) || + !receive_n_data(fd, &state->check_mtime, sizeof(state->check_mtime))) + return INCREMENTAL_ERROR; + if (!receive_n_data(fd, &state->check_mtime_nsec, sizeof(state->check_mtime_nsec)) || + state->check_mtime_nsec < 0 || state->check_mtime_nsec >= 1000000000LL) { + send_error_detail(fd, "invalid check mtime nanoseconds"); + return INCREMENTAL_ERROR; + } + if ((config->checksum || config->verify_basis)) { + uint8_t wire_len; + if (!receive_n_data(fd, &wire_len, sizeof(wire_len)) || wire_len == 0 || + wire_len > CHECKSUM_MAX_DIGEST_LEN || + wire_len != checksum_digest_len((ChecksumAlgo)config->checksum_algo)) { + send_error_detail(fd, "invalid check digest length"); + return INCREMENTAL_ERROR; + } + state->check_digest_len = wire_len; + if (!receive_n_data(fd, state->check_digest, state->check_digest_len)) + return INCREMENTAL_ERROR; + } + /* The sender transmits the source metadata with every basis-configured check + so a basis hit can be materialized with the SOURCE's attributes (rsync + copies/copies-then-fixes; the receiver would otherwise only have the basis + inode's stat). The block is symmetric and consumed unconditionally here, + whether or not this file ends up as a basis hit. */ + if (config_has_basis(config) && config->use_metadata) { + int meta_ok = 1; + state->source_metadata = metadata_receive(fd, &meta_ok); + if (!meta_ok) + return INCREMENTAL_ERROR; + } + + /* A basis-configured run may materialize a file larger than the whole-file + payload bound: a basis hit is streamed from the basis path (bounded + buffers), so the check size is not itself an allocation. Every other + path (delta/append/full) still applies MAX_RECEIVE_WHOLE_FILE_SIZE, and a + miss simply falls through to the normal transfer with its own bound. */ + if (!config_has_basis(config) && state->check_size > MAX_RECEIVE_WHOLE_FILE_SIZE) { + send_error_detail(fd, "check size exceeds receiver limit"); + return INCREMENTAL_ERROR; + } + + if (check_path[0] == '\0' || has_path_traversal(check_path)) { + char* escaped_path = output_escape(check_path, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "Invalid received check path: %s", + escaped_path ? escaped_path : ""); + free(escaped_path); + return INCREMENTAL_ERROR; + } + return INCREMENTAL_CONTINUE; +} + +/* Open the existing destination entry once, confined below the receive root, + and record its stat. */ +static IncrementalCheckOutcome incremental_check_open_destination(IncrementalCheckState* state) { + char* full_path = path_cat(state->config->receive_root_directory, state->check_path); + if (!full_path) { + send_error_detail(state->fd, "could not build destination path"); + return INCREMENTAL_ERROR; + } + state->full_path = full_path; + + char* leaf = NULL; + int parent_fd = file_open_secure_parent(full_path, &leaf, false); + if (parent_fd >= 0) { + struct stat dest_st; + if (fstatat(parent_fd, leaf, &dest_st, AT_SYMLINK_NOFOLLOW) == 0) + state->dest_exists = true; + /* O_NONBLOCK: an existing FIFO at the destination must not block this + openat(); the S_ISREG gate below rejects the non-regular entry. */ + state->old_fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); + free(leaf); + close(parent_fd); + state->has_old_file = state->old_fd >= 0 && fstat(state->old_fd, &state->old_st) == 0 && + S_ISREG(state->old_st.st_mode); + } + if (!state->has_old_file && state->old_fd >= 0) { + close(state->old_fd); + state->old_fd = -1; + } + state->old_size = state->has_old_file ? (unsigned long long)state->old_st.st_size : 0; + return INCREMENTAL_CONTINUE; +} + +/* Output parity (protocol 2.23.0): when the wire config asked for it, report a + snapshot of the pre-transfer destination entry BEFORE the ordinary verdict so + the sender can render rsync-accurate -i/--out-format columns. A missing + destination is reported explicitly (existed=false) rather than omitted, so + the sender can distinguish "new" from "unknown". */ +static IncrementalCheckOutcome incremental_check_report_dest_info(IncrementalCheckState* state) { + if (!state->config->report_dest_info) + return INCREMENTAL_CONTINUE; + OutputDestState info; + memset(&info, 0, sizeof(info)); + info.known = true; + info.existed = state->has_old_file; + if (state->has_old_file) { + info.size = (unsigned long long)state->old_st.st_size; + info.mtime_sec = (long long)state->old_st.st_mtime; +#ifdef __linux__ + info.mtime_nsec = state->old_st.st_mtim.tv_nsec; +#endif + info.mode = (uint32_t)state->old_st.st_mode; + info.uid = (int32_t)state->old_st.st_uid; + info.gid = (int32_t)state->old_st.st_gid; + } + if (!send_status(state->fd, STATUS_DEST_INFO) || !format_dest_state_send(state->fd, &info)) + return INCREMENTAL_ERROR; + return INCREMENTAL_CONTINUE; +} + +/* --ignore-existing short-circuit. The receiver must answer "skip" (STATUS_OK) + BEFORE the sender transmits any payload, otherwise the whole file crosses the + wire only to be discarded at write time. rsync skips an existing destination + entry regardless of its content or type, so the reply depends only on the + lstat existence probe; the ordinary --ignore-existing checks inside + file_receive remain as defense-in-depth for the frame types that have no + per-file check (directories/symlinks/specials/hard-links). */ +static IncrementalCheckOutcome +incremental_check_ignore_existing(const IncrementalCheckState* state) { + if (!state->config->ignore_existing || !state->dest_exists) + return INCREMENTAL_CONTINUE; + if (!send_status(state->fd, STATUS_OK)) + return INCREMENTAL_ERROR; + return INCREMENTAL_SKIP; +} + +/* Metadata for a materialized basis hit: prefer the SOURCE metadata the sender + transmitted with the check frame (rsync copies then fixes the destination to + the source's attributes); fall back to the basis inode's own stat when + metadata was not negotiated. Consumes state->source_metadata on success. */ +static FileMetadata* basis_take_metadata(IncrementalCheckState* state, + const struct stat* basis_st) { + if (state->source_metadata) { + FileMetadata* meta = state->source_metadata; + state->source_metadata = NULL; + return meta; + } + return file_metadata_create(NULL, basis_st, false, false); +} + +/* --link-dest relink of an already up-to-date destination. rsync hard-links a + destination entry to a matching basis even when the entry is already correct, + so a run over an existing tree still maximizes sharing with the basis. Only a + link-dest basis triggers this (copy-dest/compare-dest leave an up-to-date + destination untouched, matching rsync). The ordinary basis path further down + handles every not-up-to-date case, so this helper only adds the relink that + the quick-skip would otherwise short-circuit. */ +static IncrementalCheckOutcome incremental_check_link_dest_relink(IncrementalCheckState* state, + File** out_file) { + const Config* config = state->config; + if (!config_has_basis(config) || config->ignore_times || config->dry_run) + return INCREMENTAL_CONTINUE; + if (!state->has_old_file) + return INCREMENTAL_CONTINUE; + BasisMatch basis; + basis_match_find(config, state->check_path, state->check_size, (time_t)state->check_mtime, + (long)state->check_mtime_nsec, state->check_digest, state->check_digest_len, + true, &basis); + /* Only a link-dest hit relinks; a copy-dest/compare-dest hit (or a miss) lets + the up-to-date check below keep the existing destination. */ + if (!basis.hit || basis.type != BASIS_DEST_LINK) { + basis_match_free(&basis); + return INCREMENTAL_CONTINUE; + } + /* Already the basis inode: nothing to do, leave the destination alone. */ + if (basis.st.st_dev == state->old_st.st_dev && basis.st.st_ino == state->old_st.st_ino) { + basis_match_free(&basis); + return INCREMENTAL_CONTINUE; + } + File* materialized = file_create(state->check_path); + if (materialized) { + data_destroy(materialized->data); + materialized->data = data_create_reserve((size_t)state->check_size); + if (!materialized->data) { + file_destroy(materialized); + materialized = NULL; + } + } + if (materialized) { + materialized->metadata = basis_take_metadata(state, &basis.st); + materialized->skip = true; + materialized->basis_link = basis.basis_path; + basis.basis_path = NULL; + if (!materialized->metadata) { + file_destroy(materialized); + materialized = NULL; + } + } + if (materialized) { + if (!send_status(state->fd, STATUS_OK)) { + basis_match_free(&basis); + file_destroy(materialized); + return INCREMENTAL_ERROR; + } + basis_match_free(&basis); + *out_file = materialized; + return INCREMENTAL_FILE; + } + basis_match_free(&basis); + return INCREMENTAL_CONTINUE; +} + +/* Metadata-only (and, when --checksum forces it, content) up-to-date decision. + Loads the old contents only when a checksum comparison or delta needs them. */ +static IncrementalCheckOutcome incremental_check_quick_skip(IncrementalCheckState* state, + bool* out_try_delta) { + int fd = state->fd; + const Config* config = state->config; + bool has_old_file = state->has_old_file; + unsigned long long old_size = state->old_size; + struct stat st = state->old_st; + + bool size_equal = has_old_file && old_size == state->check_size; + bool match_by_metadata = false; + if (size_equal && !config->ignore_times && !config->size_only) { + long long old_mtime_nsec = 0; +#ifdef __linux__ + old_mtime_nsec = st.st_mtim.tv_nsec; +#endif + match_by_metadata = + metadata_mtime_matches(st.st_mtime, old_mtime_nsec, (time_t)state->check_mtime, + (long)state->check_mtime_nsec, config->modify_window); + } + + bool try_delta = config->use_delta && !config->whole_file && has_old_file && + delta_should_attempt(old_size, state->check_size, config->delta_max_file_size); + bool checksum_needs_read = size_equal && !config->ignore_times && config->checksum; + /* --dry-run must never read the destination file's CONTENTS: a client could + otherwise use `--dry-run --checksum` against a read-only module as a + 1-bit content oracle (hash match / mismatch) and force arbitrary reads. + Decide from metadata alone; when metadata is inconclusive (checksum or + delta would have required the body) report would-transfer. The real + (non-dry-run) behavior below is unchanged. */ + bool need_old_data = !config->dry_run && (checksum_needs_read || try_delta); + if (config->dry_run) + try_delta = false; + *out_try_delta = try_delta; + + if (need_old_data && has_old_file && old_size > 0 && old_size <= MAX_RECEIVE_WHOLE_FILE_SIZE && + old_size <= SIZE_MAX) { + state->old_data = protocol_alloc((size_t)old_size); + if (state->old_data) { + size_t got = 0; + while (got < (size_t)old_size) { + ssize_t n = read(state->old_fd, (char*)state->old_data + got, (size_t)old_size - got); + if (n <= 0) { + free(state->old_data); + state->old_data = NULL; + break; + } + got += (size_t)n; + } + } + } + + bool match = false; + if (config->dry_run) { + /* Metadata-only decision: a size match plus a matching mtime is treated as + up to date; --checksum/--delta cannot be verified without reading, so an + otherwise inconclusive comparison is a would-transfer. */ + match = size_equal && !config->ignore_times && (config->size_only || match_by_metadata); + } else if (checksum_needs_read) { + uint8_t old_digest[CHECKSUM_MAX_DIGEST_LEN]; + size_t old_len = 0; + bool hashed = checksum_digest((ChecksumAlgo)config->checksum_algo, config->checksum_seed, + old_size == 0 ? "" : state->old_data, (size_t)old_size, + old_digest, sizeof(old_digest), &old_len); + match = hashed && old_len == state->check_digest_len && state->check_digest_len > 0 && + memcmp(old_digest, state->check_digest, state->check_digest_len) == 0; + } else if (size_equal && !config->ignore_times) { + match = config->size_only || match_by_metadata; + } + + if (match) { + if (!send_status(fd, STATUS_OK)) + return INCREMENTAL_ERROR; + return INCREMENTAL_SKIP; + } + return INCREMENTAL_CONTINUE; +} + +/* Server-contacting --dry-run no-mutation short-circuit. Runs after the + quick-skip decision and before any path that could touch the destination. + When dry_run is set and the file is not already up to date the receiver must + materialize nothing (no basis link/copy, no append/delta/full transfer) and + the sender must send no data, so answer STATUS_DRY_RUN_TRANSFER and stop. + + The basis lookup is content-blind: under the default metadata quick-check a + hit needs no basis bytes and is honored here just as in a real run; under + --verify-basis a real run hashes the basis against the client-supplied digest, + which in a dry-run is a 1-bit content oracle, so no basis bytes may be read + and an otherwise-matching entry is reported as would-transfer. Everything + read here (the destination file's metadata, basis candidates' metadata) is + read-only. */ +static IncrementalCheckOutcome incremental_check_dry_run_shortcut(IncrementalCheckState* state, + bool* skipped, + bool* would_transfer) { + const Config* config = state->config; + if (!config->dry_run) + return INCREMENTAL_CONTINUE; + + bool skip_via_compare = false; + if (config_has_basis(config) && !config->ignore_times) { + BasisMatch basis; + /* hash_content=false: a dry-run must not read or hash the basis file, so + under --verify-basis no compare-dest hit can be confirmed and an + otherwise-matching file is reported as would-transfer. Without + --verify-basis the metadata quick-check confirms it without touching any + basis bytes. */ + basis_match_find(config, state->check_path, state->check_size, (time_t)state->check_mtime, + (long)state->check_mtime_nsec, state->check_digest, state->check_digest_len, + false, &basis); + if (basis.hit && basis.type == BASIS_DEST_COMPARE && !state->has_old_file) + skip_via_compare = true; + basis_match_free(&basis); + } + Status reply = skip_via_compare ? STATUS_OK : STATUS_DRY_RUN_TRANSFER; + if (!send_status(state->fd, reply)) + return INCREMENTAL_ERROR; + if (skip_via_compare) + *skipped = true; + else if (would_transfer) + *would_transfer = true; + return INCREMENTAL_DRY_RUN; +} + +/* Alternate basis directories (--compare-dest/--copy-dest/--link-dest): a hit + either suppresses the transfer (compare-dest) or materializes the file from + the basis without a data frame. */ +static IncrementalCheckOutcome incremental_check_try_basis(IncrementalCheckState* state, + File** out_file) { + int fd = state->fd; + const Config* config = state->config; + if (!config_has_basis(config)) + return INCREMENTAL_CONTINUE; + + BasisMatch basis; + basis_match_find(config, state->check_path, state->check_size, (time_t)state->check_mtime, + (long)state->check_mtime_nsec, state->check_digest, state->check_digest_len, + true, &basis); + if (basis.hit) { + if (basis.type == BASIS_DEST_COMPARE) { + basis_match_free(&basis); + if (!state->has_old_file) { + if (!send_status(fd, STATUS_OK)) + return INCREMENTAL_ERROR; + return INCREMENTAL_SKIP; + } + } else { + /* Copy/link installs source their bytes from the basis PATH at install + time (bounded buffers), so no whole-file content buffer is needed here + even for an over-limit basis. */ + File* materialized = file_create(state->check_path); + if (materialized) { + data_destroy(materialized->data); + materialized->data = data_create_reserve((size_t)state->check_size); + if (!materialized->data) { + file_destroy(materialized); + materialized = NULL; + } + } + if (materialized) { + materialized->metadata = basis_take_metadata(state, &basis.st); + materialized->skip = true; /* receiver must not ack this as a data file */ + if (basis.type == BASIS_DEST_LINK) + materialized->basis_link = basis.basis_path; + else + materialized->basis_copy = basis.basis_path; + basis.basis_path = NULL; + if (!materialized->metadata) { + file_destroy(materialized); + materialized = NULL; + } + } + if (materialized) { + if (!send_status(fd, STATUS_OK)) { + basis_match_free(&basis); + file_destroy(materialized); + return INCREMENTAL_ERROR; + } + basis_match_free(&basis); + *out_file = materialized; + return INCREMENTAL_FILE; + } + /* Materialization setup failed: fall through to the normal transfer. */ + } + } + basis_match_free(&basis); + return INCREMENTAL_CONTINUE; +} + +/* --append / --append-verify tail resume: when the destination is a SHORTER + file in an append mode, negotiate the resume offset and receive only the + tail. Produces the reconstructed file, or falls through to delta/full. */ +static IncrementalCheckOutcome incremental_check_try_append_resume(IncrementalCheckState* state, + File** out_file) { + int fd = state->fd; + const Config* config = state->config; + const char* check_path = state->check_path; + unsigned long long old_size = state->old_size; + unsigned long long check_size = state->check_size; + + bool append_resume = (config->append || config->append_verify) && state->has_old_file && + append_resume_eligible(old_size, check_size); + if (!append_resume) + return INCREMENTAL_CONTINUE; + + /* Ensure the retained prefix (== the whole, shorter destination file) is in + memory; it is needed both to rebuild the full file and, for + --append-verify, to checksum it. A load failure is not fatal: the resume is + simply not possible and we fall through to the other paths. */ + if (state->old_data == NULL && old_size > 0 && old_size <= MAX_RECEIVE_WHOLE_FILE_SIZE && + old_size <= SIZE_MAX) { + state->old_data = protocol_alloc((size_t)old_size); + if (state->old_data) { + size_t got = 0; + while (got < (size_t)old_size) { + ssize_t n = read(state->old_fd, (char*)state->old_data + got, (size_t)old_size - got); + if (n <= 0) { + free(state->old_data); + state->old_data = NULL; + break; + } + got += (size_t)n; + } + } + } + if (state->old_data == NULL && old_size != 0) + return INCREMENTAL_CONTINUE; + + if (!send_status(fd, STATUS_APPEND) || !send_n_data(fd, &old_size, sizeof(old_size))) + return INCREMENTAL_ERROR; + bool verify = config->append_verify; + bool full_fallback = false; + if (verify) { + Status sig_status; + if (!receive_status(fd, &sig_status)) + return INCREMENTAL_ERROR; + if (sig_status != STATUS_APPEND_SIG) { + send_status(fd, STATUS_ERROR); + return INCREMENTAL_ERROR; + } + uint64_t src_prefix_hash; + if (!receive_n_data(fd, &src_prefix_hash, sizeof(src_prefix_hash))) + return INCREMENTAL_ERROR; + /* Compare the retained prefix against the source prefix. A mismatch must + never be silently appended to: fall back to a full transfer so the result + is a byte-identical source copy. */ + uint64_t dst_prefix_hash = + old_size == 0 ? delta_xxhash64("", 0) : delta_xxhash64(state->old_data, (size_t)old_size); + if (dst_prefix_hash == src_prefix_hash) { + if (!send_status(fd, STATUS_APPEND_OK)) + return INCREMENTAL_ERROR; + } else { + if (!send_status(fd, STATUS_NEXT)) + return INCREMENTAL_ERROR; + full_fallback = true; + } + } + + if (full_fallback) { + /* Retained prefix differed: receive the sender's full transfer. */ + free(state->old_data); + state->old_data = NULL; + if (state->old_fd >= 0) { + close(state->old_fd); + state->old_fd = -1; + } + *out_file = receive_full_file(fd, config, check_path); + return INCREMENTAL_FILE; + } + + /* Receive the tail (STATUS_APPEND_DATA + metadata + tail bytes). */ + Status tail_status; + if (!receive_status(fd, &tail_status)) + return INCREMENTAL_ERROR; + if (tail_status != STATUS_APPEND_DATA) { + send_status(fd, STATUS_ERROR); + return INCREMENTAL_ERROR; + } + FileMetadata* meta = NULL; + FileXattrList* append_xattrs = NULL; + if (config->use_metadata) { + int meta_ok = 1; + meta = metadata_receive(fd, &meta_ok); + if (!meta_ok) + return INCREMENTAL_ERROR; + } + if (config->use_xattrs) { + int xok = 0; + append_xattrs = xattr_receive(fd, &xok, config->preserve_acls); + if (!xok) { + xattr_list_free(append_xattrs); + return INCREMENTAL_ERROR; + } + } + Data* tail = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); + if (tail == NULL) { + xattr_list_free(append_xattrs); + return INCREMENTAL_ERROR; + } + if (config->use_compression && + !compression_should_skip_with_suffixes(check_path, config->skip_compress_suffixes, + config->skip_compress_set ? config->skip_compress_count + : -1)) { + Data* uncompressed = data_decompress_limited(tail, MAX_RECEIVE_WHOLE_FILE_SIZE); + ProtocolSession* owner = tail->owner; + data_destroy(tail); + if (uncompressed == NULL) { + xattr_list_free(append_xattrs); + return INCREMENTAL_ERROR; + } + if (!data_charge_session(uncompressed, owner, uncompressed->size)) { + data_destroy(uncompressed); + xattr_list_free(append_xattrs); + return INCREMENTAL_ERROR; + } + if (uncompressed->size > MAX_FILE_DATA_SIZE) { + data_destroy(uncompressed); + xattr_list_free(append_xattrs); + return INCREMENTAL_ERROR; + } + tail = uncompressed; + } + /* The tail must complete the file exactly; anything else is a protocol + violation (never a truncated or overrun file). */ + unsigned long long expected_tail; + if (!append_tail_length(old_size, check_size, &expected_tail) || + tail->size != (size_t)expected_tail) { + send_status(fd, STATUS_ERROR); + data_destroy(tail); + xattr_list_free(append_xattrs); + return INCREMENTAL_ERROR; + } + size_t full_size = (size_t)check_size; + void* full = protocol_alloc(full_size ? full_size : 1); + if (!full) { + data_destroy(tail); + xattr_list_free(append_xattrs); + return INCREMENTAL_ERROR; + } + if (old_size > 0 && state->old_data) + memcpy(full, state->old_data, (size_t)old_size); + if (tail->size > 0) + memcpy((char*)full + old_size, tail->data, tail->size); + data_destroy(tail); + free(state->old_data); + state->old_data = NULL; + + File* file = file_create(check_path); + if (!file) { + free(full); + xattr_list_free(append_xattrs); + return INCREMENTAL_ERROR; + } + file->metadata = meta; + file->xattrs = append_xattrs; + append_xattrs = NULL; + data_destroy(file->data); + file->data = data_create(full, full_size); + if (!file->data) { /* data_create already freed full on failure */ + file_destroy(file); + return INCREMENTAL_ERROR; + } + *out_file = file; + return INCREMENTAL_FILE; +} + +/* Block delta transfer against the existing destination content. */ +static IncrementalCheckOutcome incremental_check_try_delta(IncrementalCheckState* state, + bool try_delta, File** out_file) { + if (try_delta && state->old_data != NULL) { + bool delta_failed = false; + File* delta_file = receive_delta_file(state->fd, state->config, state->check_path, + state->old_data, state->old_size, &delta_failed); + state->old_data = NULL; /* receive_delta_file consumes the snapshot on every path */ + if (delta_file) { + *out_file = delta_file; + return INCREMENTAL_FILE; + } + if (delta_failed) + return INCREMENTAL_ERROR; + } + free(state->old_data); + state->old_data = NULL; + return INCREMENTAL_CONTINUE; +} + +/* -y/--fuzzy similar-file delta basis. Reaching this point means the file + must be transferred and the destination's own content could not serve as a + delta basis; try an existing similar-named sibling in the same directory. */ +static IncrementalCheckOutcome incremental_check_try_fuzzy(IncrementalCheckState* state, + File** out_file) { + const Config* config = state->config; + if (!config->fuzzy || !config->use_delta) + return INCREMENTAL_CONTINUE; + unsigned long long fuzzy_size = 0; + void* fuzzy_basis = fuzzy_basis_find_and_load(config, state->check_path, state->check_size, + (time_t)state->check_mtime, + (long)state->check_mtime_nsec, &fuzzy_size); + if (fuzzy_basis != NULL) { + bool fuzzy_failed = false; + File* fuzzy_file = receive_delta_file(state->fd, config, state->check_path, fuzzy_basis, + fuzzy_size, &fuzzy_failed); + fuzzy_basis = NULL; /* receive_delta_file consumes the buffer on every path */ + if (fuzzy_file) { + *out_file = fuzzy_file; + return INCREMENTAL_FILE; + } + if (fuzzy_failed) + return INCREMENTAL_ERROR; + } + free(fuzzy_basis); + return INCREMENTAL_CONTINUE; +} + +/* Final fallback: tell the sender to transmit the whole file and receive it. */ +static File* incremental_check_receive_full(IncrementalCheckState* state) { + if (!send_status(state->fd, STATUS_NEXT)) + return NULL; + if (state->old_fd >= 0) { + close(state->old_fd); + state->old_fd = -1; + } + return receive_full_file(state->fd, state->config, state->check_path); +} + +/* Core implementation. `would_transfer` (may be NULL) is set true only on the + * server-contacting --dry-run path, when the file is not up to date and the + * receiver answered STATUS_DRY_RUN_TRANSFER; the caller then knows no File is + * returned and nothing was stored. */ +File* receive_incremental_check_ex(int fd, const Config* config, bool* skipped, + bool* would_transfer) { + if (would_transfer) + *would_transfer = false; + if (!config || !skipped) { + send_status(fd, STATUS_ERROR); + return NULL; + } + *skipped = false; + + IncrementalCheckState state; + incremental_check_state_init(&state, fd, config); + + File* result = NULL; + bool try_delta = false; + IncrementalCheckOutcome outcome; + + outcome = incremental_check_receive_request(&state); + if (outcome == INCREMENTAL_ERROR) + goto done; + + outcome = incremental_check_open_destination(&state); + if (outcome == INCREMENTAL_ERROR) + goto done; + + outcome = incremental_check_report_dest_info(&state); + if (outcome == INCREMENTAL_ERROR) + goto done; + + /* --ignore-existing must answer before any data is requested; it takes + precedence over the metadata up-to-date check below. */ + outcome = incremental_check_ignore_existing(&state); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome == INCREMENTAL_SKIP) { + *skipped = true; + goto done; + } + + /* A --link-dest hit relinks even an already up-to-date destination before the + quick-skip can suppress it (rsync parity). */ + outcome = incremental_check_link_dest_relink(&state, &result); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome == INCREMENTAL_FILE) + goto done; + + outcome = incremental_check_quick_skip(&state, &try_delta); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome == INCREMENTAL_SKIP) { + *skipped = true; + goto done; + } + + /* Dry-run resolves here (no mutation) or falls through to the normal path. */ + outcome = incremental_check_dry_run_shortcut(&state, skipped, would_transfer); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome != INCREMENTAL_CONTINUE) + goto done; + + outcome = incremental_check_try_basis(&state, &result); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome == INCREMENTAL_SKIP) { + *skipped = true; + goto done; + } + if (outcome == INCREMENTAL_FILE) + goto done; + + outcome = incremental_check_try_append_resume(&state, &result); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome == INCREMENTAL_FILE) + goto done; + + outcome = incremental_check_try_delta(&state, try_delta, &result); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome == INCREMENTAL_FILE) + goto done; + + outcome = incremental_check_try_fuzzy(&state, &result); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome == INCREMENTAL_FILE) + goto done; + + result = incremental_check_receive_full(&state); + +done: + incremental_check_state_cleanup(&state); + return result; +} + +File* receive_incremental_check(int fd, const Config* config, bool* skipped) { + return receive_incremental_check_ex(fd, config, skipped, NULL); +} diff --git a/src/shared/incremental_check.h b/src/shared/incremental_check.h new file mode 100644 index 0000000..862673d --- /dev/null +++ b/src/shared/incremental_check.h @@ -0,0 +1,41 @@ +#ifndef INCREMENTAL_CHECK_H +#define INCREMENTAL_CHECK_H + +#include "config.h" +#include "file_types.h" +#include "protocol.h" +#include + +/* Incremental-check module: the per-file STATUS_CHECK state machine, the + * incremental delta / alternate-basis / fuzzy matching helpers and the shared + * xattr receive helper. These declarations are re-exported by the + * file_receive.h facade. */ + +/* Whole-file payload bound shared by the plain receive path and the + * incremental check paths. */ +#define MAX_FILE_DATA_SIZE MAX_RECEIVE_WHOLE_FILE_SIZE + +/* Receive a file's xattr block (when the config enables xattr transport) and + * attach it to `file`. Returns false on a malformed/oversized frame. */ +bool receive_file_xattrs(File* file, int fd, const Config* config); + +File* receive_incremental_check(int fd, const Config* config, bool* skipped); +/* Extended variant used by the receiver. `would_transfer` (may be NULL) is set + * true only on the server-contacting --dry-run path when the file is not up to + * date: the receiver has already sent STATUS_DRY_RUN_TRANSFER and returns NULL + * without storing anything. On that path `*skipped` is true for an up-to-date + * (STATUS_OK) file and both flags are false for a genuine error. */ +File* receive_incremental_check_ex(int fd, const Config* config, bool* skipped, + bool* would_transfer); + +/* Testable basis quick-check / verification policy. file_basis_quick_match is + * rsync's metadata quick-check for a basis candidate (equal size is required + * separately by the caller; this adds the --size-only / mtime / --modify-window + * leg). file_basis_content_required reports whether a hit must ALSO be + * confirmed by a whole-file content digest (--verify-basis; false is the + * default rsync-parity behavior). */ +bool file_basis_quick_match(const Config* config, const struct stat* st, time_t check_mtime, + long check_mtime_nsec); +bool file_basis_content_required(const Config* config); + +#endif From 707bb659e80e4fb5c2da16b8163086b4fa6ce88e Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 13:46:17 +0200 Subject: [PATCH 04/10] refactor(server): decompose handler into phases --- src/server/server.c | 455 ++++++++++++++++++++++++++------------------ 1 file changed, 269 insertions(+), 186 deletions(-) diff --git a/src/server/server.c b/src/server/server.c index e3a0a35..5978d71 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -705,69 +705,85 @@ static const char* server_module_gate(const Config* config, void* context) { return module_gate_install_root(config, module); } -void handler(int file_descriptor) { - SSL* ssl = io_get_ssl(); +/* Per-connection state threaded through the handler phase helpers below. The + * fields are a faithful split of the former handler() locals: the protocol + * session, the config-frame gate context, the accepted config, the optional + * multithreaded pipeline context and the teardown bookkeeping all live here so + * the single `done` epilogue in handler() can release them exactly as before. */ +typedef struct ServerSession { + int fd; + SSL* ssl; ProtocolSession session; - protocol_session_init(&session, file_descriptor, file_descriptor); - protocol_session_set_ssl(&session, ssl); - protocol_session_bind(&session); ModuleGateContext gate_ctx; - gate_ctx.ssl = ssl; - gate_ctx.fd = file_descriptor; - gate_ctx.super_mode_override = -1; - gate_ctx.has_peer_ip = false; - gate_ctx.peer_ip[0] = '\0'; - gate_ctx.is_local = false; - /* All teardown state starts empty so the single `done` epilogue is safe to - * reach from any error path (including before the config frame arrives). */ - Config* config = NULL; - PipelineContextReceiver* context = NULL; - char* joined_destination = NULL; - bool charset_ready = false; - config = config_receive_with_validate(file_descriptor, server_module_gate, &gate_ctx); - if (config == NULL) { + Config* config; + PipelineContextReceiver* context; + char* joined_destination; + bool charset_ready; +} ServerSession; + +/* Phase 1 -- config receipt + validation. Receives the client config frame + * through the module gate, applies the super-mode override the gate recorded + * exactly once, and installs the per-connection protocol/compression state. + * Returns false when the config frame was refused (the gate has already + * answered the client); the caller jumps to the shared `done` epilogue. */ +static bool server_accept_config(ServerSession* state) { + state->config = config_receive_with_validate(state->fd, server_module_gate, &state->gate_ctx); + if (state->config == NULL) { log_message(LOG_LEVEL_ERROR, "Failed to receive config"); - goto done; + return false; } /* Apply the super-mode veto the gate decided on (operator --no-super, or a * daemon module without the `client owner = yes` opt-in) exactly once, so * every downstream gate (identity_apply_ownership via privilege_super_permitted, * device-node creation) sees SUPER_MODE_OFF. The gate never mutated the * received config. */ - if (gate_ctx.super_mode_override != -1) - config->super_mode = (SuperMode)gate_ctx.super_mode_override; + if (state->gate_ctx.super_mode_override != -1) + state->config->super_mode = (SuperMode)state->gate_ctx.super_mode_override; /* Install the codec this connection negotiated before the receiver/writer * threads start (the server forks per connection, so the process-global * codec is private to this session). */ - compression_set_algo((CompressionAlgo)config->compression_algo); + compression_set_algo((CompressionAlgo)state->config->compression_algo); /* If the client requested ownership but the effective super mode forbids it * (operator --no-super, a privileged standalone receiver's secure default, or * a daemon module without `client owner = yes`), say so ONCE per connection so * a successful -a/-o/-g transfer is not mistaken for preserved ownership. */ - if (config->super_mode == SUPER_MODE_OFF && identity_ownership_requested(config)) + if (state->config->super_mode == SUPER_MODE_OFF && identity_ownership_requested(state->config)) log_message(LOG_LEVEL_WARNING, "requested ownership will NOT be applied: super-user activities are disabled " "for this connection (operator veto, or module without `client owner = yes`)"); - protocol_set_8_bit_output(config->eight_bit_output); + protocol_set_8_bit_output(state->config->eight_bit_output); /* Server-side per-message protocol deadline for every frame from here on. * `timeout` is not serialized, so this is the server's own config (the server * has no --timeout CLI and defaults it to 0). A client's --timeout tightens * only that client's own protocol I/O; the server floors its own deadline at * SERVER_IO_TIMEOUT_SEC so a silent peer can never hold a session slot * forever (the socket layer gets the same floor at startup). */ - protocol_session_set_io_timeout(&session, protocol_server_io_timeout_sec(config->timeout)); + protocol_session_set_io_timeout(&state->session, + protocol_server_io_timeout_sec(state->config->timeout)); + return true; +} + +/* Phase 2 -- security gates. The ORDER here is load-bearing and must not be + * merged or reordered: transport/authentication (plaintext refusal, TLS + * client-CN verification), then daemon-root confinement (absolute-destination + * rejection, traversal + within-authorized-root), then delete/force + * authorization -- exactly the sequence the former handler() used. Returns + * false after logging the matching rejection; the caller jumps to the shared + * `done` epilogue. */ +static bool server_apply_security_gates(ServerSession* state) { + Config* config = state->config; const char* authorized_root = utils_get_authorized_root_path(); if (!authorized_root) { log_message(LOG_LEVEL_ERROR, "No server-side destination root configured"); - goto done; + return false; } - if (!allow_unauthenticated && ssl == NULL) { + if (!allow_unauthenticated && state->ssl == NULL) { log_message(LOG_LEVEL_ERROR, "Rejected unauthenticated plaintext connection"); - goto done; + return false; } - if (ssl && required_client_cn && !tls_client_identity_allowed(ssl)) { + if (state->ssl && required_client_cn && !tls_client_identity_allowed(state->ssl)) { log_message(LOG_LEVEL_ERROR, "Rejected TLS client with unauthorized identity"); - goto done; + return false; } /* Daemon mode: the module's root is the authorized root (installed by server_module_gate), and the client's destination is a MODULE-RELATIVE @@ -777,27 +793,27 @@ void handler(int file_descriptor) { if (g_daemon_conf && config->receive_root_directory && config->receive_root_directory[0] == '/') { log_message(LOG_LEVEL_ERROR, "Rejected absolute daemon destination (must be relative to the " "selected module root)"); - goto done; + return false; } char* destination = config->receive_root_directory; if (destination && destination[0] != '/') - joined_destination = path_cat(authorized_root, destination); - if (joined_destination) - destination = joined_destination; + state->joined_destination = path_cat(authorized_root, destination); + if (state->joined_destination) + destination = state->joined_destination; if (!destination || has_path_traversal(destination) || !path_is_within_root(authorized_root, destination)) { log_message(LOG_LEVEL_ERROR, "Rejected destination outside authorized root"); - free(joined_destination); - joined_destination = NULL; - goto done; + free(state->joined_destination); + state->joined_destination = NULL; + return false; } - if (joined_destination) { + if (state->joined_destination) { free(config->receive_root_directory); - config->receive_root_directory = joined_destination; - joined_destination = NULL; + config->receive_root_directory = state->joined_destination; + state->joined_destination = NULL; } if (!config->receive_root_directory) { - goto done; + return false; } config->use_delete = config->use_delete && allow_delete; /* --force (receiver-side) is deletion authority too: it lets an incoming @@ -807,6 +823,18 @@ void handler(int file_descriptor) { * --delete-missing-args, so a client cannot use --force to bypass the delete * policy. */ config->force_delete = config->force_delete && allow_delete; + return true; +} + +/* Phase 3 -- session preparation. Installs the negotiated conversion, applies + * the remaining deletion policy, materializes the destination root (--mkpath), + * creates the --delay-updates staging tree, snapshots the identity policy, and + * publishes the --keep-dirlinks/--trust-sender globals and the daemon MOTD. + * All of it must happen before any receiver/writer thread is spawned. Returns + * false after logging the matching failure; the caller jumps to the shared + * `done` epilogue. */ +static bool server_prepare_session(ServerSession* state) { + Config* config = state->config; /* --iconv (protocol 2.16.0): install the receiver-side wire->local conversion now that the client's full CONVERT_SPEC has been received and validated, before any received file name is decoded. The server's own --iconv (if @@ -817,14 +845,14 @@ void handler(int file_descriptor) { if (!charset_wire_init_receiver(config->iconv_spec, server_iconv_spec)) { log_message(LOG_LEVEL_ERROR, "--iconv: unsupported charset conversion requested (LOCAL[,REMOTE])"); - goto done; + return false; } - charset_ready = true; + state->charset_ready = true; } /* --delete-missing-args deletes destination mirrors receiver-side, so it is - deletion and stays gated by the same --allow-delete server policy. When - the server policy is off the flag is inert (the missing entries are still - skipped via its implied --ignore-missing-args, but nothing is deleted). */ + * deletion and stays gated by the same --allow-delete server policy. When + * the server policy is off the flag is inert (the missing entries are still + * skipped via its implied --ignore-missing-args, but nothing is deleted). */ config->delete_missing_args = config->delete_missing_args && allow_delete; /* --mkpath: create the destination root (and its missing leading components) * before anything else; without it the root must pre-exist. The precondition @@ -839,7 +867,7 @@ void handler(int file_descriptor) { log_message(LOG_LEVEL_ERROR, "destination root is not available: %s", escaped_root ? escaped_root : ""); free(escaped_root); - goto done; + return false; } /* A --delay-updates transfer stages under a private 0700 directory inside the receive root. Create it up front (wiping leftovers of any previously @@ -849,7 +877,7 @@ void handler(int file_descriptor) { config->delay_context = delay_updates_context_create(config->receive_root_directory); if (!config->delay_context || !delay_updates_prepare(config->delay_context)) { log_message(LOG_LEVEL_ERROR, "Failed to initialize --delay-updates staging area"); - goto done; + return false; } } /* Preserve the negotiated identity policy for the fd-relative ownership @@ -859,7 +887,7 @@ void handler(int file_descriptor) { rather than silently applying the wrong ownership policy. */ if (!identity_set_active(config)) { log_message(LOG_LEVEL_ERROR, "Failed to activate identity policy"); - goto done; + return false; } /* Persist the negotiated --keep-dirlinks policy once, here at config-accept, before any multithreaded receiver/writer threads are spawned, so the @@ -887,140 +915,195 @@ void handler(int file_descriptor) { Wave C note in config.h). */ if (g_daemon_conf) { char* motd = motd_read_file(g_daemon_conf->global.motd_file); - if (!motd_send(file_descriptor, motd ? motd : "")) { + if (!motd_send(state->fd, motd ? motd : "")) { free(motd); log_message(LOG_LEVEL_ERROR, "Failed to send daemon MOTD"); - goto done; + return false; } free(motd); } - if (config->use_multithreading) { - Queue* q = queue_create(100, file_destroy); - if (q == NULL) - goto done; - context = pipeline_context_receiver_create(config, q, file_descriptor, ssl); - if (context == NULL) { - queue_destroy(q); - goto done; - } - protocol_session_set_max_alloc(&context->session, config->max_alloc); - protocol_session_set_io_timeout(&context->session, - protocol_server_io_timeout_sec(config->timeout)); - atomic_store(&context->session.total_allocated_bytes, - atomic_load(&session.total_allocated_bytes)); - pipeline_context_receiver_set_queue_byte_limit(context, RECEIVER_QUEUE_MAX_BYTES); - thrd_t receiver = {0}; - thrd_t writer = {0}; - bool receiver_created = thrd_create(&receiver, receive_thread, context) == thrd_success; - bool writer_created = false; - if (receiver_created) - writer_created = thrd_create(&writer, write_thread, context) == thrd_success; - if (!receiver_created || !writer_created) { - log_perror("Error creating Threads"); - if (receiver_created) { - mtx_lock(&context->mutex); - atomic_store(&context->cancelled, true); - cnd_broadcast(&context->condition_not_full); - cnd_broadcast(&context->condition_not_empty); - mtx_unlock(&context->mutex); - /* Unblock a worker parked in socket I/O without closing the fd (the - * child owns the single close). shutdown() only affects sockets; for - * the --stdio pipe the receiver's per-message poll timeout still - * bounds the join, so do nothing there rather than close a descriptor - * another thread may still be using. */ - struct stat fd_stat; - if (fstat(file_descriptor, &fd_stat) == 0 && S_ISSOCK(fd_stat.st_mode)) - shutdown(file_descriptor, SHUT_RDWR); - thrd_join(receiver, NULL); - } - if (writer_created) - thrd_join(writer, NULL); - goto done; - } - int receiver_result; - int writer_result; - thrd_join(receiver, &receiver_result); - thrd_join(writer, &writer_result); - bool transfer_ok = receiver_result == thrd_success && writer_result == thrd_success; - if (transfer_ok && !config->dry_run) { - /* Commit-style (late) deletion: receive_thread handed the keep-set - manifest here instead of deleting while write_thread might still be - draining, so by now every file is on disk and the whole transfer is - known to have succeeded. Remove the extras before publishing a - --delay-updates run; the walker skips the staging directory. A - server-contacting --dry-run deletes nothing (no manifest is sent). */ - if (context->deferred_manifest) { - size_t deleted = 0; - DeletePathObserver observer = config->report_deletes ? receiver_record_deleted_path : NULL; - DeleteCommitResult deletion = manifest_delete_all_observed( - config, context->deferred_manifest, &deleted, observer, (void*)context->deleted_paths); - context->stats.deleted_files += deleted; - if (deletion == DELETE_COMMIT_ERROR) { - transfer_ok = false; - } else if (deletion == DELETE_COMMIT_LIMIT_REACHED) { - /* The transfer still succeeds; the terminal frame reports the capped - deletion so the sender exits 25 like rsync. */ - context->delete_limit_reached = true; - } - delete_manifest_free(context->deferred_manifest); - context->deferred_manifest = NULL; - } - /* --delete-delay: receive_thread snapshotted each plan's extras as it - arrived; with the disk writer drained, commit the deferred removals. - --delete-during already applied its plans on the receive thread. */ - if (context->deferred_plans) { - /* Defence in depth (the enclosing block already excludes dry-run): a - -n run never commits a deletion. */ - if (config->report_deletes) - delete_plan_session_set_delete_observer( - context->deferred_plans, receiver_record_deleted_path, (void*)context->deleted_paths); - DeleteCommitResult deletion = - config->dry_run ? DELETE_COMMIT_OK - : delete_plan_session_commit(context->deferred_plans, config); - context->stats.deleted_files += delete_plan_session_deleted(context->deferred_plans); - if (deletion == DELETE_COMMIT_ERROR) { - transfer_ok = false; - } else if (deletion == DELETE_COMMIT_LIMIT_REACHED) { - context->delete_limit_reached = true; - } - delete_plan_session_destroy(context->deferred_plans); - context->deferred_plans = NULL; - } - } - if (transfer_ok && !config->dry_run) { - /* --delay-updates: receive_thread has finished the whole protocol stream - (including manifest/delete handling) and write_thread has drained its - queue, so every staged file is complete. Publish atomically before the - success/outcome frame so a --remove-source-files sender only learns of - files that were actually installed. */ - if (config->delay_updates && config->delay_context && - !delay_updates_publish(config->delay_context, config)) { - transfer_ok = false; - } - /* P7 Wave D: all writers have joined and the late deletion (and - --delay-updates publication) has committed above, so it is finally safe - to stamp directory times; a directory's mtime must not be clobbered by - its children or by an extra removal. */ - if (transfer_ok) - dir_metadata_list_apply(&context->dir_times, config->receive_root_directory, config); - } - if (transfer_ok) { - Status final_status = context->delete_limit_reached ? STATUS_DELETE_LIMIT : STATUS_OK; - /* Emit the optional wire-stats record first (protocol 2.25.0), then the - success/outcome frame, exactly like the single-threaded receiver. */ - if (!receiver_send_stats_frame(file_descriptor, config, &context->stats, - context->would_delete, context->deleted_paths) || - !receiver_send_final_success(file_descriptor, config, &context->outcomes, final_status)) - transfer_ok = false; - } else { - send_error_detail(file_descriptor, "transfer failed on receiver"); - } - if (!transfer_ok) - log_message(LOG_LEVEL_ERROR, "Transfer failed"); - } else { - if (receiver_receive_files(config, file_descriptor) != 0) - log_message(LOG_LEVEL_ERROR, "Transfer failed"); + return true; +} + +/* Phase 4a -- transfer via the multithreaded receiver. Spawns the receive/write + * thread pair, joins them, then commits the late deletion, --delay-updates + * publication and directory times before emitting the terminal stats/success + * frame. On any failure the helper just returns; the caller's `done` epilogue + * releases the pipeline context (which owns the config and queue) exactly as the + * former inline code did. */ +static void server_run_mt_receiver(ServerSession* state) { + Config* config = state->config; + Queue* q = queue_create(100, file_destroy); + if (q == NULL) + return; + state->context = pipeline_context_receiver_create(config, q, state->fd, state->ssl); + if (state->context == NULL) { + queue_destroy(q); + return; } + protocol_session_set_max_alloc(&state->context->session, config->max_alloc); + protocol_session_set_io_timeout(&state->context->session, + protocol_server_io_timeout_sec(config->timeout)); + atomic_store(&state->context->session.total_allocated_bytes, + atomic_load(&state->session.total_allocated_bytes)); + pipeline_context_receiver_set_queue_byte_limit(state->context, RECEIVER_QUEUE_MAX_BYTES); + thrd_t receiver = {0}; + thrd_t writer = {0}; + bool receiver_created = thrd_create(&receiver, receive_thread, state->context) == thrd_success; + bool writer_created = false; + if (receiver_created) + writer_created = thrd_create(&writer, write_thread, state->context) == thrd_success; + if (!receiver_created || !writer_created) { + log_perror("Error creating Threads"); + if (receiver_created) { + mtx_lock(&state->context->mutex); + atomic_store(&state->context->cancelled, true); + cnd_broadcast(&state->context->condition_not_full); + cnd_broadcast(&state->context->condition_not_empty); + mtx_unlock(&state->context->mutex); + /* Unblock a worker parked in socket I/O without closing the fd (the + * child owns the single close). shutdown() only affects sockets; for + * the --stdio pipe the receiver's per-message poll timeout still + * bounds the join, so do nothing there rather than close a descriptor + * another thread may still be using. */ + struct stat fd_stat; + if (fstat(state->fd, &fd_stat) == 0 && S_ISSOCK(fd_stat.st_mode)) + shutdown(state->fd, SHUT_RDWR); + thrd_join(receiver, NULL); + } + if (writer_created) + thrd_join(writer, NULL); + return; + } + int receiver_result; + int writer_result; + thrd_join(receiver, &receiver_result); + thrd_join(writer, &writer_result); + bool transfer_ok = receiver_result == thrd_success && writer_result == thrd_success; + PipelineContextReceiver* context = state->context; + if (transfer_ok && !config->dry_run) { + /* Commit-style (late) deletion: receive_thread handed the keep-set + manifest here instead of deleting while write_thread might still be + draining, so by now every file is on disk and the whole transfer is + known to have succeeded. Remove the extras before publishing a + --delay-updates run; the walker skips the staging directory. A + server-contacting --dry-run deletes nothing (no manifest is sent). */ + if (context->deferred_manifest) { + size_t deleted = 0; + DeletePathObserver observer = config->report_deletes ? receiver_record_deleted_path : NULL; + DeleteCommitResult deletion = manifest_delete_all_observed( + config, context->deferred_manifest, &deleted, observer, (void*)context->deleted_paths); + context->stats.deleted_files += deleted; + if (deletion == DELETE_COMMIT_ERROR) { + transfer_ok = false; + } else if (deletion == DELETE_COMMIT_LIMIT_REACHED) { + /* The transfer still succeeds; the terminal frame reports the capped + deletion so the sender exits 25 like rsync. */ + context->delete_limit_reached = true; + } + delete_manifest_free(context->deferred_manifest); + context->deferred_manifest = NULL; + } + /* --delete-delay: receive_thread snapshotted each plan's extras as it + arrived; with the disk writer drained, commit the deferred removals. + --delete-during already applied its plans on the receive thread. */ + if (context->deferred_plans) { + /* Defence in depth (the enclosing block already excludes dry-run): a + -n run never commits a deletion. */ + if (config->report_deletes) + delete_plan_session_set_delete_observer( + context->deferred_plans, receiver_record_deleted_path, (void*)context->deleted_paths); + DeleteCommitResult deletion = + config->dry_run ? DELETE_COMMIT_OK + : delete_plan_session_commit(context->deferred_plans, config); + context->stats.deleted_files += delete_plan_session_deleted(context->deferred_plans); + if (deletion == DELETE_COMMIT_ERROR) { + transfer_ok = false; + } else if (deletion == DELETE_COMMIT_LIMIT_REACHED) { + context->delete_limit_reached = true; + } + delete_plan_session_destroy(context->deferred_plans); + context->deferred_plans = NULL; + } + } + if (transfer_ok && !config->dry_run) { + /* --delay-updates: receive_thread has finished the whole protocol stream + (including manifest/delete handling) and write_thread has drained its + queue, so every staged file is complete. Publish atomically before the + success/outcome frame so a --remove-source-files sender only learns of + files that were actually installed. */ + if (config->delay_updates && config->delay_context && + !delay_updates_publish(config->delay_context, config)) { + transfer_ok = false; + } + /* P7 Wave D: all writers have joined and the late deletion (and + --delay-updates publication) has committed above, so it is finally safe + to stamp directory times; a directory's mtime must not be clobbered by + its children or by an extra removal. */ + if (transfer_ok) + dir_metadata_list_apply(&context->dir_times, config->receive_root_directory, config); + } + if (transfer_ok) { + Status final_status = context->delete_limit_reached ? STATUS_DELETE_LIMIT : STATUS_OK; + /* Emit the optional wire-stats record first (protocol 2.25.0), then the + success/outcome frame, exactly like the single-threaded receiver. */ + if (!receiver_send_stats_frame(state->fd, config, &context->stats, context->would_delete, + context->deleted_paths) || + !receiver_send_final_success(state->fd, config, &context->outcomes, final_status)) + transfer_ok = false; + } else { + send_error_detail(state->fd, "transfer failed on receiver"); + } + if (!transfer_ok) + log_message(LOG_LEVEL_ERROR, "Transfer failed"); +} + +/* Phase 4b -- transfer via the single-threaded receiver. Failure is logged + * exactly as before; the caller's `done` epilogue then releases the config. */ +static void server_run_st_receiver(ServerSession* state) { + if (receiver_receive_files(state->config, state->fd) != 0) + log_message(LOG_LEVEL_ERROR, "Transfer failed"); +} + +/* Phase 4 dispatch -- choose the receiver implementation the config asks for. + * Both helpers own their success/failure logging; the caller falls through to + * the shared `done` epilogue either way. */ +static void server_run_transfer(ServerSession* state) { + if (state->config->use_multithreading) + server_run_mt_receiver(state); + else + server_run_st_receiver(state); +} + +void handler(int file_descriptor) { + /* Single per-connection state; every phase helper below advances it and + * returns false on a logged failure. All teardown state starts empty so the + * single `done` epilogue is safe to reach from any error path (including + * before the config frame arrives). */ + ServerSession state; + state.fd = file_descriptor; + state.ssl = io_get_ssl(); + protocol_session_init(&state.session, file_descriptor, file_descriptor); + protocol_session_set_ssl(&state.session, state.ssl); + protocol_session_bind(&state.session); + state.gate_ctx.ssl = state.ssl; + state.gate_ctx.fd = file_descriptor; + state.gate_ctx.super_mode_override = -1; + state.gate_ctx.has_peer_ip = false; + state.gate_ctx.peer_ip[0] = '\0'; + state.gate_ctx.is_local = false; + state.config = NULL; + state.context = NULL; + state.joined_destination = NULL; + state.charset_ready = false; + + if (!server_accept_config(&state)) + goto done; + if (!server_apply_security_gates(&state)) + goto done; + if (!server_prepare_session(&state)) + goto done; + server_run_transfer(&state); done: /* Single cleanup epilogue: every error path jumps here, so the iconv @@ -1029,22 +1112,22 @@ done: * connection fd is deliberately NOT closed here -- the child functions own * its single close (plain_child_fn / tls_child_fn), and the --stdio call * site must leave stdin/stdout open. */ - if (charset_ready) + if (state.charset_ready) charset_wire_free(); /* The delay-updates staging tree is released by config_delete (which the branch below always reaches), so it is cleaned exactly once. */ identity_clear_active(); protocol_session_unbind(); - if (context != NULL) { + if (state.context != NULL) { /* context owns both the config and the queue it was created with. */ - pipeline_context_receiver_destroy(context); - context = NULL; - config = NULL; + pipeline_context_receiver_destroy(state.context); + state.context = NULL; + state.config = NULL; } else { - config_delete(config); - config = NULL; + config_delete(state.config); + state.config = NULL; } - free(joined_destination); + free(state.joined_destination); } #ifndef FASTSYNC_SERVER_AS_LIB From ade9be86007720e8efbe7b8c20c8519cea1c342a Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 13:49:28 +0200 Subject: [PATCH 05/10] refactor(receiver): per-status dispatch and shared pending teardown --- src/server/receiver.c | 495 ++++++++++++++++++++++++++---------------- 1 file changed, 302 insertions(+), 193 deletions(-) diff --git a/src/server/receiver.c b/src/server/receiver.c index 178ca34..10c4c45 100644 --- a/src/server/receiver.c +++ b/src/server/receiver.c @@ -303,6 +303,264 @@ int receiver_process(Config* config, int file_descriptor, const ReceiverSink* si return receiver_process_pending(config, file_descriptor, sink, NULL, NULL); } +/* Per-connection state threaded through the status handlers below. The parked + keep-set / per-directory session live here so one teardown helper can release + them on every exit path. */ +typedef struct { + Config* config; + int fd; + const ReceiverSink* sink; + DeleteManifest** pending_manifest; + DeletePlanSession** pending_plans; + /* Parked keep-set for the late/commit timing. Every exit path frees it + exactly once; the only exception is the successful FINISHED handoff, which + transfers ownership to *pending_manifest (used by the -m receiver). */ + DeleteManifest* deferred_manifest; + /* Per-directory delete session for --delete-during/--delete-delay. During the + loop it applies plans inline (during) or snapshots their extras (delay); on + a successful FINISHED it is either committed here or handed to + *pending_plans so the -m caller commits after its disk writer drained. */ + DeletePlanSession* plan_session; + bool early_delete; + bool per_dir_delete; + bool delete_limit_noted; +} ReceiverPendingState; + +/* Outcome of one frame handler. NEXT reads the following status frame; FAIL + tears the connection down without a peer STATUS_ERROR; ERROR tears it down + and (when the sink owns error reporting) emits STATUS_ERROR. */ +typedef enum { + RECEIVER_STEP_NEXT, + RECEIVER_STEP_FAIL, + RECEIVER_STEP_ERROR, +} ReceiverStep; + +static ReceiverStep receiver_handle_keepalive(ReceiverPendingState* state) { + if (!send_status(state->fd, STATUS_KEEPALIVE)) + return RECEIVER_STEP_FAIL; + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_abort(ReceiverPendingState* state) { + (void)state; + log_message(LOG_LEVEL_INFO, "Received abort from client, cleaning up"); + return RECEIVER_STEP_FAIL; +} + +static ReceiverStep receiver_handle_check(ReceiverPendingState* state) { + bool skipped = false; + bool would_transfer = false; + File* file = receive_incremental_check_ex(state->fd, state->config, &skipped, &would_transfer); + if (state->config->dry_run) { + /* Server-contacting --dry-run: the reply has already been sent + (STATUS_OK = up to date, STATUS_DRY_RUN_TRANSFER = would transfer) and + nothing may be stored. Both flags false means a genuine protocol + error (STATUS_ERROR already sent or sent by receive_error below). */ + if (!skipped && !would_transfer) + return RECEIVER_STEP_ERROR; + } else if (!skipped && (!file || !state->sink->store_file(file, state->sink->context))) { + return RECEIVER_STEP_ERROR; + } + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_chunk(ReceiverPendingState* state) { + Chunk* chunk = receive_chunk_data(state->fd, state->config); + if (!chunk || !receiver_process_chunk(chunk, state->sink)) + return RECEIVER_STEP_ERROR; + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_check_batch(ReceiverPendingState* state) { + if (!receiver_process_batch(state->config, state->fd)) + return RECEIVER_STEP_FAIL; + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_mkdir(ReceiverPendingState* state) { + File* dir = file_receive_directory(state->fd, state->config); + if (!dir || !state->sink->store_file(dir, state->sink->context)) + return RECEIVER_STEP_ERROR; + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_dir_times(ReceiverPendingState* state) { + if (!receiver_process_dir_times(state->fd, state->config, state->sink)) + return RECEIVER_STEP_ERROR; + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_hardlink(ReceiverPendingState* state) { + File* file = file_receive_hardlink(state->fd); + if (!file || !state->sink->store_file(file, state->sink->context)) + return RECEIVER_STEP_ERROR; + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_symlink(ReceiverPendingState* state) { + File* sym = file_receive_symlink(state->fd, state->config); + if (!sym || !state->sink->store_file(sym, state->sink->context)) + return RECEIVER_STEP_ERROR; + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_special(ReceiverPendingState* state) { + File* file = file_receive_special(state->fd); + if (!file || !state->sink->store_file(file, state->sink->context)) + return RECEIVER_STEP_ERROR; + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_manifest(ReceiverPendingState* state) { + Config* config = state->config; + int fd = state->fd; + const ReceiverSink* sink = state->sink; + DeleteManifest* manifest = receive_manifest_entries(fd); + if (!manifest) + return RECEIVER_STEP_FAIL; /* receive_manifest_entries already sent STATUS_ERROR */ + if (config->dry_run) { + /* Server-contacting --dry-run mutates nothing, so a keep-set manifest + is consumed and discarded. The early-delete mode still needs its ACK + so a sender blocked on the delete handshake is not left hanging. + When would-delete reporting is armed, enumerate (read-only) the + destination extras so the terminal STATUS_STATS frame can list them. */ + if (config->use_delete && sink->would_delete) { + size_t count = 0; + if (!manifest_would_delete_list(config, manifest, sink->would_delete, &count)) + log_message(LOG_LEVEL_WARNING, "dry-run: could not enumerate would-delete paths"); + } + delete_manifest_free(manifest); + if (state->early_delete && !send_status(fd, STATUS_OK)) + return RECEIVER_STEP_FAIL; + return RECEIVER_STEP_NEXT; + } + if (state->early_delete) { + /* --delete-before: the whole-tree manifest is authoritative the moment + it arrives, before any file data. Delete now and acknowledge so the + sender only starts streaming once the deletion committed (or failed). + A later transfer failure does not restore these deletions. A + --max-delete-capped commit still succeeds and the transfer proceeds; + the terminal success frame reports the cap. */ + size_t deleted = 0; + DeletePathObserver observer = + (config->report_deletes && sink->deleted_paths) ? receiver_record_deleted_path : NULL; + DeleteCommitResult deletion = + (config->use_delete || config->delete_missing_args) + ? manifest_delete_all_observed(config, manifest, &deleted, observer, + (void*)sink->deleted_paths) + : DELETE_COMMIT_OK; + receiver_tally_deleted(sink, deleted); + delete_manifest_free(manifest); + if (deletion == DELETE_COMMIT_ERROR) { + send_status(fd, STATUS_ERROR); + return RECEIVER_STEP_FAIL; + } + if (deletion == DELETE_COMMIT_LIMIT_REACHED && sink->note_delete_limit) + sink->note_delete_limit(sink->context); + if (!send_status(fd, STATUS_OK)) + return RECEIVER_STEP_FAIL; + } else if (config->use_delete || config->delete_missing_args) { + /* Plain --delete / --delete-after and the --delete-missing-args + exact-path deletions: hold the manifest and commit it only after + STATUS_FINISHED. The per-directory modes never send this frame. */ + if (state->deferred_manifest) { + log_message(LOG_LEVEL_ERROR, "Received a second delete manifest"); + delete_manifest_free(state->deferred_manifest); + state->deferred_manifest = NULL; + delete_manifest_free(manifest); + send_status(fd, STATUS_ERROR); + return RECEIVER_STEP_FAIL; + } + state->deferred_manifest = manifest; + } else { + delete_manifest_free(manifest); + } + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_delete_plan(ReceiverPendingState* state) { + Config* config = state->config; + int fd = state->fd; + const ReceiverSink* sink = state->sink; + if (!state->per_dir_delete) { + log_message(LOG_LEVEL_ERROR, "Received a per-directory delete plan without a per-dir " + "delete timing"); + send_status(fd, STATUS_ERROR); + return RECEIVER_STEP_FAIL; + } + if (!state->plan_session) { + state->plan_session = delete_plan_session_create(config); + if (state->plan_session && config->report_deletes && sink->deleted_paths) + delete_plan_session_set_delete_observer(state->plan_session, receiver_record_deleted_path, + (void*)sink->deleted_paths); + } + if (!state->plan_session || delete_plan_session_receive(state->plan_session, config, fd) != 0) + return RECEIVER_STEP_FAIL; + if (delete_plan_session_limit_reached(state->plan_session) && !state->delete_limit_noted && + sink->note_delete_limit) { + sink->note_delete_limit(sink->context); + state->delete_limit_noted = true; + } + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_file(ReceiverPendingState* state) { + File* file = file_receive(state->config, state->fd); + if (!file) { + log_message(LOG_LEVEL_ERROR, "Failed to receive file"); + return RECEIVER_STEP_ERROR; + } + if (!state->sink->store_file(file, state->sink->context)) + return RECEIVER_STEP_ERROR; + return RECEIVER_STEP_NEXT; +} + +/* One dispatch per admitted frame type; STATUS_NEXT (and any other + data-bearing status) falls through to the regular file receiver. */ +static ReceiverStep receiver_dispatch_status(ReceiverPendingState* state, Status status) { + switch (status) { + case STATUS_KEEPALIVE: + return receiver_handle_keepalive(state); + case STATUS_ABORT: + return receiver_handle_abort(state); + case STATUS_CHECK: + return receiver_handle_check(state); + case STATUS_CHUNK: + return receiver_handle_chunk(state); + case STATUS_CHECK_BATCH: + return receiver_handle_check_batch(state); + case STATUS_MKDIR: + return receiver_handle_mkdir(state); + case STATUS_DIR_TIMES: + return receiver_handle_dir_times(state); + case STATUS_HARDLINK: + return receiver_handle_hardlink(state); + case STATUS_SYMLINK: + return receiver_handle_symlink(state); + case STATUS_SPECIAL: + return receiver_handle_special(state); + case STATUS_MANIFEST: + return receiver_handle_manifest(state); + case STATUS_DELETE_PLAN: + return receiver_handle_delete_plan(state); + default: + return receiver_handle_file(state); + } +} + +/* Release the parked keep-set / per-directory session exactly once on every + failure exit. Never commit a deletion for a failed stream. */ +static void receiver_drop_pending(ReceiverPendingState* state) { + if (state->deferred_manifest) { + delete_manifest_free(state->deferred_manifest); + state->deferred_manifest = NULL; + } + if (state->plan_session) { + delete_plan_session_destroy(state->plan_session); + state->plan_session = NULL; + } +} + /* Runs the whole receive loop. The delete manifest may legitimately arrive either FIRST (--delete-before / --delete-during: the sender transmits the validated keep-set before any file data) or LAST (--delete-after / @@ -312,9 +570,9 @@ int receiver_process(Config* config, int file_descriptor, const ReceiverSink* si deletion has committed (or failed); in the late modes the manifest is held and the deletion is committed only after the terminal STATUS_FINISHED proves the whole transfer succeeded. A plain --delete defaults to the per-directory - delete-during plan mode (no manifest at all). See - receiver_process_pending() for how the -m receiver defers that commit until - its disk writer has drained. */ + delete-during plan mode (no manifest at all). See the per-frame handlers + above for how the -m receiver defers that commit until its disk writer has + drained. */ int receiver_process_pending(Config* config, int file_descriptor, const ReceiverSink* sink, DeleteManifest** pending_manifest, DeletePlanSession** pending_plans) { Status status; @@ -329,166 +587,29 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver last_progress = session_start; if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor, sink)) return -1; - bool early_delete = config_delete_timing_early(config); - bool per_dir_delete = config_delete_timing_per_dir(config); - /* Parked keep-set for the late/commit timing. Every exit path below frees it - exactly once; the only exception is the successful FINISHED handoff, which - transfers ownership to *pending_manifest (used by the -m receiver). */ - DeleteManifest* deferred_manifest = NULL; - /* Per-directory delete session for --delete-during/--delete-delay. During the - loop it applies plans inline (during) or snapshots their extras (delay); on - a successful FINISHED it is either committed here or handed to - *pending_plans so the -m caller commits after its disk writer drained. */ - DeletePlanSession* plan_session = NULL; - bool delete_limit_noted = false; + ReceiverPendingState state = { + .config = config, + .fd = file_descriptor, + .sink = sink, + .pending_manifest = pending_manifest, + .pending_plans = pending_plans, + .deferred_manifest = NULL, + .plan_session = NULL, + .early_delete = config_delete_timing_early(config), + .per_dir_delete = config_delete_timing_per_dir(config), + .delete_limit_noted = false, + }; + bool notify_peer = false; while (status == STATUS_NEXT || status == STATUS_CHUNK || status == STATUS_CHECK || status == STATUS_KEEPALIVE || status == STATUS_ABORT || status == STATUS_CHECK_BATCH || status == STATUS_MKDIR || status == STATUS_MANIFEST || status == STATUS_HARDLINK || status == STATUS_SYMLINK || status == STATUS_SPECIAL || status == STATUS_DIR_TIMES || status == STATUS_DELETE_PLAN) { - if (status == STATUS_KEEPALIVE) { - if (!send_status(file_descriptor, STATUS_KEEPALIVE)) - goto fail; - goto next_status; - } - if (status == STATUS_ABORT) { - log_message(LOG_LEVEL_INFO, "Received abort from client, cleaning up"); + ReceiverStep step = receiver_dispatch_status(&state, status); + if (step == RECEIVER_STEP_FAIL) goto fail; - } - if (status == STATUS_CHECK) { - bool skipped = false; - bool would_transfer = false; - File* file = receive_incremental_check_ex(file_descriptor, config, &skipped, &would_transfer); - if (config->dry_run) { - /* Server-contacting --dry-run: the reply has already been sent - (STATUS_OK = up to date, STATUS_DRY_RUN_TRANSFER = would transfer) and - nothing may be stored. Both flags false means a genuine protocol - error (STATUS_ERROR already sent or sent by receive_error below). */ - if (!skipped && !would_transfer) - goto receive_error; - } else if (!skipped && (!file || !sink->store_file(file, sink->context))) { - goto receive_error; - } - } else if (status == STATUS_CHUNK) { - Chunk* chunk = receive_chunk_data(file_descriptor, config); - if (!chunk || !receiver_process_chunk(chunk, sink)) - goto receive_error; - } else if (status == STATUS_CHECK_BATCH) { - if (!receiver_process_batch(config, file_descriptor)) - goto fail; - goto next_status; - } else if (status == STATUS_MKDIR) { - File* dir = file_receive_directory(file_descriptor, config); - if (!dir || !sink->store_file(dir, sink->context)) - goto receive_error; - } else if (status == STATUS_DIR_TIMES) { - if (!receiver_process_dir_times(file_descriptor, config, sink)) - goto receive_error; - } else if (status == STATUS_HARDLINK) { - File* file = file_receive_hardlink(file_descriptor); - if (!file || !sink->store_file(file, sink->context)) - goto receive_error; - } else if (status == STATUS_SYMLINK) { - File* sym = file_receive_symlink(file_descriptor, config); - if (!sym || !sink->store_file(sym, sink->context)) - goto receive_error; - } else if (status == STATUS_SPECIAL) { - File* file = file_receive_special(file_descriptor); - if (!file || !sink->store_file(file, sink->context)) - goto receive_error; - } else if (status == STATUS_MANIFEST) { - DeleteManifest* manifest = receive_manifest_entries(file_descriptor); - if (!manifest) - goto fail; /* receive_manifest_entries already sent STATUS_ERROR */ - if (config->dry_run) { - /* Server-contacting --dry-run mutates nothing, so a keep-set manifest - is consumed and discarded. The early-delete mode still needs its ACK - so a sender blocked on the delete handshake is not left hanging. - When would-delete reporting is armed, enumerate (read-only) the - destination extras so the terminal STATUS_STATS frame can list them. */ - if (config->use_delete && sink->would_delete) { - size_t count = 0; - if (!manifest_would_delete_list(config, manifest, sink->would_delete, &count)) - log_message(LOG_LEVEL_WARNING, "dry-run: could not enumerate would-delete paths"); - } - delete_manifest_free(manifest); - if (early_delete && !send_status(file_descriptor, STATUS_OK)) - goto fail; - goto next_status; - } - if (early_delete) { - /* --delete-before: the whole-tree manifest is authoritative the moment - it arrives, before any file data. Delete now and acknowledge so the - sender only starts streaming once the deletion committed (or failed). - A later transfer failure does not restore these deletions. A - --max-delete-capped commit still succeeds and the transfer proceeds; - the terminal success frame reports the cap. */ - size_t deleted = 0; - DeletePathObserver observer = - (config->report_deletes && sink->deleted_paths) ? receiver_record_deleted_path : NULL; - DeleteCommitResult deletion = - (config->use_delete || config->delete_missing_args) - ? manifest_delete_all_observed(config, manifest, &deleted, observer, - (void*)sink->deleted_paths) - : DELETE_COMMIT_OK; - receiver_tally_deleted(sink, deleted); - delete_manifest_free(manifest); - if (deletion == DELETE_COMMIT_ERROR) { - send_status(file_descriptor, STATUS_ERROR); - goto fail; - } - if (deletion == DELETE_COMMIT_LIMIT_REACHED && sink->note_delete_limit) - sink->note_delete_limit(sink->context); - if (!send_status(file_descriptor, STATUS_OK)) - goto fail; - } else if (config->use_delete || config->delete_missing_args) { - /* Plain --delete / --delete-after and the --delete-missing-args - exact-path deletions: hold the manifest and commit it only after - STATUS_FINISHED. The per-directory modes never send this frame. */ - if (deferred_manifest) { - log_message(LOG_LEVEL_ERROR, "Received a second delete manifest"); - delete_manifest_free(deferred_manifest); - deferred_manifest = NULL; - delete_manifest_free(manifest); - send_status(file_descriptor, STATUS_ERROR); - goto fail; - } - deferred_manifest = manifest; - } else { - delete_manifest_free(manifest); - } - goto next_status; - } else if (status == STATUS_DELETE_PLAN) { - if (!per_dir_delete) { - log_message(LOG_LEVEL_ERROR, "Received a per-directory delete plan without a per-dir " - "delete timing"); - send_status(file_descriptor, STATUS_ERROR); - goto fail; - } - if (!plan_session) { - plan_session = delete_plan_session_create(config); - if (plan_session && config->report_deletes && sink->deleted_paths) - delete_plan_session_set_delete_observer(plan_session, receiver_record_deleted_path, - (void*)sink->deleted_paths); - } - if (!plan_session || delete_plan_session_receive(plan_session, config, file_descriptor) != 0) - goto fail; - if (delete_plan_session_limit_reached(plan_session) && !delete_limit_noted && - sink->note_delete_limit) { - sink->note_delete_limit(sink->context); - delete_limit_noted = true; - } - goto next_status; - } else { - File* file = file_receive(config, file_descriptor); - if (!file) { - log_message(LOG_LEVEL_ERROR, "Failed to receive file"); - goto receive_error; - } - if (!sink->store_file(file, sink->context)) - goto receive_error; - } - next_status: + if (step == RECEIVER_STEP_ERROR) + goto receive_error; if (!receive_status(file_descriptor, &status)) goto receive_error; if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor, sink)) @@ -507,19 +628,19 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver disk writer may still be draining; the caller commits after the writer has joined so no extra file is removed unless the transfer is known to have succeeded. */ - if (deferred_manifest) { - if (pending_manifest) { - *pending_manifest = deferred_manifest; - deferred_manifest = NULL; + if (state.deferred_manifest) { + if (state.pending_manifest) { + *state.pending_manifest = state.deferred_manifest; + state.deferred_manifest = NULL; } else { size_t deleted = 0; DeletePathObserver observer = (config->report_deletes && sink->deleted_paths) ? receiver_record_deleted_path : NULL; DeleteCommitResult deletion = manifest_delete_all_observed( - config, deferred_manifest, &deleted, observer, (void*)sink->deleted_paths); + config, state.deferred_manifest, &deleted, observer, (void*)sink->deleted_paths); receiver_tally_deleted(sink, deleted); - delete_manifest_free(deferred_manifest); - deferred_manifest = NULL; + delete_manifest_free(state.deferred_manifest); + state.deferred_manifest = NULL; if (deletion == DELETE_COMMIT_ERROR) { send_status(file_descriptor, STATUS_ERROR); goto fail; @@ -533,28 +654,28 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver nothing yet and applies its decompressed snapshot here. The -m receiver hands the session to its caller instead, which commits after the disk writer drained. */ - if (plan_session) { + if (state.plan_session) { if (config->report_deletes && sink->deleted_paths) - delete_plan_session_set_delete_observer(plan_session, receiver_record_deleted_path, + delete_plan_session_set_delete_observer(state.plan_session, receiver_record_deleted_path, (void*)sink->deleted_paths); - if (pending_plans) { - *pending_plans = plan_session; - plan_session = NULL; + if (state.pending_plans) { + *state.pending_plans = state.plan_session; + state.plan_session = NULL; } else if (config->dry_run) { /* Central dry-run no-op: never commit a deletion for a -n run. */ - delete_plan_session_destroy(plan_session); - plan_session = NULL; + delete_plan_session_destroy(state.plan_session); + state.plan_session = NULL; } else { - DeleteCommitResult deletion = delete_plan_session_commit(plan_session, config); - bool limit = delete_plan_session_limit_reached(plan_session); - receiver_tally_deleted(sink, delete_plan_session_deleted(plan_session)); - delete_plan_session_destroy(plan_session); - plan_session = NULL; + DeleteCommitResult deletion = delete_plan_session_commit(state.plan_session, config); + bool limit = delete_plan_session_limit_reached(state.plan_session); + receiver_tally_deleted(sink, delete_plan_session_deleted(state.plan_session)); + delete_plan_session_destroy(state.plan_session); + state.plan_session = NULL; if (deletion == DELETE_COMMIT_ERROR) { send_status(file_descriptor, STATUS_ERROR); goto fail; } - if (limit && !delete_limit_noted && sink->note_delete_limit) + if (limit && !state.delete_limit_noted && sink->note_delete_limit) sink->note_delete_limit(sink->context); } } @@ -568,26 +689,14 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver } return 0; +receive_error: + notify_peer = true; fail: /* Failure exits that must not (or already did) report a STATUS_ERROR. The parked keep-set/session is dropped: never commit a deletion for a failed stream. */ - if (deferred_manifest) { - delete_manifest_free(deferred_manifest); - deferred_manifest = NULL; - } - if (plan_session) - delete_plan_session_destroy(plan_session); - return -1; - -receive_error: - if (deferred_manifest) { - delete_manifest_free(deferred_manifest); - deferred_manifest = NULL; - } - if (plan_session) - delete_plan_session_destroy(plan_session); - if (sink->send_error) + receiver_drop_pending(&state); + if (notify_peer && sink->send_error) send_status(file_descriptor, STATUS_ERROR); return -1; } From 934defa9659837ae935703618c46590d890a1120 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 13:52:57 +0200 Subject: [PATCH 06/10] refactor(delete): consolidate delete engine into delete.c --- CMakeLists.txt | 1 + src/shared/delete.c | 656 +++++++++++++++++++++++++++++++++++++ src/shared/delete.h | 151 +++++++++ src/shared/delete_commit.c | 193 ++--------- src/shared/delete_commit.h | 6 +- src/shared/delete_plan.c | 49 +-- src/shared/delete_plan.h | 1 + src/shared/utils.c | 635 ----------------------------------- src/shared/utils.h | 94 ------ tests/test_shared_utils.c | 1 + 10 files changed, 845 insertions(+), 942 deletions(-) create mode 100644 src/shared/delete.c create mode 100644 src/shared/delete.h diff --git a/CMakeLists.txt b/CMakeLists.txt index eb01dbb..8f5dbf2 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -99,6 +99,7 @@ set(SHARED_SRCS src/shared/daemon_limits.c src/shared/data.c src/shared/delay_updates.c + src/shared/delete.c src/shared/delete_commit.c src/shared/delete_plan.c src/shared/delta.c diff --git a/src/shared/delete.c b/src/shared/delete.c new file mode 100644 index 0000000..3e4fb62 --- /dev/null +++ b/src/shared/delete.c @@ -0,0 +1,656 @@ +#include "delete.h" + +#include "delay_updates.h" +#include "filter.h" +#include "log.h" +#include "utils.h" +#include +#include +#include +#include +#include +#include +#include +#include + +/* Build the keep-set index from the exact manifest entries only. A lookup of + `rel` succeeds iff `rel` is a kept entry, a kept directory, or an ancestor + directory of kept content (the old is_dir_in_manifest predicate); the sorted + view answers "is an ancestor of kept content" without materializing any + per-component prefix copy, so the index is O(manifest size) memory. */ +static bool build_keep_index(const ArrayList* manifest, PathIndex* index) { + if (!manifest || manifest->size <= 0) + return path_index_build(index, NULL, 0); + return path_index_build(index, (const char* const*)manifest->items, (size_t)manifest->size); +} + +static bool keep_is_dir(const PathIndex* index, const char* rel_path) { + return path_index_contains(index, rel_path) || path_index_has_descendant(index, rel_path); +} + +static bool keep_is_file(const PathIndex* index, const char* rel_path) { + return path_index_contains(index, rel_path); +} + +/* True when child_rel is, or lies below, a protected entry. A prefix "a" + therefore protects "a" and "a/b/c" but not "ab". Entries with top_level_only + set only protect DIRECT children of the receive root (at_root); nested + directories that share such a name stay ordinary destination content. */ +bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips, + int skip_count) { + for (int i = 0; i < skip_count; i++) { + if (skips[i].top_level_only && !at_root) + continue; + size_t prefix_len = strlen(skips[i].prefix); + if (strncmp(child_rel, skips[i].prefix, prefix_len) == 0 && + (child_rel[prefix_len] == '\0' || child_rel[prefix_len] == '/')) + return true; + } + return false; +} + +/* Per-run deletion budget and tallies. `max_delete` is the cap on the number + of entries the walker may remove (SIZE_MAX = unlimited); once it is reached + the remaining extras are counted in `skipped` and left in place, matching + rsync's partial --max-delete behavior. */ +typedef struct { + size_t max_delete; + size_t deleted; + size_t skipped; + bool limit_hit; +} DeleteBudget; + +/* True when direct children of the directory named by `rel` may be removed. + With no synchronization info (dirs == NULL) the whole tree is deletable; when + a dirs index is supplied only its exact entries are (the receive root is the + "." sentinel). */ +static bool is_synced_dir(const PathIndex* dirs, const char* rel) { + if (!dirs) + return true; + return path_index_contains(dirs, rel[0] == '\0' ? "." : rel); +} + +/* Unsigned byte-wise string compare, matching rsync's u_strcmp (a signed + strcmp would order bytes >= 0x80 differently). */ +static int delete_name_cmp(const char* a, const char* b) { + const unsigned char* pa = (const unsigned char*)a; + const unsigned char* pb = (const unsigned char*)b; + while (*pa != '\0' && *pa == *pb) { + pa++; + pb++; + } + return (int)*pa - (int)*pb; +} + +bool delete_dir_entries_collect(int dirfd, DeleteDirEntry** out, size_t* count, + bool* operation_ok) { + *out = NULL; + *count = 0; + if (operation_ok) + *operation_ok = true; + int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + if (scanfd < 0) + return false; + DIR* dir = fdopendir(scanfd); + if (!dir) { + close(scanfd); + return false; + } + DeleteDirEntry* entries = NULL; + size_t used = 0; + size_t capacity = 0; + bool ok = true; + const struct dirent* entry; + while ((entry = readdir(dir)) != NULL) { + if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) + continue; + struct stat st; + if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) { + if (errno != ENOENT && operation_ok) + *operation_ok = false; + continue; + } + if (used == capacity) { + size_t next = capacity == 0 ? 16 : capacity * 2; + DeleteDirEntry* grown = realloc(entries, next * sizeof(*grown)); + if (!grown) { + ok = false; + break; + } + entries = grown; + capacity = next; + } + entries[used].name = str_dup(entry->d_name); + if (!entries[used].name) { + ok = false; + break; + } + entries[used].is_dir = S_ISDIR(st.st_mode); + used++; + } + closedir(dir); + if (!ok) { + delete_dir_entries_free(entries, used); + return false; + } + *out = entries; + *count = used; + return true; +} + +void delete_dir_entries_free(DeleteDirEntry* entries, size_t count) { + if (!entries) + return; + for (size_t i = 0; i < count; i++) + free(entries[i].name); + free(entries); +} + +/* rsync's extraneous-entry order: subdirectories before files, each group in + descending name order. */ +int delete_dir_entry_cmp_desc(const void* a, const void* b) { + const DeleteDirEntry* ea = a; + const DeleteDirEntry* eb = b; + if (ea->is_dir != eb->is_dir) + return ea->is_dir ? -1 : 1; + return -delete_name_cmp(ea->name, eb->name); +} + +/* rsync's kept-subdirectory order: plain ascending name. */ +int delete_dir_entry_cmp_asc(const void* a, const void* b) { + const DeleteDirEntry* ea = a; + const DeleteDirEntry* eb = b; + return delete_name_cmp(ea->name, eb->name); +} + +/* How the shared classification/descent walk disposes of an extra it has + identified. LIST records the destination-relative path without touching disk + (the -n/--dry-run would-delete enumeration); DELETE unlinks/rmdirs it, charges + the shared --max-delete budget and notifies the observer. Both modes classify + and traverse identically, so the dry-run enumeration and the real deletion + cannot drift. */ +typedef enum { DELETE_WALK_MODE_DELETE, DELETE_WALK_MODE_LIST } DeleteWalkMode; + +typedef struct { + DeleteWalkMode mode; + DeleteBudget* budget; /* DELETE mode */ + ArrayList* out; /* LIST mode: receives strdup'd relative paths */ + size_t* recorded; /* LIST mode */ + DeletePathObserver observer; /* DELETE mode */ + void* observer_context; /* DELETE mode */ +} DeleteWalkState; + +/* Remove the extras directly inside the directory open on `dirfd` (DELETE mode) + or record the paths that WOULD be removed (LIST mode), recursing into every + child directory so kept content below a synchronized prefix is reached. + `all_removed` reports whether every child entry was removed (so the caller may + rmdir this directory). A child directory is never removed when it is itself a + synchronized directory or holds kept content; with a dirs index supplied, + direct children of a non-synchronized directory are never extras at all (they + are left in place but still descended into). Symlinks are unlinked like any + other non-directory extra (never followed). + + Entries are processed in rsync's order (extraneous subdirectories in + descending name order, then extraneous files, then kept subdirectories in + ascending order) rather than readdir() order, so `--max-delete` leaves the + same survivors and the `--info=del`/dry-run line order matches rsync. */ +static bool delete_walk_fd(int dirfd, const char* rel_path, const PathIndex* keep, + const PathIndex* dirs, DeleteWalkState* state, + const DeleteSkipEntry* skips, int skip_count, + const FilterRuleList* protect_rules, bool parent_deletable, + bool* all_removed) { + DeleteDirEntry* entries = NULL; + size_t count = 0; + bool collect_ok = true; + if (!delete_dir_entries_collect(dirfd, &entries, &count, &collect_ok)) + return false; + bool operation_ok = collect_ok; + bool local_survives = false; + bool* shielded = calloc(count ? count : 1, sizeof(bool)); + bool* is_extra = calloc(count ? count : 1, sizeof(bool)); + if (!shielded || !is_extra) { + free(shielded); + free(is_extra); + delete_dir_entries_free(entries, count); + return false; + } + /* A directory is deletable when it or ANY ancestor is synchronized; the + `parent_deletable` flag carries that down the recursion so dest-only + directories below a synchronized root are removed wholesale. */ + bool deletable = parent_deletable || is_synced_dir(dirs, rel_path); + bool at_root = rel_path[0] == '\0'; + + /* Reproduce rsync's traversal order: extraneous subdirectories in descending + name order, then extraneous files in descending name order, and kept + subdirectories only afterwards (ascending). Sorting up front also fixes the + identity of the survivors under a partial --max-delete. */ + if (count > 1) + qsort(entries, count, sizeof(*entries), delete_dir_entry_cmp_desc); + size_t dir_count = 0; + while (dir_count < count && entries[dir_count].is_dir) + dir_count++; + + /* Classify every entry up front (the verdict does not depend on processing + order) so the ordered passes below can act on it. */ + for (size_t i = 0; i < count; i++) { + char* child_rel = path_cat((char*)rel_path, entries[i].name); + if (!child_rel) { + operation_ok = false; + continue; + } + /* A --delay-updates run keeps its staging directory as a direct child of + the receive root, and basis-dir snapshots live below it too. Their + contents are not manifest entries, so descending into them would delete + every staged / basis file as an "extra". Only the staging name (a + top-level-only prefix) and the basis prefixes are protected: a nested + destination directory that happens to be called .fastsync-stage is + ordinary content. */ + if (path_under_skip_prefix(child_rel, at_root, skips, skip_count)) { + shielded[i] = true; + local_survives = true; + } else if (protect_rules && + filter_rules_apply_side(protect_rules, child_rel, entries[i].name, entries[i].is_dir, + FILTER_SIDE_RECEIVER) == FILTER_ACTION_PROTECT) { + /* A first-match protect rule shields the extra; for a directory the whole + subtree is shielded (rsync prunes an excluded directory), so do not + descend. */ + shielded[i] = true; + local_survives = true; + } else if (entries[i].is_dir) { + bool child_synced = dirs && path_index_contains(dirs, child_rel); + is_extra[i] = deletable && !child_synced && !keep_is_dir(keep, child_rel); + if (!is_extra[i]) + local_survives = true; + } else { + is_extra[i] = deletable && !keep_is_file(keep, child_rel); + if (!is_extra[i]) + local_survives = true; + } + free(child_rel); + } + + /* Pass 1: extraneous subdirectories, descending. */ + for (size_t i = 0; i < dir_count; i++) { + if (!is_extra[i]) + continue; + char* child_rel = path_cat((char*)rel_path, entries[i].name); + if (!child_rel) { + operation_ok = false; + continue; + } + int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + bool child_all_removed = false; + if (childfd >= 0) { + if (!delete_walk_fd(childfd, child_rel, keep, dirs, state, skips, skip_count, protect_rules, + deletable, &child_all_removed)) + operation_ok = false; + close(childfd); + } else if (errno != ENOENT) { + operation_ok = false; + } + if (child_all_removed && deletable) { + if (state->mode == DELETE_WALK_MODE_LIST) { + /* Record the directory with rsync's trailing slash. */ + size_t len = strlen(child_rel); + char* copy = malloc(len + 2); + if (!copy) { + operation_ok = false; + } else { + memcpy(copy, child_rel, len); + copy[len] = '/'; + copy[len + 1] = '\0'; + if (!array_list_add(state->out, copy)) { + free(copy); + operation_ok = false; + } else { + (*state->recorded)++; + } + } + } else if (state->budget->deleted >= state->budget->max_delete) { + state->budget->limit_hit = true; + state->budget->skipped++; + local_survives = true; + } else if (unlinkat(dirfd, entries[i].name, AT_REMOVEDIR) != 0) { + /* ENOENT: already gone (fine). ENOTEMPTY/EEXIST: the directory still + holds entries the walker leaves in place (a protected excluded + prefix, a kept file the manifest protects, a symlink); rsync leaves + such a directory behind, so this is not an error. Only genuine I/O + failures abort the deletion. */ + if (errno != ENOENT && errno != ENOTEMPTY && errno != EEXIST) + operation_ok = false; + local_survives = true; + } else { + state->budget->deleted++; + /* rsync reports a removed directory with a trailing slash. */ + if (state->observer) { + size_t len = strlen(child_rel); + char* with_slash = malloc(len + 2); + if (with_slash) { + memcpy(with_slash, child_rel, len); + with_slash[len] = '/'; + with_slash[len + 1] = '\0'; + state->observer(state->observer_context, with_slash); + free(with_slash); + } else { + state->observer(state->observer_context, child_rel); + } + } + } + } else { + local_survives = true; + } + free(child_rel); + } + + /* Pass 2: extraneous files, descending. */ + for (size_t i = dir_count; i < count; i++) { + if (!is_extra[i]) + continue; + if (state->mode == DELETE_WALK_MODE_LIST) { + char* child_rel = path_cat((char*)rel_path, entries[i].name); + if (!child_rel) { + operation_ok = false; + continue; + } + char* copy = str_dup(child_rel); + if (!copy || !array_list_add(state->out, copy)) { + free(copy); + operation_ok = false; + } else { + (*state->recorded)++; + } + free(child_rel); + } else if (state->budget->deleted >= state->budget->max_delete) { + state->budget->limit_hit = true; + state->budget->skipped++; + local_survives = true; + } else if (unlinkat(dirfd, entries[i].name, 0) != 0) { + if (errno != ENOENT) + operation_ok = false; + local_survives = true; + } else { + state->budget->deleted++; + char* child_rel = path_cat((char*)rel_path, entries[i].name); + if (child_rel) { + if (state->observer) + state->observer(state->observer_context, child_rel); + char* escaped_path = output_escape(child_rel, log_get_8_bit_output()); + fprintf(stderr, " Deleted: %s\n", escaped_path ? escaped_path : ""); + free(escaped_path); + } + free(child_rel); + } + } + + /* Pass 3: kept subdirectories, ascending (rsync descends into these only + after the parent's own extras have been handled). */ + for (size_t i = dir_count; i-- > 0;) { + if (is_extra[i] || shielded[i]) + continue; + char* child_rel = path_cat((char*)rel_path, entries[i].name); + if (!child_rel) { + operation_ok = false; + continue; + } + int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + bool child_all_removed = false; + if (childfd >= 0) { + if (!delete_walk_fd(childfd, child_rel, keep, dirs, state, skips, skip_count, protect_rules, + deletable, &child_all_removed)) + operation_ok = false; + close(childfd); + } else if (errno != ENOENT) { + operation_ok = false; + } + /* A kept/synchronized directory is never removed. */ + local_survives = true; + free(child_rel); + } + + free(shielded); + free(is_extra); + delete_dir_entries_free(entries, count); + *all_removed = !local_survives; + return operation_ok; +} + +/* Open the receive root following the same authorized-root confinement the + walker uses, or dest_root directly when no authorized root is installed. */ +static int open_destination_root(const char* dest_root) { + int root_fd = utils_get_authorized_root_fd(); + if (root_fd >= 0) { + if (utils_get_authorized_root_path()) + return utils_open_authorized_destination(dest_root); + if (dest_root == NULL) + return dup(root_fd); + return -1; + } + return open(dest_root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); +} + +bool delete_extras_list(const char* dest_root, const ArrayList* manifest, + const ArrayList* synced_dirs, const DeleteSkipEntry* skips, int skip_count, + const FilterRuleList* protect_rules, ArrayList* out, size_t* count_out) { + if (count_out) + *count_out = 0; + if (!manifest || !out) + return false; + PathIndex keep; + if (!build_keep_index(manifest, &keep)) + return false; + PathIndex dirs; + bool have_dirs = synced_dirs != NULL; + if (have_dirs && + !path_index_build(&dirs, (const char* const*)synced_dirs->items, (size_t)synced_dirs->size)) { + path_index_free(&keep); + return false; + } + int rootfd = open_destination_root(dest_root); + if (rootfd < 0) { + path_index_free(&keep); + if (have_dirs) + path_index_free(&dirs); + return false; + } + bool all_removed = false; + size_t recorded = 0; + DeleteWalkState state = {.mode = DELETE_WALK_MODE_LIST, + .budget = NULL, + .out = out, + .recorded = &recorded, + .observer = NULL, + .observer_context = NULL}; + bool ok = delete_walk_fd(rootfd, "", &keep, have_dirs ? &dirs : NULL, &state, skips, skip_count, + protect_rules, false, &all_removed); + if (close(rootfd) != 0) + ok = false; + path_index_free(&keep); + if (have_dirs) + path_index_free(&dirs); + if (count_out) + *count_out = recorded; + return ok; +} + +DeleteWalkResult delete_extras_limited_observed(const char* dest_root, const ArrayList* manifest, + const ArrayList* synced_dirs, size_t max_delete, + const DeleteSkipEntry* skips, int skip_count, + const FilterRuleList* protect_rules, + size_t* deleted_out, size_t* skipped_out, + DeletePathObserver observer, + void* observer_context) { + if (deleted_out) + *deleted_out = 0; + if (skipped_out) + *skipped_out = 0; + if (!manifest) + return DELETE_WALK_ERROR; + /* Index the keep-set (and the synchronized-dir set, when supplied) once so + membership is answered in O(path length) instead of scanning every entry + for every destination entry. */ + PathIndex keep; + if (!build_keep_index(manifest, &keep)) + return DELETE_WALK_ERROR; + PathIndex dirs; + bool have_dirs = synced_dirs != NULL; + if (have_dirs && + !path_index_build(&dirs, (const char* const*)synced_dirs->items, (size_t)synced_dirs->size)) { + path_index_free(&keep); + return DELETE_WALK_ERROR; + } + int rootfd = open_destination_root(dest_root); + if (rootfd < 0) { + path_index_free(&keep); + if (have_dirs) + path_index_free(&dirs); + return DELETE_WALK_ERROR; + } + DeleteBudget budget = {.max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false}; + bool all_removed = false; + DeleteWalkState state = {.mode = DELETE_WALK_MODE_DELETE, + .budget = &budget, + .out = NULL, + .recorded = NULL, + .observer = observer, + .observer_context = observer_context}; + bool ok = delete_walk_fd(rootfd, "", &keep, have_dirs ? &dirs : NULL, &state, skips, skip_count, + protect_rules, false, &all_removed); + if (close(rootfd) != 0) + ok = false; + path_index_free(&keep); + if (have_dirs) + path_index_free(&dirs); + if (deleted_out) + *deleted_out = budget.deleted; + if (skipped_out) + *skipped_out = budget.skipped; + if (!ok) + return DELETE_WALK_ERROR; + return budget.limit_hit ? DELETE_WALK_LIMIT_REACHED : DELETE_WALK_OK; +} + +DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest, + const ArrayList* synced_dirs, size_t max_delete, + const DeleteSkipEntry* skips, int skip_count, + const FilterRuleList* protect_rules, size_t* deleted_out, + size_t* skipped_out) { + return delete_extras_limited_observed(dest_root, manifest, synced_dirs, max_delete, skips, + skip_count, protect_rules, deleted_out, skipped_out, NULL, + NULL); +} + +bool delete_extras(const char* dest_root, const ArrayList* manifest) { + return delete_extras_limited(dest_root, manifest, NULL, SIZE_MAX, NULL, 0, NULL, NULL, NULL) == + DELETE_WALK_OK; +} + +/* Build the delete-walk protection prefix for one basis directory. The walker + compares paths relative to the receive root, so a relative entry is already + in the right form; an absolute entry that lies below the root is converted to + its root-relative form, and one outside the root returns NULL (the walk + cannot reach it, and it is not protected data beneath the root). Exposed so + tests can exercise the root-of-"/" child mapping directly. */ +char* file_receive_basis_delete_relative(const Config* config, const char* path) { + if (!path) + return NULL; + if (path[0] != '/') + return str_dup(path); + const char* root = config->receive_root_directory; + if (!root || root[0] != '/') + return NULL; + size_t root_len = strlen(root); + while (root_len > 1 && root[root_len - 1] == '/') + root_len--; + if (strncmp(path, root, root_len) != 0) + return NULL; + if (root_len == 1) { + /* `root` is "/" (the only single-character absolute root): every absolute + path is below it, and the child relative form is everything after the + leading '/'. */ + if (path[1] == '\0') + return NULL; /* identical to the root, not a child */ + return str_dup(path + 1); + } + if (path[root_len] != '/') + return NULL; /* identical or a sibling sharing a name prefix */ + return str_dup(path + root_len + 1); +} + +bool delete_skips_build(const Config* config, const ArrayList* protected_paths, + const ArrayList* size_skipped, bool basis_root_relative, + DeleteSkipSet* out) { + if (!out) + return false; + out->entries = NULL; + out->owned_prefixes = NULL; + out->count = 0; + out->owned_count = 0; + if (!config) + return false; + int protected_count = protected_paths ? protected_paths->size : 0; + int size_skipped_count = size_skipped ? size_skipped->size : 0; + int count = + (config->delay_updates ? 1 : 0) + config->basis_count + protected_count + size_skipped_count; + if (count == 0) + return true; + out->entries = calloc((size_t)count, sizeof(DeleteSkipEntry)); + if (!out->entries) + return false; + if (basis_root_relative && config->basis_count > 0) { + out->owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); + if (!out->owned_prefixes) { + free(out->entries); + out->entries = NULL; + return false; + } + out->owned_count = config->basis_count; + } + int idx = 0; + if (config->delay_updates) { + out->entries[idx].prefix = DELAY_UPDATES_STAGING_DIR; + out->entries[idx].top_level_only = true; + idx++; + } + for (int i = 0; i < config->basis_count; i++) { + const char* prefix = config->basis_dirs[i].path; + if (basis_root_relative) { + /* An absolute basis outside the receive root is unreachable by this walk, + so it contributes no protection prefix (and no slot). */ + char* relative = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); + if (!relative) + continue; + out->owned_prefixes[i] = relative; + prefix = relative; + } + out->entries[idx].prefix = prefix; + out->entries[idx].top_level_only = false; + idx++; + } + for (int i = 0; i < protected_count; i++) { + out->entries[idx].prefix = (const char*)protected_paths->items[i]; + out->entries[idx].top_level_only = false; + idx++; + } + for (int i = 0; i < size_skipped_count; i++) { + out->entries[idx].prefix = (const char*)size_skipped->items[i]; + out->entries[idx].top_level_only = false; + idx++; + } + out->count = idx; + return true; +} + +void delete_skips_free(DeleteSkipSet* set) { + if (!set) + return; + if (set->owned_prefixes) { + for (int i = 0; i < set->owned_count; i++) + free(set->owned_prefixes[i]); + } + free(set->owned_prefixes); + free(set->entries); + set->entries = NULL; + set->owned_prefixes = NULL; + set->count = 0; + set->owned_count = 0; +} diff --git a/src/shared/delete.h b/src/shared/delete.h new file mode 100644 index 0000000..235db8c --- /dev/null +++ b/src/shared/delete.h @@ -0,0 +1,151 @@ +#ifndef DELETE_H +#define DELETE_H + +#include "array_list.h" +#include "config.h" +#include +#include + +/* Delete engine. + * + * This module owns destination-relative delete traversal: the ordered directory + * walker that reproduces rsync's extraneous-entry order, the skip-prefix + * protection set shared by every delete pass, and the read-only enumeration + * that mirrors the walker for -n/--dry-run. The budgeted manifest commit + * (delete_commit.c) and the per-directory delete plans (delete_plan.c) are + * built on the primitives exported here. */ + +/* Result of a bounded extra-file deletion run. */ +typedef enum { + /* Every extra entry was removed (or there were none). */ + DELETE_WALK_OK = 0, + /* The numeric cap for this run was reached before every extra was removed. + The walker removed exactly the entries the cap allowed and skipped (without + removing) the rest, matching rsync's partial --max-delete behavior. */ + DELETE_WALK_LIMIT_REACHED, + /* A traversal or unlink failure aborted the deletion (partial removal is + possible, mirroring the delete pass). */ + DELETE_WALK_ERROR +} DeleteWalkResult; + +/* One protected entry for the delete walker. When top_level_only is true the + prefix is skipped only as a DIRECT child of dest_root (the --delay-updates + staging directory, which must not hide genuine extras inside a nested + destination directory that happens to share the staging name); otherwise the + prefix is skipped at any depth (the --compare-dest/--copy-dest/--link-dest + basis trees, and the sender-side protected filter-excluded prefixes, which + are never destination content). */ +typedef struct { + const char* prefix; + bool top_level_only; +} DeleteSkipEntry; + +/* A built skip-prefix set. `entries`/`count` are what path_under_skip_prefix() + consumes. `owned_prefixes` holds any prefix strings the builder had to + allocate (root-relative basis-dir conversions); it is NULL when every prefix + is borrowed from the config or the caller's lists. Release with + delete_skips_free(). */ +typedef struct { + DeleteSkipEntry* entries; + char** owned_prefixes; + int count; + int owned_count; +} DeleteSkipSet; + +/* True when child_rel is, or lies below, one of the protected entries (a prefix + "a" protects "a" and "a/b/c" but not "ab"; top_level_only entries protect + only DIRECT children of the destination root, i.e. child_rel has no '/'). */ +bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips, + int skip_count); + +/* One destination-directory entry collected up front so the delete walkers can + reproduce rsync's traversal order instead of readdir() order. rsync processes + a directory's extraneous subdirectories first (descending name, depth-first), + then its extraneous files (descending name), and only afterwards descends into + its kept subdirectories (ascending name). */ +typedef struct { + char* name; + bool is_dir; +} DeleteDirEntry; +/* Collect the entries of the directory open on `dirfd` (excluding "." and ".."), + stat'ing each with AT_SYMLINK_NOFOLLOW. On success *out is a malloc'd array of + *count entries whose names the caller frees with delete_dir_entries_free(). + Returns false on an allocation/readdir failure; a vanished entry (ENOENT) is + skipped, any other stat failure is reported through *operation_ok while the + walk continues. */ +bool delete_dir_entries_collect(int dirfd, DeleteDirEntry** out, size_t* count, bool* operation_ok); +void delete_dir_entries_free(DeleteDirEntry* entries, size_t count); +/* Sort comparators: `_desc` orders subdirectories before files and each group by + descending name (rsync's extraneous-entry order); `_asc` orders plain ascending + name (rsync's kept-subdirectory order). */ +int delete_dir_entry_cmp_desc(const void* a, const void* b); +int delete_dir_entry_cmp_asc(const void* a, const void* b); + +/* Remove files/dirs/symlinks under dest_root that are not listed in manifest + without ever descending into a protected prefix (see DeleteSkipEntry). When + `synced_dirs` is non-NULL, extras are only removed directly inside a directory + whose destination-relative path is an exact entry in that list (the receive + root is the "." sentinel); directories outside the synchronized set are still + descended into so kept content below a listed directory is preserved, but + nothing in them is removed. A NULL `synced_dirs` keeps the legacy behavior of + treating the whole destination tree as deletable. `max_delete` caps the + number of removed entries (SIZE_MAX = unlimited): the walker removes up to the + cap and returns DELETE_WALK_LIMIT_REACHED when more extras remained. + `deleted_out`/`skipped_out` optionally receive the number of entries removed + and the number skipped because of the cap. */ +DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest, + const ArrayList* synced_dirs, size_t max_delete, + const DeleteSkipEntry* skips, int skip_count, + const FilterRuleList* protect_rules, size_t* deleted_out, + size_t* skipped_out); + +/* Optional per-deletion observer: called for each destination-relative path + actually removed (a file, symlink, or directory), in removal order, so the + receiver can stream rsync's `--info=del`/`--info=remove` lines. */ +typedef void (*DeletePathObserver)(void* context, const char* rel_path); + +/* `delete_extras_limited_observed` is delete_extras_limited with an optional + * observer; the observer is invoked only for entries truly removed. When + * `protect_rules` is non-NULL its receiver-side verdict is evaluated for every + * candidate extra: a first-match PROTECT leaves the entry (and, for a + * directory, its whole subtree) in place, while RISK/NONE fall through to the + * ordinary skip-prefix/keep-set logic. */ +DeleteWalkResult delete_extras_limited_observed(const char* dest_root, const ArrayList* manifest, + const ArrayList* synced_dirs, size_t max_delete, + const DeleteSkipEntry* skips, int skip_count, + const FilterRuleList* protect_rules, + size_t* deleted_out, size_t* skipped_out, + DeletePathObserver observer, + void* observer_context); +/* Read-only companion to delete_extras_limited: walk the destination exactly as + the delete pass would and APPEND (strdup'd) destination-relative paths that + WOULD be removed, without touching disk. Used for -n/--dry-run --delete + would-delete reporting. Returns true on a clean walk; the caller owns the + strings appended to `out` and receives their count in *count_out. */ +bool delete_extras_list(const char* dest_root, const ArrayList* manifest, + const ArrayList* synced_dirs, const DeleteSkipEntry* skips, int skip_count, + const FilterRuleList* protect_rules, ArrayList* out, size_t* count_out); +bool delete_extras(const char* dest_root, const ArrayList* manifest); + +/* Build the delete walk's skip-prefix set from the config's --delay-updates + staging directory, its --compare-dest/--copy-dest/--link-dest basis dirs, and + the caller-supplied protection lists, in that order. `protected_paths` and + `size_skipped` are borrowed (may be NULL); every entry in them is protected at + any depth. The staging directory is protected only as a DIRECT child of the + receive root. `basis_root_relative` selects how a basis path becomes a + prefix: true converts an absolute path under the receive root to its + root-relative form (the whole-tree commit walk; an unreachable path + contributes no slot), false keeps the configured path verbatim (the + per-directory plan walk). On success the caller releases `*out` with + delete_skips_free(); returns false on allocation failure. */ +bool delete_skips_build(const Config* config, const ArrayList* protected_paths, + const ArrayList* size_skipped, bool basis_root_relative, + DeleteSkipSet* out); +void delete_skips_free(DeleteSkipSet* set); + +/* Convert one basis-directory path to the receive-root-relative protection + prefix the delete walker uses (NULL when it lies outside the root). Exposed + for unit tests of the root-of-"/" and normalization edge cases. */ +char* file_receive_basis_delete_relative(const Config* config, const char* path); + +#endif diff --git a/src/shared/delete_commit.c b/src/shared/delete_commit.c index a522dd8..07f72db 100644 --- a/src/shared/delete_commit.c +++ b/src/shared/delete_commit.c @@ -126,38 +126,6 @@ typedef struct { bool limit_hit; } DeleteBudgetState; -/* Build the delete-walk protection prefix for one basis directory. The walker - compares paths relative to the receive root, so a relative entry is already - in the right form; an absolute entry that lies below the root is converted to - its root-relative form, and one outside the root returns NULL (the walk - cannot reach it, and it is not protected data beneath the root). Exposed so - tests can exercise the root-of-"/" child mapping directly. */ -char* file_receive_basis_delete_relative(const Config* config, const char* path) { - if (!path) - return NULL; - if (path[0] != '/') - return str_dup(path); - const char* root = config->receive_root_directory; - if (!root || root[0] != '/') - return NULL; - size_t root_len = strlen(root); - while (root_len > 1 && root[root_len - 1] == '/') - root_len--; - if (strncmp(path, root, root_len) != 0) - return NULL; - if (root_len == 1) { - /* `root` is "/" (the only single-character absolute root): every absolute - path is below it, and the child relative form is everything after the - leading '/'. */ - if (path[1] == '\0') - return NULL; /* identical to the root, not a child */ - return str_dup(path + 1); - } - if (path[root_len] != '/') - return NULL; /* identical or a sibling sharing a name prefix */ - return str_dup(path + root_len + 1); -} - /* Remove every destination entry under the receive root that is not in the keep-set, bounded by the shared budget (a smaller client --max-delete=NUM replaces the server hard bound; rsync deletes up to the bound and skips the @@ -174,55 +142,13 @@ static bool delete_extras_budgeted_observed(const Config* config, DeleteManifest if (!config || !manifest || !manifest->keeps) return false; fprintf(stderr, "Deleting files not in manifest...\n"); - /* Protected entries: - - the --delay-updates staging name, protected only as a DIRECT child of the - receive root (a nested destination directory that happens to be named - .fastsync-stage is ordinary content); - - alternate basis directories (--compare-dest / --copy-dest / --link-dest) - at any depth: they are extra comparison snapshots the user pointed at, - not destination content, and deleting them would destroy the very files a - --link-dest run just linked into place; - - the sender-side protected prefixes (source paths excluded by filters and - paths pruned by --max-size/--min-size), at any depth, so their destination - mirror survives --delete unless --delete-excluded opts back into removing - the filter-excluded ones (size-pruned entries are always protected). */ - int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count + - (manifest->protected ? manifest->protected->size : 0); - DeleteSkipEntry* skips = NULL; - char** owned_prefixes = NULL; - int used = 0; - if (skip_count > 0) { - skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); - owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); - if (!skips || (config->basis_count > 0 && !owned_prefixes)) { - free(skips); - free(owned_prefixes); - return false; - } - int idx = 0; - if (config->delay_updates) { - skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; - skips[idx].top_level_only = true; - idx++; - } - for (int i = 0; i < config->basis_count; i++) { - /* An absolute basis outside the receive root is unreachable by this walk, - so it contributes no protection prefix (and no slot). */ - char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); - if (!prefix) - continue; - owned_prefixes[i] = prefix; - skips[idx].prefix = prefix; - skips[idx].top_level_only = false; - idx++; - } - for (int i = 0; i < manifest->protected->size; i++) { - skips[idx].prefix = (const char*)manifest->protected->items[i]; - skips[idx].top_level_only = false; - idx++; - } - used = idx; - } + /* Protected entries: the --delay-updates staging name (only as a DIRECT child + of the receive root), the alternate basis directories and the sender-side + protected prefixes (filter-excluded and size-pruned source mirrors), all at + any depth. See delete_skips_build(). */ + DeleteSkipSet skips; + if (!delete_skips_build(config, manifest->protected, NULL, true, &skips)) + return false; /* Clamp rather than subtract: an accounting bug where deleted already exceeds max_delete must never underflow into an effectively unlimited budget. */ size_t remaining; @@ -235,14 +161,9 @@ static bool delete_extras_budgeted_observed(const Config* config, DeleteManifest size_t deleted = 0; size_t skipped = 0; DeleteWalkResult result = delete_extras_limited_observed( - config->receive_root_directory, manifest->keeps, manifest->dirs, remaining, skips, used, - config->protect_rules, &deleted, &skipped, observer, observer_context); - if (owned_prefixes) { - for (int i = 0; i < config->basis_count; i++) - free(owned_prefixes[i]); - } - free(owned_prefixes); - free(skips); + config->receive_root_directory, manifest->keeps, manifest->dirs, remaining, skips.entries, + skips.count, config->protect_rules, &deleted, &skipped, observer, observer_context); + delete_skips_free(&skips); budget->deleted += deleted; budget->skipped += skipped; if (result == DELETE_WALK_LIMIT_REACHED) { @@ -303,35 +224,12 @@ static bool delete_missing_args_budgeted_observed(const Config* config, DeleteMa if (!manifest->missing || manifest->missing->size == 0) return true; fprintf(stderr, "Deleting destination mirrors of missing source arguments...\n"); - int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count; - DeleteSkipEntry* skips = NULL; - char** owned_prefixes = NULL; - int used = 0; - if (skip_count > 0) { - skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); - owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); - if (!skips || (config->basis_count > 0 && !owned_prefixes)) { - free(skips); - free(owned_prefixes); - return false; - } - int idx = 0; - if (config->delay_updates) { - skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; - skips[idx].top_level_only = true; - idx++; - } - for (int i = 0; i < config->basis_count; i++) { - char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); - if (!prefix) - continue; - owned_prefixes[i] = prefix; - skips[idx].prefix = prefix; - skips[idx].top_level_only = false; - idx++; - } - used = idx; - } + /* The staging directory and basis snapshots stay protected exactly as in the + extras walker (the missing-args path overrides the ordinary protected + prefixes, so those are not passed here). */ + DeleteSkipSet skips; + if (!delete_skips_build(config, NULL, NULL, true, &skips)) + return false; bool ok = true; for (int i = 0; i < manifest->missing->size; i++) { const char* rel = (const char*)manifest->missing->items[i]; @@ -343,7 +241,7 @@ static bool delete_missing_args_budgeted_observed(const Config* config, DeleteMa continue; } bool at_root = strchr(rel, '/') == NULL; - if (path_under_skip_prefix(rel, at_root, skips, used)) { + if (path_under_skip_prefix(rel, at_root, skips.entries, skips.count)) { char* escaped = output_escape(rel, log_get_8_bit_output()); log_message(LOG_LEVEL_WARNING, "missing-args path '%s' is protected (staging directory or basis snapshot); " @@ -472,12 +370,7 @@ static bool delete_missing_args_budgeted_observed(const Config* config, DeleteMa if (!ok) break; } - if (owned_prefixes) { - for (int i = 0; i < config->basis_count; i++) - free(owned_prefixes[i]); - } - free(owned_prefixes); - free(skips); + delete_skips_free(&skips); return ok; } @@ -489,52 +382,12 @@ bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, *count_out = 0; if (!config || !manifest || !manifest->keeps || !out) return false; - int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count + - (manifest->protected ? manifest->protected->size : 0); - DeleteSkipEntry* skips = NULL; - char** owned_prefixes = NULL; - int used = 0; - if (skip_count > 0) { - skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); - owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); - if (!skips || (config->basis_count > 0 && !owned_prefixes)) { - free(skips); - free(owned_prefixes); - return false; - } - int idx = 0; - if (config->delay_updates) { - skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; - skips[idx].top_level_only = true; - idx++; - } - for (int i = 0; i < config->basis_count; i++) { - /* Normalize exactly like the real commit path: a relative entry is - already root-relative, an absolute one inside the receive root is - converted, and one outside contributes no protection prefix. */ - char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); - if (!prefix) - continue; - owned_prefixes[i] = prefix; - skips[idx].prefix = prefix; - skips[idx].top_level_only = false; - idx++; - } - for (int i = 0; i < manifest->protected->size; i++) { - skips[idx].prefix = (const char*)manifest->protected->items[i]; - skips[idx].top_level_only = false; - idx++; - } - used = idx; - } + DeleteSkipSet skips; + if (!delete_skips_build(config, manifest->protected, NULL, true, &skips)) + return false; bool ok = delete_extras_list(config->receive_root_directory, manifest->keeps, manifest->dirs, - skips, used, config->protect_rules, out, count_out); - if (owned_prefixes) { - for (int i = 0; i < config->basis_count; i++) - free(owned_prefixes[i]); - } - free(owned_prefixes); - free(skips); + skips.entries, skips.count, config->protect_rules, out, count_out); + delete_skips_free(&skips); return ok; } diff --git a/src/shared/delete_commit.h b/src/shared/delete_commit.h index c2d374b..16f00ae 100644 --- a/src/shared/delete_commit.h +++ b/src/shared/delete_commit.h @@ -3,7 +3,7 @@ #include "array_list.h" #include "config.h" -#include "utils.h" +#include "delete.h" #include /* Delete-commit module: delete-manifest receive plus the budgeted extras and @@ -102,9 +102,5 @@ DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteMani clean walk; `*count_out` receives the number of paths appended. */ bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, ArrayList* out, size_t* count_out); -/* Convert one basis-directory path to the receive-root-relative protection - prefix the delete walker uses (NULL when it lies outside the root). Exposed - for unit tests of the root-of-"/" and normalization edge cases. */ -char* file_receive_basis_delete_relative(const Config* config, const char* path); #endif diff --git a/src/shared/delete_plan.c b/src/shared/delete_plan.c index 58eb52e..f69213b 100644 --- a/src/shared/delete_plan.c +++ b/src/shared/delete_plan.c @@ -2,6 +2,7 @@ #include "charset.h" #include "delay_updates.h" +#include "delete.h" #include "file.h" #include "log.h" #include "utils.h" @@ -601,9 +602,8 @@ static int open_plan_dir(const Config* config, const char* dir) { return fd; } -typedef struct PlanSkips { - DeleteSkipEntry* entries; - int count; +typedef struct { + DeleteSkipSet set; /* Receiver-side delete-protection rules received on the config frame (NULL when the sender sent none). Evaluated per extra so a protect/risk rule is honored under --delete-during/--delete-delay exactly like the whole-tree @@ -613,39 +613,12 @@ typedef struct PlanSkips { static bool build_plan_skips(const Config* config, const DeletePlanSession* session, PlanSkips* out) { - out->entries = NULL; - out->count = 0; out->protect_rules = config->protect_rules; - int count = (config->delay_updates ? 1 : 0) + config->basis_count + - session->protected_prefixes->size + session->size_skipped->size; - if (count == 0) - return true; - out->entries = calloc((size_t)count, sizeof(DeleteSkipEntry)); - if (!out->entries) - return false; - int idx = 0; - if (config->delay_updates) { - out->entries[idx].prefix = DELAY_UPDATES_STAGING_DIR; - out->entries[idx].top_level_only = true; - idx++; - } - for (int i = 0; i < config->basis_count; i++) { - out->entries[idx].prefix = config->basis_dirs[i].path; - out->entries[idx].top_level_only = false; - idx++; - } - for (int i = 0; i < session->protected_prefixes->size; i++) { - out->entries[idx].prefix = (const char*)session->protected_prefixes->items[i]; - out->entries[idx].top_level_only = false; - idx++; - } - for (int i = 0; i < session->size_skipped->size; i++) { - out->entries[idx].prefix = (const char*)session->size_skipped->items[i]; - out->entries[idx].top_level_only = false; - idx++; - } - out->count = idx; - return true; + /* The per-directory plan walk keeps each basis path verbatim (it does not + convert an absolute under-root path to its root-relative form, unlike the + whole-tree commit walk). */ + return delete_skips_build(config, session->protected_prefixes, session->size_skipped, false, + &out->set); } static bool budget_available(const DeletePlanSession* session) { @@ -796,7 +769,7 @@ static bool process_children(int dirfd, const char* dir_rel, const ArrayList* ke operation_ok = false; continue; } - if (path_under_skip_prefix(child_rel, at_root, skips->entries, skips->count)) { + if (path_under_skip_prefix(child_rel, at_root, skips->set.entries, skips->set.count)) { shielded[i] = true; local_survives = true; free(child_rel); @@ -883,7 +856,7 @@ static bool apply_plan_dir(DeletePlanSession* session, const Config* config, con bool survives = false; bool ok = process_children(dirfd, dir, dirs, files, strcmp(dir, ".") == 0, false, &skips, session, &survives); - free(skips.entries); + delete_skips_free(&skips.set); close(dirfd); if (!ok) log_message(LOG_LEVEL_ERROR, "deletion failed while removing extraneous files"); @@ -1024,7 +997,7 @@ static bool apply_deferred_path(DeletePlanSession* session, const Config* config } bool survives = false; bool ok = process_children(dirfd, rel, NULL, NULL, false, true, &skips, session, &survives); - free(skips.entries); + delete_skips_free(&skips.set); close(dirfd); if (!ok) { close(parent_fd); diff --git a/src/shared/delete_plan.h b/src/shared/delete_plan.h index 93929c8..9472d65 100644 --- a/src/shared/delete_plan.h +++ b/src/shared/delete_plan.h @@ -3,6 +3,7 @@ #include "array_list.h" #include "config.h" +#include "delete.h" #include "file_receive.h" #include "protocol.h" #include "utils.h" diff --git a/src/shared/utils.c b/src/shared/utils.c index 947a4d9..f758e4e 100644 --- a/src/shared/utils.c +++ b/src/shared/utils.c @@ -625,641 +625,6 @@ bool format_human_bytes(unsigned long long bytes, char* buffer, size_t buffer_si return written >= 0 && (size_t)written < buffer_size; } -/* Build the keep-set index from the exact manifest entries only. A lookup of - `rel` succeeds iff `rel` is a kept entry, a kept directory, or an ancestor - directory of kept content (the old is_dir_in_manifest predicate); the sorted - view answers "is an ancestor of kept content" without materializing any - per-component prefix copy, so the index is O(manifest size) memory. */ -static bool build_keep_index(const ArrayList* manifest, PathIndex* index) { - if (!manifest || manifest->size <= 0) - return path_index_build(index, NULL, 0); - return path_index_build(index, (const char* const*)manifest->items, (size_t)manifest->size); -} - -static bool keep_is_dir(const PathIndex* index, const char* rel_path) { - return path_index_contains(index, rel_path) || path_index_has_descendant(index, rel_path); -} - -static bool keep_is_file(const PathIndex* index, const char* rel_path) { - return path_index_contains(index, rel_path); -} - -/* True when child_rel is, or lies below, a protected entry. A prefix "a" - therefore protects "a" and "a/b/c" but not "ab". Entries with top_level_only - set only protect DIRECT children of the receive root (at_root); nested - directories that share such a name stay ordinary destination content. */ -bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips, - int skip_count) { - for (int i = 0; i < skip_count; i++) { - if (skips[i].top_level_only && !at_root) - continue; - size_t prefix_len = strlen(skips[i].prefix); - if (strncmp(child_rel, skips[i].prefix, prefix_len) == 0 && - (child_rel[prefix_len] == '\0' || child_rel[prefix_len] == '/')) - return true; - } - return false; -} - -/* Per-run deletion budget and tallies. `max_delete` is the cap on the number - of entries the walker may remove (SIZE_MAX = unlimited); once it is reached - the remaining extras are counted in `skipped` and left in place, matching - rsync's partial --max-delete behavior. */ -typedef struct { - size_t max_delete; - size_t deleted; - size_t skipped; - bool limit_hit; -} DeleteBudget; - -/* True when direct children of the directory named by `rel` may be removed. - With no synchronization info (dirs == NULL) the whole tree is deletable; when - a dirs index is supplied only its exact entries are (the receive root is the - "." sentinel). */ -static bool is_synced_dir(const PathIndex* dirs, const char* rel) { - if (!dirs) - return true; - return path_index_contains(dirs, rel[0] == '\0' ? "." : rel); -} - -/* Unsigned byte-wise string compare, matching rsync's u_strcmp (a signed - strcmp would order bytes >= 0x80 differently). */ -static int delete_name_cmp(const char* a, const char* b) { - const unsigned char* pa = (const unsigned char*)a; - const unsigned char* pb = (const unsigned char*)b; - while (*pa != '\0' && *pa == *pb) { - pa++; - pb++; - } - return (int)*pa - (int)*pb; -} - -bool delete_dir_entries_collect(int dirfd, DeleteDirEntry** out, size_t* count, - bool* operation_ok) { - *out = NULL; - *count = 0; - if (operation_ok) - *operation_ok = true; - int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - if (scanfd < 0) - return false; - DIR* dir = fdopendir(scanfd); - if (!dir) { - close(scanfd); - return false; - } - DeleteDirEntry* entries = NULL; - size_t used = 0; - size_t capacity = 0; - bool ok = true; - const struct dirent* entry; - while ((entry = readdir(dir)) != NULL) { - if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) - continue; - struct stat st; - if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) { - if (errno != ENOENT && operation_ok) - *operation_ok = false; - continue; - } - if (used == capacity) { - size_t next = capacity == 0 ? 16 : capacity * 2; - DeleteDirEntry* grown = realloc(entries, next * sizeof(*grown)); - if (!grown) { - ok = false; - break; - } - entries = grown; - capacity = next; - } - entries[used].name = str_dup(entry->d_name); - if (!entries[used].name) { - ok = false; - break; - } - entries[used].is_dir = S_ISDIR(st.st_mode); - used++; - } - closedir(dir); - if (!ok) { - delete_dir_entries_free(entries, used); - return false; - } - *out = entries; - *count = used; - return true; -} - -void delete_dir_entries_free(DeleteDirEntry* entries, size_t count) { - if (!entries) - return; - for (size_t i = 0; i < count; i++) - free(entries[i].name); - free(entries); -} - -/* rsync's extraneous-entry order: subdirectories before files, each group in - descending name order. */ -int delete_dir_entry_cmp_desc(const void* a, const void* b) { - const DeleteDirEntry* ea = a; - const DeleteDirEntry* eb = b; - if (ea->is_dir != eb->is_dir) - return ea->is_dir ? -1 : 1; - return -delete_name_cmp(ea->name, eb->name); -} - -/* rsync's kept-subdirectory order: plain ascending name. */ -int delete_dir_entry_cmp_asc(const void* a, const void* b) { - const DeleteDirEntry* ea = a; - const DeleteDirEntry* eb = b; - return delete_name_cmp(ea->name, eb->name); -} - -/* Remove the extras directly inside the directory open on `dirfd`, recursing - into every child directory so kept content below a synchronized prefix is - reached. `all_removed` reports whether every child entry was removed (so the - caller may rmdir this directory). A child directory is never removed when it - is itself a synchronized directory or holds kept content; with a dirs index - supplied, direct children of a non-synchronized directory are never extras at - all (they are left in place but still descended into). Symlinks are unlinked - like any other non-directory extra (never followed). - - Entries are processed in rsync's order (extraneous subdirectories in - descending name order, then extraneous files, then kept subdirectories in - ascending order) rather than readdir() order, so `--max-delete` leaves the - same survivors and the `--info=del`/dry-run line order matches rsync. */ -static bool delete_extras_fd(int dirfd, const char* rel_path, const PathIndex* keep, - const PathIndex* dirs, DeleteBudget* budget, - const DeleteSkipEntry* skips, int skip_count, - const FilterRuleList* protect_rules, bool parent_deletable, - bool* all_removed, DeletePathObserver observer, - void* observer_context) { - DeleteDirEntry* entries = NULL; - size_t count = 0; - bool collect_ok = true; - if (!delete_dir_entries_collect(dirfd, &entries, &count, &collect_ok)) - return false; - bool operation_ok = collect_ok; - bool local_survives = false; - bool* shielded = calloc(count ? count : 1, sizeof(bool)); - bool* is_extra = calloc(count ? count : 1, sizeof(bool)); - if (!shielded || !is_extra) { - free(shielded); - free(is_extra); - delete_dir_entries_free(entries, count); - return false; - } - /* A directory is deletable when it or ANY ancestor is synchronized; the - `parent_deletable` flag carries that down the recursion so dest-only - directories below a synchronized root are removed wholesale. */ - bool deletable = parent_deletable || is_synced_dir(dirs, rel_path); - bool at_root = rel_path[0] == '\0'; - - /* Reproduce rsync's traversal order: extraneous subdirectories in descending - name order, then extraneous files in descending name order, and kept - subdirectories only afterwards (ascending). Sorting up front also fixes the - identity of the survivors under a partial --max-delete. */ - if (count > 1) - qsort(entries, count, sizeof(*entries), delete_dir_entry_cmp_desc); - size_t dir_count = 0; - while (dir_count < count && entries[dir_count].is_dir) - dir_count++; - - /* Classify every entry up front (the verdict does not depend on processing - order) so the ordered passes below can act on it. */ - for (size_t i = 0; i < count; i++) { - char* child_rel = path_cat((char*)rel_path, entries[i].name); - if (!child_rel) { - operation_ok = false; - continue; - } - /* A --delay-updates run keeps its staging directory as a direct child of - the receive root, and basis-dir snapshots live below it too. Their - contents are not manifest entries, so descending into them would delete - every staged / basis file as an "extra". Only the staging name (a - top-level-only prefix) and the basis prefixes are protected: a nested - destination directory that happens to be called .fastsync-stage is - ordinary content. */ - if (path_under_skip_prefix(child_rel, at_root, skips, skip_count)) { - shielded[i] = true; - local_survives = true; - } else if (protect_rules && - filter_rules_apply_side(protect_rules, child_rel, entries[i].name, entries[i].is_dir, - FILTER_SIDE_RECEIVER) == FILTER_ACTION_PROTECT) { - /* A first-match protect rule shields the extra; for a directory the whole - subtree is shielded (rsync prunes an excluded directory), so do not - descend. */ - shielded[i] = true; - local_survives = true; - } else if (entries[i].is_dir) { - bool child_synced = dirs && path_index_contains(dirs, child_rel); - is_extra[i] = deletable && !child_synced && !keep_is_dir(keep, child_rel); - if (!is_extra[i]) - local_survives = true; - } else { - is_extra[i] = deletable && !keep_is_file(keep, child_rel); - if (!is_extra[i]) - local_survives = true; - } - free(child_rel); - } - - /* Pass 1: extraneous subdirectories, descending. */ - for (size_t i = 0; i < dir_count; i++) { - if (!is_extra[i]) - continue; - char* child_rel = path_cat((char*)rel_path, entries[i].name); - if (!child_rel) { - operation_ok = false; - continue; - } - int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - bool child_all_removed = false; - if (childfd >= 0) { - if (!delete_extras_fd(childfd, child_rel, keep, dirs, budget, skips, skip_count, - protect_rules, deletable, &child_all_removed, observer, - observer_context)) - operation_ok = false; - close(childfd); - } else if (errno != ENOENT) { - operation_ok = false; - } - if (child_all_removed && deletable) { - if (budget->deleted >= budget->max_delete) { - budget->limit_hit = true; - budget->skipped++; - local_survives = true; - } else if (unlinkat(dirfd, entries[i].name, AT_REMOVEDIR) != 0) { - /* ENOENT: already gone (fine). ENOTEMPTY/EEXIST: the directory still - holds entries the walker leaves in place (a protected excluded - prefix, a kept file the manifest protects, a symlink); rsync leaves - such a directory behind, so this is not an error. Only genuine I/O - failures abort the deletion. */ - if (errno != ENOENT && errno != ENOTEMPTY && errno != EEXIST) - operation_ok = false; - local_survives = true; - } else { - budget->deleted++; - /* rsync reports a removed directory with a trailing slash. */ - if (observer) { - size_t len = strlen(child_rel); - char* with_slash = malloc(len + 2); - if (with_slash) { - memcpy(with_slash, child_rel, len); - with_slash[len] = '/'; - with_slash[len + 1] = '\0'; - observer(observer_context, with_slash); - free(with_slash); - } else { - observer(observer_context, child_rel); - } - } - } - } else { - local_survives = true; - } - free(child_rel); - } - - /* Pass 2: extraneous files, descending. */ - for (size_t i = dir_count; i < count; i++) { - if (!is_extra[i]) - continue; - if (budget->deleted >= budget->max_delete) { - budget->limit_hit = true; - budget->skipped++; - local_survives = true; - } else if (unlinkat(dirfd, entries[i].name, 0) != 0) { - if (errno != ENOENT) - operation_ok = false; - local_survives = true; - } else { - budget->deleted++; - char* child_rel = path_cat((char*)rel_path, entries[i].name); - if (child_rel) { - if (observer) - observer(observer_context, child_rel); - char* escaped_path = output_escape(child_rel, log_get_8_bit_output()); - fprintf(stderr, " Deleted: %s\n", escaped_path ? escaped_path : ""); - free(escaped_path); - } - free(child_rel); - } - } - - /* Pass 3: kept subdirectories, ascending (rsync descends into these only - after the parent's own extras have been handled). */ - for (size_t i = dir_count; i-- > 0;) { - if (is_extra[i] || shielded[i]) - continue; - char* child_rel = path_cat((char*)rel_path, entries[i].name); - if (!child_rel) { - operation_ok = false; - continue; - } - int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - bool child_all_removed = false; - if (childfd >= 0) { - if (!delete_extras_fd(childfd, child_rel, keep, dirs, budget, skips, skip_count, - protect_rules, deletable, &child_all_removed, observer, - observer_context)) - operation_ok = false; - close(childfd); - } else if (errno != ENOENT) { - operation_ok = false; - } - /* A kept/synchronized directory is never removed. */ - local_survives = true; - free(child_rel); - } - - free(shielded); - free(is_extra); - delete_dir_entries_free(entries, count); - *all_removed = !local_survives; - return operation_ok; -} - -/* Read-only mirror of delete_extras_fd: records the paths that WOULD be removed - without unlinking anything. A child directory is reported after its own - reportable children (depth-first), matching the delete pass's ordering. */ -static bool list_extras_fd(int dirfd, const char* rel_path, const PathIndex* keep, - const PathIndex* dirs, ArrayList* out, size_t* recorded, - const DeleteSkipEntry* skips, int skip_count, - const FilterRuleList* protect_rules, bool parent_deletable, - bool* all_removed) { - DeleteDirEntry* entries = NULL; - size_t count = 0; - bool collect_ok = true; - if (!delete_dir_entries_collect(dirfd, &entries, &count, &collect_ok)) - return false; - bool operation_ok = collect_ok; - bool local_survives = false; - bool* shielded = calloc(count ? count : 1, sizeof(bool)); - bool* is_extra = calloc(count ? count : 1, sizeof(bool)); - if (!shielded || !is_extra) { - free(shielded); - free(is_extra); - delete_dir_entries_free(entries, count); - return false; - } - bool deletable = parent_deletable || is_synced_dir(dirs, rel_path); - bool at_root = rel_path[0] == '\0'; - - /* Mirror the delete walk's rsync order (extraneous subdirectories descending, - then extraneous files descending, then kept subdirectories ascending). */ - if (count > 1) - qsort(entries, count, sizeof(*entries), delete_dir_entry_cmp_desc); - size_t dir_count = 0; - while (dir_count < count && entries[dir_count].is_dir) - dir_count++; - - for (size_t i = 0; i < count; i++) { - char* child_rel = path_cat((char*)rel_path, entries[i].name); - if (!child_rel) { - operation_ok = false; - continue; - } - if (path_under_skip_prefix(child_rel, at_root, skips, skip_count)) { - shielded[i] = true; - local_survives = true; - } else if (protect_rules && - filter_rules_apply_side(protect_rules, child_rel, entries[i].name, entries[i].is_dir, - FILTER_SIDE_RECEIVER) == FILTER_ACTION_PROTECT) { - /* Mirror the delete walk: a protected entry is never reported as a - would-delete and a protected directory's subtree is not enumerated. */ - shielded[i] = true; - local_survives = true; - } else if (entries[i].is_dir) { - bool child_synced = dirs && path_index_contains(dirs, child_rel); - is_extra[i] = deletable && !child_synced && !keep_is_dir(keep, child_rel); - if (!is_extra[i]) - local_survives = true; - } else { - is_extra[i] = deletable && !keep_is_file(keep, child_rel); - if (!is_extra[i]) - local_survives = true; - } - free(child_rel); - } - - /* Pass 1: extraneous subdirectories, descending (recorded after contents). */ - for (size_t i = 0; i < dir_count; i++) { - if (!is_extra[i]) - continue; - char* child_rel = path_cat((char*)rel_path, entries[i].name); - if (!child_rel) { - operation_ok = false; - continue; - } - int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - bool child_all_removed = false; - if (childfd >= 0) { - if (!list_extras_fd(childfd, child_rel, keep, dirs, out, recorded, skips, skip_count, - protect_rules, deletable, &child_all_removed)) - operation_ok = false; - close(childfd); - } else if (errno != ENOENT) { - operation_ok = false; - } - if (child_all_removed && deletable) { - size_t len = strlen(child_rel); - char* copy = malloc(len + 2); - if (!copy) { - operation_ok = false; - } else { - memcpy(copy, child_rel, len); - copy[len] = '/'; - copy[len + 1] = '\0'; - if (!array_list_add(out, copy)) { - free(copy); - operation_ok = false; - } else { - (*recorded)++; - } - } - } else { - local_survives = true; - } - free(child_rel); - } - - /* Pass 2: extraneous files, descending. */ - for (size_t i = dir_count; i < count; i++) { - if (!is_extra[i]) - continue; - char* child_rel = path_cat((char*)rel_path, entries[i].name); - if (!child_rel) { - operation_ok = false; - continue; - } - char* copy = str_dup(child_rel); - if (!copy || !array_list_add(out, copy)) { - free(copy); - operation_ok = false; - } else { - (*recorded)++; - } - free(child_rel); - } - - /* Pass 3: kept subdirectories, ascending. */ - for (size_t i = dir_count; i-- > 0;) { - if (is_extra[i] || shielded[i]) - continue; - char* child_rel = path_cat((char*)rel_path, entries[i].name); - if (!child_rel) { - operation_ok = false; - continue; - } - int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - bool child_all_removed = false; - if (childfd >= 0) { - if (!list_extras_fd(childfd, child_rel, keep, dirs, out, recorded, skips, skip_count, - protect_rules, deletable, &child_all_removed)) - operation_ok = false; - close(childfd); - } else if (errno != ENOENT) { - operation_ok = false; - } - local_survives = true; - free(child_rel); - } - - free(shielded); - free(is_extra); - delete_dir_entries_free(entries, count); - *all_removed = !local_survives; - return operation_ok; -} - -bool delete_extras_list(const char* dest_root, const ArrayList* manifest, - const ArrayList* synced_dirs, const DeleteSkipEntry* skips, int skip_count, - const FilterRuleList* protect_rules, ArrayList* out, size_t* count_out) { - if (count_out) - *count_out = 0; - if (!manifest || !out) - return false; - PathIndex keep; - if (!build_keep_index(manifest, &keep)) - return false; - PathIndex dirs; - bool have_dirs = synced_dirs != NULL; - if (have_dirs && - !path_index_build(&dirs, (const char* const*)synced_dirs->items, (size_t)synced_dirs->size)) { - path_index_free(&keep); - return false; - } - int rootfd; - int root_fd = utils_get_authorized_root_fd(); - if (root_fd >= 0) { - if (utils_get_authorized_root_path()) - rootfd = utils_open_authorized_destination(dest_root); - else if (dest_root == NULL) - rootfd = dup(root_fd); - else - rootfd = -1; - } else { - rootfd = open(dest_root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - } - if (rootfd < 0) { - path_index_free(&keep); - if (have_dirs) - path_index_free(&dirs); - return false; - } - bool all_removed = false; - size_t recorded = 0; - bool ok = list_extras_fd(rootfd, "", &keep, have_dirs ? &dirs : NULL, out, &recorded, skips, - skip_count, protect_rules, false, &all_removed); - if (close(rootfd) != 0) - ok = false; - path_index_free(&keep); - if (have_dirs) - path_index_free(&dirs); - if (count_out) - *count_out = recorded; - return ok; -} - -DeleteWalkResult delete_extras_limited_observed(const char* dest_root, const ArrayList* manifest, - const ArrayList* synced_dirs, size_t max_delete, - const DeleteSkipEntry* skips, int skip_count, - const FilterRuleList* protect_rules, - size_t* deleted_out, size_t* skipped_out, - DeletePathObserver observer, - void* observer_context) { - if (deleted_out) - *deleted_out = 0; - if (skipped_out) - *skipped_out = 0; - if (!manifest) - return DELETE_WALK_ERROR; - /* Index the keep-set (and the synchronized-dir set, when supplied) once so - membership is answered in O(path length) instead of scanning every entry - for every destination entry. */ - PathIndex keep; - if (!build_keep_index(manifest, &keep)) - return DELETE_WALK_ERROR; - PathIndex dirs; - bool have_dirs = synced_dirs != NULL; - if (have_dirs && - !path_index_build(&dirs, (const char* const*)synced_dirs->items, (size_t)synced_dirs->size)) { - path_index_free(&keep); - return DELETE_WALK_ERROR; - } - int rootfd; - int root_fd = utils_get_authorized_root_fd(); - if (root_fd >= 0) { - if (utils_get_authorized_root_path()) - rootfd = utils_open_authorized_destination(dest_root); - else if (dest_root == NULL) - rootfd = dup(root_fd); - else - rootfd = -1; - } else { - rootfd = open(dest_root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - } - if (rootfd < 0) { - path_index_free(&keep); - if (have_dirs) - path_index_free(&dirs); - return DELETE_WALK_ERROR; - } - DeleteBudget budget = {.max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false}; - bool all_removed = false; - bool ok = - delete_extras_fd(rootfd, "", &keep, have_dirs ? &dirs : NULL, &budget, skips, skip_count, - protect_rules, false, &all_removed, observer, observer_context); - if (close(rootfd) != 0) - ok = false; - path_index_free(&keep); - if (have_dirs) - path_index_free(&dirs); - if (deleted_out) - *deleted_out = budget.deleted; - if (skipped_out) - *skipped_out = budget.skipped; - if (!ok) - return DELETE_WALK_ERROR; - return budget.limit_hit ? DELETE_WALK_LIMIT_REACHED : DELETE_WALK_OK; -} - -DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest, - const ArrayList* synced_dirs, size_t max_delete, - const DeleteSkipEntry* skips, int skip_count, - const FilterRuleList* protect_rules, size_t* deleted_out, - size_t* skipped_out) { - return delete_extras_limited_observed(dest_root, manifest, synced_dirs, max_delete, skips, - skip_count, protect_rules, deleted_out, skipped_out, NULL, - NULL); -} - -bool delete_extras(const char* dest_root, const ArrayList* manifest) { - return delete_extras_limited(dest_root, manifest, NULL, SIZE_MAX, NULL, 0, NULL, NULL, NULL) == - DELETE_WALK_OK; -} - bool has_path_traversal(const char* path) { if (!path) return true; diff --git a/src/shared/utils.h b/src/shared/utils.h index aedb9ce..492200f 100644 --- a/src/shared/utils.h +++ b/src/shared/utils.h @@ -106,101 +106,7 @@ int env_choice_first(const char* env_name, int (*resolve)(const char*), bool* sp ssize_t utils_getdelim_bounded(FILE* stream, char** line, size_t* cap, int delim, size_t max_len); char* path_cat(const char* path1, const char* path2); bool glob_match(const char* pattern, const char* str); -/* Result of a bounded extra-file deletion run. */ -typedef enum { - /* Every extra entry was removed (or there were none). */ - DELETE_WALK_OK = 0, - /* The numeric cap for this run was reached before every extra was removed. - The walker removed exactly the entries the cap allowed and skipped (without - removing) the rest, matching rsync's partial --max-delete behavior. */ - DELETE_WALK_LIMIT_REACHED, - /* A traversal or unlink failure aborted the deletion (partial removal is - possible, mirroring the delete pass). */ - DELETE_WALK_ERROR -} DeleteWalkResult; -/* One protected entry for the delete walker. When top_level_only is true the - prefix is skipped only as a DIRECT child of dest_root (the --delay-updates - staging directory, which must not hide genuine extras inside a nested - destination directory that happens to share the staging name); otherwise the - prefix is skipped at any depth (the --compare-dest/--copy-dest/--link-dest - basis trees, and the sender-side protected filter-excluded prefixes, which - are never destination content). */ -typedef struct { - const char* prefix; - bool top_level_only; -} DeleteSkipEntry; -/* True when child_rel is, or lies below, one of the protected entries (a prefix - "a" protects "a" and "a/b/c" but not "ab"; top_level_only entries protect - only DIRECT children of the destination root, i.e. child_rel has no '/'). */ -bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips, - int skip_count); -/* One destination-directory entry collected up front so the delete walkers can - reproduce rsync's traversal order instead of readdir() order. rsync processes - a directory's extraneous subdirectories first (descending name, depth-first), - then its extraneous files (descending name), and only afterwards descends into - its kept subdirectories (ascending name). */ -typedef struct { - char* name; - bool is_dir; -} DeleteDirEntry; -/* Collect the entries of the directory open on `dirfd` (excluding "." and ".."), - stat'ing each with AT_SYMLINK_NOFOLLOW. On success *out is a malloc'd array of - *count entries whose names the caller frees with delete_dir_entries_free(). - Returns false on an allocation/readdir failure; a vanished entry (ENOENT) is - skipped, any other stat failure is reported through *operation_ok while the - walk continues. */ -bool delete_dir_entries_collect(int dirfd, DeleteDirEntry** out, size_t* count, bool* operation_ok); -void delete_dir_entries_free(DeleteDirEntry* entries, size_t count); -/* Sort comparators: `_desc` orders subdirectories before files and each group by - descending name (rsync's extraneous-entry order); `_asc` orders plain ascending - name (rsync's kept-subdirectory order). */ -int delete_dir_entry_cmp_desc(const void* a, const void* b); -int delete_dir_entry_cmp_asc(const void* a, const void* b); -/* Remove files/dirs/symlinks under dest_root that are not listed in manifest - without ever descending into a protected prefix (see DeleteSkipEntry). When - `synced_dirs` is non-NULL, extras are only removed directly inside a directory - whose destination-relative path is an exact entry in that list (the receive - root is the "." sentinel); directories outside the synchronized set are still - descended into so kept content below a listed directory is preserved, but - nothing in them is removed. A NULL `synced_dirs` keeps the legacy behavior of - treating the whole destination tree as deletable. `max_delete` caps the - number of removed entries (SIZE_MAX = unlimited): the walker removes up to the - cap and returns DELETE_WALK_LIMIT_REACHED when more extras remained. - `deleted_out`/`skipped_out` optionally receive the number of entries removed - and the number skipped because of the cap. */ -DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest, - const ArrayList* synced_dirs, size_t max_delete, - const DeleteSkipEntry* skips, int skip_count, - const FilterRuleList* protect_rules, size_t* deleted_out, - size_t* skipped_out); -/* Optional per-deletion observer: called for each destination-relative path - actually removed (a file, symlink, or directory), in removal order, so the - receiver can stream rsync's `--info=del`/`--info=remove` lines. */ -typedef void (*DeletePathObserver)(void* context, const char* rel_path); - -/* `delete_extras_limited_observed` is delete_extras_limited with an optional - * observer; the observer is invoked only for entries truly removed. When - * `protect_rules` is non-NULL its receiver-side verdict is evaluated for every - * candidate extra: a first-match PROTECT leaves the entry (and, for a - * directory, its whole subtree) in place, while RISK/NONE fall through to the - * ordinary skip-prefix/keep-set logic. */ -DeleteWalkResult delete_extras_limited_observed(const char* dest_root, const ArrayList* manifest, - const ArrayList* synced_dirs, size_t max_delete, - const DeleteSkipEntry* skips, int skip_count, - const FilterRuleList* protect_rules, - size_t* deleted_out, size_t* skipped_out, - DeletePathObserver observer, - void* observer_context); -/* Read-only companion to delete_extras_limited: walk the destination exactly as - the delete pass would and APPEND (strdup'd) destination-relative paths that - WOULD be removed, without touching disk. Used for -n/--dry-run --delete - would-delete reporting. Returns true on a clean walk; the caller owns the - strings appended to `out` and receives their count in *count_out. */ -bool delete_extras_list(const char* dest_root, const ArrayList* manifest, - const ArrayList* synced_dirs, const DeleteSkipEntry* skips, int skip_count, - const FilterRuleList* protect_rules, ArrayList* out, size_t* count_out); -bool delete_extras(const char* dest_root, const ArrayList* manifest); /* Open the existing destination directory at `dest_root`, confined to the authorized root with an O_NOFOLLOW component walk (the same confinement the deletion walker uses for its root). Returns a new fd the caller owns, or -1 diff --git a/tests/test_shared_utils.c b/tests/test_shared_utils.c index b822f1e..5fafa7e 100644 --- a/tests/test_shared_utils.c +++ b/tests/test_shared_utils.c @@ -1,4 +1,5 @@ #include "test_shared_utils.h" +#include "delete.h" #include "utils.h" #include "protocol.h" #include "test_utils.h" From b549887138a889783b15872dbdc1bb916693db7e Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 14:03:10 +0200 Subject: [PATCH 07/10] refactor(config): group CLI-parse state; drop old_args field --- src/client/change_list.c | 2 +- src/client/client_cli.c | 65 +++++++++++++------------- src/client/client_manifest.c | 2 +- src/client/client_send.c | 4 +- src/shared/config.c | 23 +++++----- src/shared/config.h | 88 ++++++++++++++++++++---------------- tests/test_change_list.c | 2 +- tests/test_client_cli.c | 45 +++++++++--------- tests/test_config.c | 8 ++-- 9 files changed, 126 insertions(+), 113 deletions(-) diff --git a/src/client/change_list.c b/src/client/change_list.c index 9f3d728..a2c5096 100644 --- a/src/client/change_list.c +++ b/src/client/change_list.c @@ -245,7 +245,7 @@ static char* change_render_name_uptodate(const ChangeEvent* event) { * resolves to xxh128, so an explicit selection and the default both render the * selected algorithm's digest. */ static ChecksumAlgo out_format_checksum_algo(const Config* config) { - return (ChecksumAlgo)config->checksum_transfer_algo; + return (ChecksumAlgo)config->cli.checksum_transfer_algo; } /* Render a digest as rsync's sum_as_hex: xxh128 prints the HIGH 64-bit half diff --git a/src/client/client_cli.c b/src/client/client_cli.c index fb0411b..12f0c6d 100644 --- a/src/client/client_cli.c +++ b/src/client/client_cli.c @@ -170,7 +170,7 @@ static int set_positive_int_option(int* dest, const char* value, const char* opt * name is a hard error with rsync's exit code 4, never a silent no-op. */ static int set_compression_choice(Config* config, const char* value) { if (!value) { - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } int algo; @@ -178,7 +178,7 @@ static int set_compression_choice(Config* config, const char* value) { algo = compression_choice_resolve(); if (algo < 0) { log_message(LOG_LEVEL_ERROR, "RSYNC_COMPRESS_LIST names no supported compression algorithm"); - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } } else { @@ -189,7 +189,7 @@ static int set_compression_choice(Config* config, const char* value) { "--compress-choice '%s' is not a supported algorithm; FastSync supports zstd, " "lz4, zlib, zlibx, none or auto", value); - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } const char* canonical = compression_algo_name((CompressionAlgo)algo); @@ -226,7 +226,7 @@ static int resolve_checksum_name(const char* name, size_t len, int* out) { * resolves to FastSync's negotiated default (xxh128). */ static int set_checksum_choice(Config* config, const char* value) { if (!value) { - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } const char* comma = strchr(value, ','); @@ -244,7 +244,7 @@ static int set_checksum_choice(Config* config, const char* value) { "--checksum-choice '%s' is invalid; FastSync supports xxh64 (or xxhash), xxh128, " "xxh3, md5, md4, sha1, none or auto, optionally as 'transfer,pre-transfer'", value); - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } int negotiated = -1; @@ -252,7 +252,7 @@ static int set_checksum_choice(Config* config, const char* value) { negotiated = checksum_choice_resolve(); if (negotiated < 0) { log_message(LOG_LEVEL_ERROR, "RSYNC_CHECKSUM_LIST names no supported checksum algorithm"); - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } } @@ -264,8 +264,8 @@ static int set_checksum_choice(Config* config, const char* value) { pre = negotiated; config->checksum_algo = pre; - config->checksum_transfer_algo = transfer; - config->checksum_choice_set = true; + config->cli.checksum_transfer_algo = transfer; + config->cli.checksum_choice_set = true; /* rsync: "none" for the transfer checksum forces --whole-file. */ if (transfer == (int)CHECKSUM_ALGO_NONE) config->whole_file = true; @@ -963,7 +963,10 @@ static const OptionEntry OPTION_TABLE[] = { * faithful no-op (accepted silently, never consumes an argument). */ {"--recursive", "-r", OPT_NOOP, 0}, {"--update", "-u", OPT_FLAG, offsetof(Config, update)}, - {"--old-args", NULL, OPT_FLAG, offsetof(Config, old_args)}, + /* rsync's --old-args: accepted for CLI compatibility as a documented no-op + * (the remote server path is always safely quoted; see usage.c). It is + * recognized but stores no Config field. */ + {"--old-args", NULL, OPT_NOOP, 0}, {"--rsh", "-e", OPT_STRING, offsetof(Config, rsh_command)}, {"--blocking-io", NULL, OPT_FLAG, offsetof(Config, blocking_io)}, {"--links", "-l", OPT_FLAG, offsetof(Config, follow_symlinks)}, @@ -1196,12 +1199,12 @@ static int apply_negation(Config* config, const char* arg) { config->preserve_times = false; config->preserve_owner = false; config->preserve_group = false; - config->metadata_explicitly_disabled = true; + config->cli.metadata_explicitly_disabled = true; /* --no-preserve is an explicit opt-out of the whole bundle: record it so * the --incremental/--delta auto-preserve in cli_finalize_config does not * silently re-enable perms/times. */ - config->preserve_perms_explicit_off = true; - config->preserve_times_explicit_off = true; + config->cli.preserve_perms_explicit_off = true; + config->cli.preserve_times_explicit_off = true; return 0; } *(bool*)((char*)config + entry->offset) = false; @@ -1209,9 +1212,9 @@ static int apply_negation(Config* config, const char* arg) { * auto-preserve the OTHER attribute without undoing this one. A later * -p/-t sets the attribute directly; this flag only gates the implication. */ if (entry->offset == offsetof(Config, preserve_perms)) - config->preserve_perms_explicit_off = true; + config->cli.preserve_perms_explicit_off = true; else if (entry->offset == offsetof(Config, preserve_times)) - config->preserve_times_explicit_off = true; + config->cli.preserve_times_explicit_off = true; return 0; } @@ -1440,7 +1443,7 @@ static bool cli_handle_range_time_options(CliParseCtx* ctx) { ctx->exit_code = -1; return true; } - config->stop_at_set = true; + config->cli.stop_at_set = true; return true; } if (strcmp(arg, "--stop-at") == 0) { @@ -1455,7 +1458,7 @@ static bool cli_handle_range_time_options(CliParseCtx* ctx) { ctx->exit_code = -1; return true; } - config->stop_at_set = true; + config->cli.stop_at_set = true; return true; } const char* threads_prefix = "--compress-threads="; @@ -1526,7 +1529,7 @@ static bool cli_handle_table_option(CliParseCtx* ctx) { return true; } if (entry->offset == offsetof(Config, compression_level)) - config->compression_level_set = true; + config->cli.compression_level_set = true; if (entry->offset == offsetof(Config, chmod_spec)) { mode_t ignored; if (!chmod_apply(0, config->chmod_spec, &ignored)) { @@ -1539,7 +1542,7 @@ static bool cli_handle_table_option(CliParseCtx* ctx) { defaults to 127.0.0.1, so a value check cannot distinguish it). Used by --dry-run to route an explicit remote target to the server. */ if (entry->offset == offsetof(Config, server_host)) - config->server_host_set = true; + config->cli.server_host_set = true; } } else if (apply_table_option(config, entry, NULL) != 0) { ctx->exit_code = -1; @@ -1813,7 +1816,7 @@ static bool cli_handle_transfer_flags(CliParseCtx* ctx) { return true; } config->compression_level = (int)level; - config->compression_level_set = true; + config->cli.compression_level_set = true; log_info_message(LOG_INFO_MISC, "Set Compression level to %ld", level); ctx->i++; } @@ -1877,7 +1880,7 @@ static int set_server_port_option(Config* config, const char* value, const char* return -1; } config->server_port = port; - config->server_port_set = true; + config->cli.server_port_set = true; return 0; } @@ -2648,7 +2651,7 @@ static int cli_finalize_config(Config* config, bool verbose, bool no_delta, bool int resolved = compression_choice_resolve(); if (resolved < 0) { log_message(LOG_LEVEL_ERROR, "RSYNC_COMPRESS_LIST names no supported compression algorithm"); - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } config->compression_algo = resolved; @@ -2659,7 +2662,7 @@ static int cli_finalize_config(Config* config, bool verbose, bool no_delta, bool * clamped to the codec's range, otherwise the codec's own default is used. */ if (config->use_compression) { CompressionAlgo algo = (CompressionAlgo)config->compression_algo; - config->compression_level = config->compression_level_set + config->compression_level = config->cli.compression_level_set ? compression_clamp_level(algo, config->compression_level) : compression_default_level(algo); log_debug_message(LOG_DEBUG_UTIL, "Client compression: %s (level %d)", @@ -2668,22 +2671,22 @@ static int cli_finalize_config(Config* config, bool verbose, bool no_delta, bool /* The negotiated checksum is always resolved (rsync negotiates one for the * delta strong sum even without --checksum): RSYNC_CHECKSUM_LIST first, then * the compiled-in order. An explicit --checksum-choice already set it. */ - if (!config->checksum_choice_set) { + if (!config->cli.checksum_choice_set) { int resolved = checksum_choice_resolve(); if (resolved < 0) { log_message(LOG_LEVEL_ERROR, "RSYNC_CHECKSUM_LIST names no supported checksum algorithm"); - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } config->checksum_algo = resolved; - config->checksum_transfer_algo = resolved; + config->cli.checksum_transfer_algo = resolved; } /* rsync parity: "none" as the pre-transfer checksum cannot be combined with * --checksum (exit 4). The check runs here because --checksum may appear on * either side of --checksum-choice. */ if (config->checksum && config->checksum_algo == (int)CHECKSUM_ALGO_NONE) { log_message(LOG_LEVEL_ERROR, "Invalid checksum-choice for --checksum: none"); - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } @@ -2762,11 +2765,11 @@ static int cli_finalize_config(Config* config, bool verbose, bool no_delta, bool * explicitly negated them (--no-perms/--no-times/--no-preserve). This runs * BEFORE the derived use_metadata bit so the transport frame is still sent * for the incremental/delta handshake even when both attributes were negated - * via --no-preserve (metadata_explicitly_disabled handles that opt-out). */ - if (preserve_implied && !config->metadata_explicitly_disabled) { - if (!config->preserve_perms_explicit_off) + * via --no-preserve (cli.metadata_explicitly_disabled handles that opt-out). */ + if (preserve_implied && !config->cli.metadata_explicitly_disabled) { + if (!config->cli.preserve_perms_explicit_off) config->preserve_perms = true; - if (!config->preserve_times_explicit_off) + if (!config->cli.preserve_times_explicit_off) config->preserve_times = true; } @@ -3153,7 +3156,7 @@ int main(int argc, char* argv[]) { int parse_ret = parse_args(config, argc, argv, positional_args, &positional_count); if (parse_ret != 0) { if (parse_ret < 0) - exit_code = config->cli_exit_code ? config->cli_exit_code : 1; + exit_code = config->cli.cli_exit_code ? config->cli.cli_exit_code : 1; goto cleanup; } diff --git a/src/client/client_manifest.c b/src/client/client_manifest.c index d476bb3..55a3f67 100644 --- a/src/client/client_manifest.c +++ b/src/client/client_manifest.c @@ -31,7 +31,7 @@ bool dry_run_targets_server(const Config* config) { return true; if (config->module && config->module[0] != '\0') return true; - if (config->server_host_set || config->server_port_set) + if (config->cli.server_host_set || config->cli.server_port_set) return true; if (config->use_tls) return true; diff --git a/src/client/client_send.c b/src/client/client_send.c index 3b77503..2f3422b 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -1578,7 +1578,7 @@ static bool send_files_run(Config* config, SendFilesState* state) { now_mono.tv_nsec = 0; } state->stop = stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, - config->stop_at_set, config->stop_at, now_mono); + config->cli.stop_at_set, config->stop_at, now_mono); state->prepared.options.stop_condition = &state->stop; /* The early-delete pre-scan above already ran; only the data pass should feed the directory-time list (otherwise every directory would be captured @@ -1916,7 +1916,7 @@ int send_files_multithreaded(Config* config) { now_mono.tv_nsec = 0; } context->stop_condition = - stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, config->stop_at_set, + stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, config->cli.stop_at_set, config->stop_at, now_mono); bool collect_excluded = config->use_delete && !config->delete_excluded; unsigned long long pre_scan_non_dir = 0; diff --git a/src/shared/config.c b/src/shared/config.c index f4b255d..eea3974 100644 --- a/src/shared/config.c +++ b/src/shared/config.c @@ -20,9 +20,9 @@ static void config_set_defaults(Config* config) { config->scanner_threads = 0; - config->metadata_explicitly_disabled = false; - config->preserve_perms_explicit_off = false; - config->preserve_times_explicit_off = false; + config->cli.preserve_perms_explicit_off = false; + config->cli.preserve_times_explicit_off = false; + config->cli.metadata_explicitly_disabled = false; config->show_progress = false; config->compression_threads = 0; config->ssh_port = 22; @@ -44,8 +44,8 @@ static void config_set_defaults(Config* config) { config->tls_ca = NULL; config->server_host = str_dup("127.0.0.1"); config->server_port = 8080; - config->server_port_set = false; - config->server_host_set = false; + config->cli.server_port_set = false; + config->cli.server_host_set = false; /* rsync defaults: --timeout=0 (I/O timeouts disabled) and --contimeout=60. * A value of 0 disables the client's own deadline on both the socket layer * (tcp_set_timeouts) and the protocol layer @@ -68,10 +68,10 @@ static void config_set_defaults(Config* config) { config->human_readable = false; config->ignore_errors = false; config->ignore_missing_args = false; - config->checksum_transfer_algo = CHECKSUM_ALGO_DEFAULT; - config->cli_exit_code = 0; - config->compression_level_set = false; - config->checksum_choice_set = false; + config->cli.checksum_transfer_algo = CHECKSUM_ALGO_DEFAULT; + config->cli.cli_exit_code = 0; + config->cli.compression_level_set = false; + config->cli.checksum_choice_set = false; config->filters = NULL; config->files_from = NULL; config->files_from_set = NULL; @@ -85,7 +85,6 @@ static void config_set_defaults(Config* config) { config->rsh_command = NULL; config->blocking_io = false; config->outbuf = OUTBUF_BLOCK; - config->old_args = false; config->remote_options = NULL; config->remote_option_count = 0; config->address = NULL; @@ -101,7 +100,7 @@ static void config_set_defaults(Config* config) { config->trust_sender = false; config->stop_after_mins = 0; config->stop_at = 0; - config->stop_at_set = false; + config->cli.stop_at_set = false; config->write_batch = NULL; config->only_write_batch = NULL; config->read_batch = NULL; @@ -342,7 +341,7 @@ bool config_derived_use_metadata(const Config* config) { config->chown_uid_set || config->chown_gid_set || config->usermap_count > 0 || config->groupmap_count > 0 || config->update) return true; - return (config->use_incremental || config->use_delta) && !config->metadata_explicitly_disabled; + return (config->use_incremental || config->use_delta) && !config->cli.metadata_explicitly_disabled; } bool config_has_basis(const Config* config) { diff --git a/src/shared/config.h b/src/shared/config.h index c4d7130..3596620 100644 --- a/src/shared/config.h +++ b/src/shared/config.h @@ -342,22 +342,60 @@ typedef enum SuperMode { SUPER_MODE_AUTO = 0, SUPER_MODE_ON = 1, SUPER_MODE_OFF CONFIG_WIRE_CODEC_FIELDS(X) \ CONFIG_WIRE_PROTECT_FIELDS(X) +/* Client-only, CLI-parse bookkeeping (never serialized). These members exist + * only so the client command-line parser can record HOW an option was + * specified (explicitly set, explicitly negated, or a parser-requested exit + * code); no other module and no wire peer ever needs them. Grouping them in + * one nested member keeps the public Config free of client-CLI-only state. */ +typedef struct { + /* Set when the user explicitly turned an attribute off with --no-perms / + * --no-times (long or short form). --incremental/--delta historically + * auto-enabled mode and mtime preservation; these flags let + * cli_finalize_config restore that behavior while still honoring the + * explicit per-attribute negation. A later -p/-t re-enables the attribute + * directly, so the flag only prevents the incremental/delta implication, + * never a POSITIVE request. */ + bool preserve_perms_explicit_off; + bool preserve_times_explicit_off; + /* Set by --no-preserve, the explicit opt-out of the whole preservation + * bundle, so the --incremental/--delta auto-preserve implication stays off. */ + bool metadata_explicitly_disabled; + /* True when --server-port/--port was explicitly given. --dry-run uses it to + * decide whether a real server handshake was requested, so a plain local + * destination (no explicit port) keeps the existing client-side dry-run + * behavior instead of dialing the default 127.0.0.1:8080. */ + bool server_port_set; + /* True when --server-host was explicitly given, and distinct from the + * "127.0.0.1" default: --dry-run uses it to route an explicit remote target + * to the server so it reports receiver state exactly like a real run, + * instead of silently running the client-side manifest. */ + bool server_host_set; + /* Codec-negotiation CLI state. The effective pre-transfer checksum is + * Config->checksum_algo (serialized); checksum_transfer_algo is the rsync + * "transfer" half of a two-name --checksum-choice form (validated and used + * only to mirror rsync's whole-file forcing, since FastSync's per-block + * strong hash is fixed). cli_exit_code carries a parser-requested process + * exit status (rsync uses 4 for an unsupported checksum/compress algorithm) + * so main() can mirror it. */ + int checksum_transfer_algo; + int cli_exit_code; + /* "The user explicitly chose" bits. They let the per-codec default level / + * checksum list be applied only when the corresponding rsync option was + * omitted (an explicit --compress-level / --checksum-choice always wins). */ + bool compression_level_set; + bool checksum_choice_set; + /* True when --stop-at was given. */ + bool stop_at_set; +} ConfigCliParse; + typedef struct Config { /* -j/--threads=N: number of parallel scanner worker threads for the -m * pipeline. 0 (the default, also set by bare -j/--threads) means "use the * scanner's built-in default" (4). CLIENT-ONLY: it is a local scheduling * concern and is NEVER serialized into the wire config frame. */ int scanner_threads; - bool metadata_explicitly_disabled; - /* CLIENT-ONLY (never serialized; not in CONFIG_WIRE_FIELDS). Set when the - * user explicitly turned an attribute off with --no-perms / --no-times (long - * or short form). --incremental/--delta historically auto-enabled mode and - * mtime preservation; these flags let cli_finalize_config restore that - * behavior while still honoring the explicit per-attribute negation. A - * later -p/-t re-enables the attribute directly, so the flag only prevents - * the incremental/delta implication, never a POSITIVE request. */ - bool preserve_perms_explicit_off; - bool preserve_times_explicit_off; + /* Client-only CLI-parse bookkeeping (never serialized). See ConfigCliParse. */ + ConfigCliParse cli; bool show_progress; int compression_threads; int ssh_port; @@ -378,18 +416,6 @@ typedef struct Config { bool use_tls; char* server_host; int server_port; - /* True when --server-port/--port was explicitly given. CLIENT-ONLY (never - * serialized): --dry-run uses it to decide whether a real server handshake - * was requested, so a plain local destination (no explicit port) keeps the - * existing client-side dry-run behavior instead of dialing the default - * 127.0.0.1:8080. */ - bool server_port_set; - /* True when --server-host was explicitly given. CLIENT-ONLY (never - * serialized), and distinct from the "127.0.0.1" default: --dry-run uses it - * to route an explicit remote target to the server so it reports receiver - * state exactly like a real run, instead of silently running the client-side - * manifest. */ - bool server_host_set; char* tls_cert; char* tls_key; char* tls_ca; @@ -435,22 +461,6 @@ typedef struct Config { * enters the keep-set. Implied by --delete-missing-args. */ bool ignore_missing_args; - /* Codec-negotiation CLI state (all client-only, never serialized). The - * effective pre-transfer checksum is Config->checksum_algo (serialized); - * checksum_transfer_algo is the rsync "transfer" half of a two-name - * --checksum-choice form (validated and used only to mirror rsync's - * whole-file forcing, since FastSync's per-block strong hash is fixed). - * cli_exit_code carries a parser-requested process exit status (rsync uses 4 - * for an unsupported checksum/compress algorithm) so main() can mirror it. */ - int checksum_transfer_algo; - int cli_exit_code; - /* Client-only "the user explicitly chose" bits. They let the per-codec - * default level / checksum list be applied only when the corresponding - * rsync option was omitted (an explicit --compress-level / --checksum-choice - * always wins). Never serialized. */ - bool compression_level_set; - bool checksum_choice_set; - // Issue #129: Advanced file selection. These fields are CLIENT-ONLY: they are // never serialized to the wire (the receiver must not learn them). ArrayList* filters; /* --filter=RULE rule strings, in order */ @@ -486,7 +496,6 @@ typedef struct Config { /* --outbuf mode (OutbufMode): stdout/stderr buffering. Client-only launch * concern: NEVER crosses the wire. */ int outbuf; - bool old_args; /* --remote-option=OPT (Phase 5, long form only): one or more extra command-line * options to append to the REMOTE server invocation over SSH. CLIENT-ONLY: * they are composed into the remote command line by ssh_build_remote_command() @@ -556,7 +565,6 @@ typedef struct Config { * process and are NEVER serialized into the config frame. */ int stop_after_mins; /* --stop-after=MINS minutes; 0 when unset */ time_t stop_at; /* --stop-at=... absolute wall-clock deadline */ - bool stop_at_set; /* true when --stop-at was given */ /* Client-only residual-batch paths. A residual batch is a self-contained * single-file record of the whole source tree (full file images using the diff --git a/tests/test_change_list.c b/tests/test_change_list.c index f9ad8a5..ef0f16d 100644 --- a/tests/test_change_list.c +++ b/tests/test_change_list.c @@ -195,7 +195,7 @@ static void test_format_C_padding_uses_transfer_algo() { {CHECKSUM_ALGO_NONE, 2}, }; for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { - config->checksum_transfer_algo = cases[i].algo; + config->cli.checksum_transfer_algo = cases[i].algo; char expected[64]; size_t n = 0; expected[n++] = '['; diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c index d86f8f6..e86aec7 100644 --- a/tests/test_client_cli.c +++ b/tests/test_client_cli.c @@ -730,7 +730,7 @@ static void test_parse_args_port_alias() { EXPECT_EQ_INT(cfg->server_port, 9000); /* The default port is 8080; the explicit bit is what lets --dry-run tell an explicit remote target from the default and route to the server. */ - EXPECT_TRUE(cfg->server_port_set); + EXPECT_TRUE(cfg->cli.server_port_set); config_delete(cfg); cfg = config_create(); @@ -738,7 +738,7 @@ static void test_parse_args_port_alias() { positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 4, argv_inline, positional_args, &positional_count), 0); EXPECT_EQ_INT(cfg->server_port, 9001); - EXPECT_TRUE(cfg->server_port_set); + EXPECT_TRUE(cfg->cli.server_port_set); config_delete(cfg); cfg = config_create(); @@ -746,7 +746,7 @@ static void test_parse_args_port_alias() { positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 4, argv_long, positional_args, &positional_count), 0); EXPECT_EQ_INT(cfg->server_port, 9002); - EXPECT_TRUE(cfg->server_port_set); + EXPECT_TRUE(cfg->cli.server_port_set); config_delete(cfg); } @@ -757,18 +757,18 @@ static void test_parse_args_server_host_sets_routing_bit() { Config* cfg = config_create(); int positional_args[2]; int positional_count = 0; - EXPECT_FALSE(cfg->server_host_set); + EXPECT_FALSE(cfg->cli.server_host_set); char* argv_space[] = {"fastsync", "--server-host", "example.test", "/src", "/dst"}; EXPECT_EQ_INT(parse_args(cfg, 5, argv_space, positional_args, &positional_count), 0); EXPECT_EQ_STR(cfg->server_host, "example.test"); - EXPECT_TRUE(cfg->server_host_set); + EXPECT_TRUE(cfg->cli.server_host_set); config_delete(cfg); cfg = config_create(); char* argv_inline[] = {"fastsync", "--server-host=example.test", "/src", "/dst"}; positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 4, argv_inline, positional_args, &positional_count), 0); - EXPECT_TRUE(cfg->server_host_set); + EXPECT_TRUE(cfg->cli.server_host_set); config_delete(cfg); } @@ -1726,7 +1726,7 @@ static void test_parse_args_no_preserve_blocks_implicit_metadata() { EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); EXPECT_FALSE(cfg->use_metadata); - EXPECT_TRUE(cfg->metadata_explicitly_disabled); + EXPECT_TRUE(cfg->cli.metadata_explicitly_disabled); config_delete(cfg); } } @@ -1777,7 +1777,7 @@ static void test_parse_args_checksum_choice_rejects_unsupported() { int positional_args[2]; int positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1); - EXPECT_EQ_INT(cfg->cli_exit_code, 4); + EXPECT_EQ_INT(cfg->cli.cli_exit_code, 4); config_delete(cfg); } } @@ -1817,7 +1817,7 @@ static void test_parse_args_checksum_choice_new_algos() { positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 5, argv4, positional_args, &positional_count), 0); EXPECT_EQ_INT(cfg->checksum_algo, single[i]); - EXPECT_EQ_INT(cfg->checksum_transfer_algo, single[i]); + EXPECT_EQ_INT(cfg->cli.checksum_transfer_algo, single[i]); config_delete(cfg); } @@ -1827,7 +1827,7 @@ static void test_parse_args_checksum_choice_new_algos() { char* argv5[] = {"fastsync", "--cc=sha1,md4", "/checksum/src", "/dst"}; positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 4, argv5, positional_args, &positional_count), 0); - EXPECT_EQ_INT(cfg->checksum_transfer_algo, (int)CHECKSUM_ALGO_SHA1); + EXPECT_EQ_INT(cfg->cli.checksum_transfer_algo, (int)CHECKSUM_ALGO_SHA1); EXPECT_EQ_INT(cfg->checksum_algo, (int)CHECKSUM_ALGO_MD4); config_delete(cfg); @@ -1849,14 +1849,14 @@ static void test_parse_args_checksum_none_with_checksum_rejected() { int positional_args[2]; int positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1); - EXPECT_EQ_INT(cfg->cli_exit_code, 4); + EXPECT_EQ_INT(cfg->cli.cli_exit_code, 4); config_delete(cfg); cfg = config_create(); char* argv2[] = {"fastsync", "--checksum", "--cc=md5,none", "/checksum/src", "/dst"}; positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 5, argv2, positional_args, &positional_count), -1); - EXPECT_EQ_INT(cfg->cli_exit_code, 4); + EXPECT_EQ_INT(cfg->cli.cli_exit_code, 4); config_delete(cfg); /* "none" as the TRANSFER checksum with a real pre-transfer checksum is @@ -1958,7 +1958,7 @@ static void test_parse_args_compress_choice_parity() { int positional_args[2]; int positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1); - EXPECT_EQ_INT(cfg->cli_exit_code, 4); + EXPECT_EQ_INT(cfg->cli.cli_exit_code, 4); config_delete(cfg); } } @@ -2078,6 +2078,9 @@ static void test_parse_args_temp_dir() { config_delete(cfg); } +/* --old-args is accepted for rsync CLI compatibility as a documented no-op (the + * remote server path is always safely quoted); it stores no Config field, so + * parsing it must simply succeed and leave the positional arguments intact. */ static void test_parse_args_old_args() { Config* cfg = config_create(); char* argv[] = {"fastsync", "--old-args", "/src", "/dst"}; @@ -2085,7 +2088,7 @@ static void test_parse_args_old_args() { int positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); - EXPECT_TRUE(cfg->old_args); + EXPECT_EQ_INT(positional_count, 2); config_delete(cfg); } @@ -2719,7 +2722,7 @@ static void test_parse_args_compression_env_list() { cfg = config_create(); positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), -1); - EXPECT_EQ_INT(cfg->cli_exit_code, 4); + EXPECT_EQ_INT(cfg->cli.cli_exit_code, 4); config_delete(cfg); unsetenv("RSYNC_COMPRESS_LIST"); } @@ -2734,7 +2737,7 @@ static void test_parse_args_checksum_env_list() { setenv("RSYNC_CHECKSUM_LIST", "md5", 1); EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); EXPECT_EQ_INT(cfg->checksum_algo, (int)CHECKSUM_ALGO_MD5); - EXPECT_EQ_INT(cfg->checksum_transfer_algo, (int)CHECKSUM_ALGO_MD5); + EXPECT_EQ_INT(cfg->cli.checksum_transfer_algo, (int)CHECKSUM_ALGO_MD5); config_delete(cfg); /* An explicit --cc wins. */ @@ -2750,7 +2753,7 @@ static void test_parse_args_checksum_env_list() { cfg = config_create(); positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), -1); - EXPECT_EQ_INT(cfg->cli_exit_code, 4); + EXPECT_EQ_INT(cfg->cli.cli_exit_code, 4); config_delete(cfg); unsetenv("RSYNC_CHECKSUM_LIST"); } @@ -4385,7 +4388,7 @@ static void test_parse_args_preserve_long_form() { } /* --no-perms/--no-times/--no-owner/--no-group (long and short) clear only - * their own attribute bit; they never set metadata_explicitly_disabled. */ + * their own attribute bit; they never set cli.metadata_explicitly_disabled. */ static void test_parse_args_preserve_negations() { struct { const char* arg; @@ -4413,7 +4416,7 @@ static void test_parse_args_preserve_negations() { bool expected = all[j] != cases[i].offset; EXPECT_TRUE(*(bool*)((char*)cfg + all[j]) == expected); } - EXPECT_FALSE(cfg->metadata_explicitly_disabled); + EXPECT_FALSE(cfg->cli.metadata_explicitly_disabled); /* -a's devices/specials keep the metadata frame on. */ EXPECT_TRUE(cfg->use_metadata); config_delete(cfg); @@ -4456,7 +4459,7 @@ static void test_parse_args_no_preserve_disables_bundle() { EXPECT_FALSE(cfg->preserve_times); EXPECT_FALSE(cfg->preserve_owner); EXPECT_FALSE(cfg->preserve_group); - EXPECT_TRUE(cfg->metadata_explicitly_disabled); + EXPECT_TRUE(cfg->cli.metadata_explicitly_disabled); EXPECT_TRUE(cfg->use_incremental); EXPECT_FALSE(cfg->use_metadata); config_delete(cfg); @@ -4509,7 +4512,7 @@ static void test_parse_args_incremental_implies_preserve() { EXPECT_EQ_INT(parse_args(cfg, 5, argv4, positional_args, &positional_count), 0); EXPECT_FALSE(cfg->preserve_perms); EXPECT_FALSE(cfg->preserve_times); - EXPECT_TRUE(cfg->metadata_explicitly_disabled); + EXPECT_TRUE(cfg->cli.metadata_explicitly_disabled); EXPECT_FALSE(cfg->use_metadata); config_delete(cfg); } diff --git a/tests/test_config.c b/tests/test_config.c index a801d70..4c87b34 100644 --- a/tests/test_config.c +++ b/tests/test_config.c @@ -2533,15 +2533,15 @@ static void test_config_derived_use_metadata() { /* Incremental/delta imply metadata unless --no-preserve disabled it. */ c->use_incremental = true; EXPECT_TRUE(config_derived_use_metadata(c)); - c->metadata_explicitly_disabled = true; + c->cli.metadata_explicitly_disabled = true; EXPECT_FALSE(config_derived_use_metadata(c)); - c->metadata_explicitly_disabled = false; + c->cli.metadata_explicitly_disabled = false; c->use_incremental = false; c->use_delta = true; EXPECT_TRUE(config_derived_use_metadata(c)); - c->metadata_explicitly_disabled = true; + c->cli.metadata_explicitly_disabled = true; EXPECT_FALSE(config_derived_use_metadata(c)); - c->metadata_explicitly_disabled = false; + c->cli.metadata_explicitly_disabled = false; c->use_delta = false; /* Flags that must NOT imply metadata on their own. */ From be20e836dedd465fc178243d4dbb4dad4d88818b Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 14:06:26 +0200 Subject: [PATCH 08/10] fix: correct throttle legacy resolution; add EXDEV temp-dir coverage --- RSYNC_COMPAT.md | 7 ++- src/shared/file_send.c | 2 +- src/shared/protocol.c | 13 ++-- src/shared/protocol.h | 13 ++-- tests/integration/test_temp_dir_exdev.py | 80 ++++++++++++++++++++++++ tests/test_protocol.c | 45 ++++++++++++- 6 files changed, 146 insertions(+), 14 deletions(-) create mode 100644 tests/integration/test_temp_dir_exdev.py diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 7657532..decd2d5 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -1013,7 +1013,12 @@ integration tests unless it is explicitly listed as a limitation. - **`--temp-dir` is confined to the receive root on the receiver:** a relative dir resolves below it; an absolute path or one containing `..` is rejected. - An `EXDEV` install falls back to a non-atomic copy instead of aborting. + An `EXDEV` install falls back to a non-atomic copy instead of aborting. (The + confined receiver path cannot be mount-tested in the CI container — no + `CAP_SYS_ADMIN` and unprivileged user namespaces are disabled — so the + cross-filesystem fallback is exercised end-to-end through the unconfined local + `--read-batch` apply against a `/dev/shm` scratch dir, in + `tests/integration/test_temp_dir_exdev.py`.) - **Deletion scoping:** the manifest carries the synchronized directories, so the extras walk only visits their subtrees; `--files-from` subsets no longer delete untransmitted paths outside the listed directories. diff --git a/src/shared/file_send.c b/src/shared/file_send.c index b5224bd..9f43724 100644 --- a/src/shared/file_send.c +++ b/src/shared/file_send.c @@ -185,7 +185,7 @@ bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_meta return false; } protocol_note_bytes_written((unsigned long long)sent); - protocol_throttle_bytes((size_t)sent); + protocol_throttle_bytes(file_descriptor, (size_t)sent); } close(fd); diff --git a/src/shared/protocol.c b/src/shared/protocol.c index fa1fff2..b6f8280 100644 --- a/src/shared/protocol.c +++ b/src/shared/protocol.c @@ -302,11 +302,14 @@ static ProtocolSession* legacy_session(int read_fd, int write_fd) { } /* Pace an out-of-band write that bypassed protocol_send_n_data (the plaintext - * sendfile fast path). The bound/legacy session is resolved exactly as - * send_n_data resolves it, so the same token-bucket state is throttled and the - * TLS and plaintext transports share identical --bwlimit semantics. */ -void protocol_throttle_bytes(size_t bytes) { - bw_throttle_session(legacy_session(-1, -1), bytes); + * sendfile fast path). The bound/legacy session is resolved exactly as the + * preceding send_n_data(fd, ...) resolved it, so the same token-bucket state is + * throttled and the TLS and plaintext transports share identical --bwlimit + * semantics. Passing the wire fd (rather than -1) is essential: the sendfile + * send left legacy_io_session.write_fd bound to it, so resolving with -1 would + * mismatch, re-initialize the session and hand out a second first-call burst. */ +void protocol_throttle_bytes(int file_descriptor, size_t bytes) { + bw_throttle_session(legacy_session(-1, file_descriptor), bytes); } bool send_n_data(int file_descriptor, const void* data, size_t data_size) { diff --git a/src/shared/protocol.h b/src/shared/protocol.h index d13c3f7..0bb0b42 100644 --- a/src/shared/protocol.h +++ b/src/shared/protocol.h @@ -225,11 +225,14 @@ unsigned long long protocol_bytes_written(void); unsigned long long protocol_bytes_read(void); void protocol_note_bytes_written(unsigned long long bytes); /* Apply --bwlimit pacing to bytes written outside protocol_send_n_data (the - * plaintext zero-copy sendfile fast path). Resolves the bound/legacy session - * exactly as send_n_data does and runs the same token-bucket throttle, so the - * sendfile transport is paced identically to the buffered/TLS paths. A no-op - * when the effective session has no bandwidth limit. */ -void protocol_throttle_bytes(size_t bytes); + * plaintext zero-copy sendfile fast path). `file_descriptor` is the wire fd + * the bytes were written to, so the legacy session is resolved exactly as the + * preceding send_n_data call resolved it (the bound TLS session still wins when + * set); resolving with the same fd avoids re-initializing the legacy session + * and granting a second first-call burst. Runs the same token-bucket throttle, + * so the sendfile transport is paced identically to the buffered/TLS paths. A + * no-op when the effective session has no bandwidth limit. */ +void protocol_throttle_bytes(int file_descriptor, size_t bytes); void protocol_session_init(ProtocolSession* session, int read_fd, int write_fd); /* Transitional bridge for helpers whose signatures still carry only an fd. */ diff --git a/tests/integration/test_temp_dir_exdev.py b/tests/integration/test_temp_dir_exdev.py new file mode 100644 index 0000000..d9236dd --- /dev/null +++ b/tests/integration/test_temp_dir_exdev.py @@ -0,0 +1,80 @@ +"""End-to-end coverage for the `--temp-dir` EXDEV (cross-filesystem) fallback. + +`file_to_disk_secure_impl` installs a completed temp file with `renameat(2)`; +when the scratch dir lives on a different filesystem the rename fails with +`EXDEV` and the engine retries with no scratch dir, writing the file directly in +the destination directory (a non-atomic copy), matching rsync. + +The daemon receiver confines `--temp-dir` to the authorized receive root, so a +genuine cross-fs scratch there would require an in-root mount point. Bind/tmpfs +mounting is not permitted in the CI container (no `CAP_SYS_ADMIN`, and +unprivileged user namespaces are disabled), so this test reaches the exact same +code path through the local `--read-batch` apply instead: it has no +authorized-root confinement, so a relative `--temp-dir` that is a symlink to a +tmpfs (`/dev/shm`) is accepted and the final install then crosses filesystems. +""" +import os +import shutil +import subprocess +import sys + +import pytest + +sys.path.insert(0, os.path.dirname(__file__)) +from common import CLIENT_CMD, get_dest_received_dir + +TMPFS = "/dev/shm" + + +def _run(args): + return subprocess.run(CLIENT_CMD + args, capture_output=True, text=True, timeout=180) + + +def _read(path): + with open(path, "rb") as fh: + return fh.read() + + +def test_read_batch_temp_dir_cross_filesystem_fallback(tmp_path): + if not os.path.isdir(TMPFS): + pytest.skip("no /dev/shm tmpfs available to force a cross-filesystem install") + + source = tmp_path / "src" + dest = tmp_path / "dst" + source.mkdir() + dest.mkdir() + files = { + "payload.bin": bytes(range(256)) * 64, + "sub/nested.txt": b"nested exdev fallback\n" * 8, + } + for rel, data in files.items(): + full = source / rel + full.parent.mkdir(parents=True, exist_ok=True) + full.write_bytes(data) + + batch = tmp_path / "tree.batch" + r = _run(["--only-write-batch", str(batch), str(source)]) + assert r.returncode == 0, (r.stdout, r.stderr) + + # A cross-filesystem scratch dir, reached through a relative --temp-dir + # symlink (the local batch apply performs no authorized-root confinement). + scratch = os.path.join(TMPFS, "fastsync_exdev_%d" % os.getpid()) + shutil.rmtree(scratch, ignore_errors=True) + os.makedirs(scratch) + os.symlink(scratch, dest / "scratch") + try: + assert os.stat(scratch).st_dev != os.stat(dest).st_dev, ( + "scratch and destination share a filesystem; EXDEV cannot be exercised" + ) + r = _run(["--read-batch", str(batch), str(dest), "--temp-dir=scratch"]) + assert r.returncode == 0, (r.stdout, r.stderr) + # The engine must report the non-atomic cross-fs fallback rather than + # silently claiming an atomic install. + assert "different filesystem" in (r.stdout + r.stderr), (r.stdout, r.stderr) + # The tree is still byte-exact and the scratch dir is left clean. + received = get_dest_received_dir(str(dest), str(source)) + for rel, data in files.items(): + assert _read(os.path.join(received, rel)) == data, f"content mismatch for {rel}" + assert os.listdir(scratch) == [], "cross-fs temp file was not cleaned up" + finally: + shutil.rmtree(scratch, ignore_errors=True) diff --git a/tests/test_protocol.c b/tests/test_protocol.c index 483ee82..3992256 100644 --- a/tests/test_protocol.c +++ b/tests/test_protocol.c @@ -1,6 +1,8 @@ #include "protocol.h" #include "test_utils.h" +#include #include +#include #include #include #include @@ -715,7 +717,7 @@ static void test_protocol_throttle_bytes_paces() { struct timespec start; clock_gettime(CLOCK_MONOTONIC, &start); - protocol_throttle_bytes(150000); + protocol_throttle_bytes(-1, 150000); struct timespec now; clock_gettime(CLOCK_MONOTONIC, &now); long long elapsed_ms = @@ -736,7 +738,7 @@ static void test_protocol_throttle_bytes_unlimited() { struct timespec start; clock_gettime(CLOCK_MONOTONIC, &start); - protocol_throttle_bytes(100000000ULL); + protocol_throttle_bytes(-1, 100000000ULL); struct timespec now; clock_gettime(CLOCK_MONOTONIC, &now); long long elapsed_ms = @@ -746,6 +748,44 @@ static void test_protocol_throttle_bytes_unlimited() { protocol_session_unbind(); } +/* Regression for the plaintext sendfile path: it calls protocol_throttle_bytes() + * immediately after send_n_data(), which already bound legacy_io_session.write_fd + * to the wire fd. Resolving the throttle session with (read=-1, write=-1) + * mismatched that fd and re-initialized the legacy session, granting a *second* + * first-call burst and discarding the accumulated debt. This drives the same + * sequence and asserts the debt from send_n_data carries into the throttle. */ +static void test_protocol_throttle_bytes_legacy_same_session() { + const size_t payload = 150000; /* 1.5x the 100 KB burst at --bwlimit=1 MB/s */ + unsigned char* buffer = malloc(payload); + EXPECT_TRUE(buffer != NULL); + memset(buffer, 0, payload); + + io_set_fds(-1, -1); + io_set_bwlimit(1000000ULL); + + int fd = open("/dev/null", O_WRONLY); + EXPECT_TRUE(fd >= 0); + + struct timespec start; + clock_gettime(CLOCK_MONOTONIC, &start); + /* send_n_data() consumes the whole 100 KB burst and sleeps ~50 ms. */ + EXPECT_TRUE(send_n_data(fd, buffer, payload)); + /* The throttle must share that session, so the 150 KB is all debt and sleeps + ~150 ms (total ~200 ms). A re-initialized session would hand out a fresh + 100 KB burst and sleep only ~50 ms (total ~100 ms). */ + protocol_throttle_bytes(fd, payload); + struct timespec now; + clock_gettime(CLOCK_MONOTONIC, &now); + long long elapsed_ms = + (now.tv_sec - start.tv_sec) * 1000LL + (now.tv_nsec - start.tv_nsec) / 1000000LL; + EXPECT_TRUE(elapsed_ms >= 150); + + close(fd); + free(buffer); + io_set_bwlimit(0); + io_set_fds(-1, -1); +} + void test_protocol() { test_send_receive_n_data(); test_send_receive_n_data_zero(); @@ -778,4 +818,5 @@ void test_protocol() { test_data_create_starts_uncharged_and_unowned(); test_protocol_throttle_bytes_paces(); test_protocol_throttle_bytes_unlimited(); + test_protocol_throttle_bytes_legacy_same_session(); } From e98729f00e5faab7cc4a55d8670e3ed4976e3345 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 14:24:39 +0200 Subject: [PATCH 09/10] refactor: const-correct delete-manifest API; apply clang-format --- src/client/client_send.c | 6 +++--- src/server/receiver.c | 2 +- src/shared/config.c | 3 ++- src/shared/delete_commit.c | 34 +++++++++++++++++----------------- src/shared/delete_commit.h | 27 +++++++++++++-------------- 5 files changed, 36 insertions(+), 36 deletions(-) diff --git a/src/client/client_send.c b/src/client/client_send.c index 2f3422b..3b9d623 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -1653,7 +1653,7 @@ static bool send_files_run(Config* config, SendFilesState* state) { /* Completion tail: send the late delete manifest and captured directory times, * finalize the receiver handshake, remove transferred sources and report stats. * Returns the rsync-compatible exit code. */ -static int send_files_finalize(Config* config, SendFilesState* state) { +static int send_files_finalize(const Config* config, SendFilesState* state) { Client* client = state->client; if (directory_scanner_failed(state->scanner)) return 1; @@ -1916,8 +1916,8 @@ int send_files_multithreaded(Config* config) { now_mono.tv_nsec = 0; } context->stop_condition = - stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, config->cli.stop_at_set, - config->stop_at, now_mono); + stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, + config->cli.stop_at_set, config->stop_at, now_mono); bool collect_excluded = config->use_delete && !config->delete_excluded; unsigned long long pre_scan_non_dir = 0; if (config->use_delete) { diff --git a/src/server/receiver.c b/src/server/receiver.c index 10c4c45..50f4e0b 100644 --- a/src/server/receiver.c +++ b/src/server/receiver.c @@ -384,7 +384,7 @@ static ReceiverStep receiver_handle_mkdir(ReceiverPendingState* state) { return RECEIVER_STEP_NEXT; } -static ReceiverStep receiver_handle_dir_times(ReceiverPendingState* state) { +static ReceiverStep receiver_handle_dir_times(const ReceiverPendingState* state) { if (!receiver_process_dir_times(state->fd, state->config, state->sink)) return RECEIVER_STEP_ERROR; return RECEIVER_STEP_NEXT; diff --git a/src/shared/config.c b/src/shared/config.c index eea3974..223daad 100644 --- a/src/shared/config.c +++ b/src/shared/config.c @@ -341,7 +341,8 @@ bool config_derived_use_metadata(const Config* config) { config->chown_uid_set || config->chown_gid_set || config->usermap_count > 0 || config->groupmap_count > 0 || config->update) return true; - return (config->use_incremental || config->use_delta) && !config->cli.metadata_explicitly_disabled; + return (config->use_incremental || config->use_delta) && + !config->cli.metadata_explicitly_disabled; } bool config_has_basis(const Config* config) { diff --git a/src/shared/delete_commit.c b/src/shared/delete_commit.c index 07f72db..0a2b66f 100644 --- a/src/shared/delete_commit.c +++ b/src/shared/delete_commit.c @@ -136,7 +136,7 @@ typedef struct { alternate basis directories are never destination content and are skipped at any depth. Returns true unless a traversal/unlink error aborted the walk; the budget's limit_hit/skipped fields report a cap-stopped run. */ -static bool delete_extras_budgeted_observed(const Config* config, DeleteManifest* manifest, +static bool delete_extras_budgeted_observed(const Config* config, const DeleteManifest* manifest, DeleteBudgetState* budget, DeletePathObserver observer, void* observer_context) { if (!config || !manifest || !manifest->keeps) @@ -177,7 +177,7 @@ static bool delete_extras_budgeted_observed(const Config* config, DeleteManifest return true; } -static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifest, +static bool delete_extras_budgeted(const Config* config, const DeleteManifest* manifest, DeleteBudgetState* budget) { return delete_extras_budgeted_observed(config, manifest, budget, NULL, NULL); } @@ -215,7 +215,8 @@ static void prefixed_delete_observer(void* context, const char* rel) { --max-delete budget: once it is exhausted the remaining requests are skipped and counted. Returns false only on a genuine error (a confinement failure on a validated path or an I/O error), which fails the run. */ -static bool delete_missing_args_budgeted_observed(const Config* config, DeleteManifest* manifest, +static bool delete_missing_args_budgeted_observed(const Config* config, + const DeleteManifest* manifest, DeleteBudgetState* budget, DeletePathObserver observer, void* observer_context) { @@ -376,8 +377,8 @@ static bool delete_missing_args_budgeted_observed(const Config* config, DeleteMa /* Public wrappers used outside the commit path (and by unit tests): no --max-delete budget. */ -bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, ArrayList* out, - size_t* count_out) { +bool manifest_would_delete_list(const Config* config, const DeleteManifest* manifest, + ArrayList* out, size_t* count_out) { if (count_out) *count_out = 0; if (!config || !manifest || !manifest->keeps || !out) @@ -391,30 +392,28 @@ bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, return ok; } -bool manifest_delete_extras(const Config* config, DeleteManifest* manifest) { +bool manifest_delete_extras(const Config* config, const DeleteManifest* manifest) { DeleteBudgetState budget = { .max_delete = SIZE_MAX, .deleted = 0, .skipped = 0, .limit_hit = false}; return delete_extras_budgeted(config, manifest, &budget); } -bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest) { +bool manifest_delete_missing_args(const Config* config, const DeleteManifest* manifest) { DeleteBudgetState budget = { .max_delete = SIZE_MAX, .deleted = 0, .skipped = 0, .limit_hit = false}; return delete_missing_args_budgeted_observed(config, manifest, &budget, NULL, NULL); } -bool manifest_delete_missing_args_limited(const Config* config, DeleteManifest* manifest, +bool manifest_delete_missing_args_limited(const Config* config, const DeleteManifest* manifest, size_t max_delete, size_t* deleted, size_t* skipped, bool* limit_hit) { return manifest_delete_missing_args_limited_observed(config, manifest, max_delete, deleted, skipped, limit_hit, NULL, NULL); } -bool manifest_delete_missing_args_limited_observed(const Config* config, DeleteManifest* manifest, - size_t max_delete, size_t* deleted, - size_t* skipped, bool* limit_hit, - DeletePathObserver observer, - void* observer_context) { +bool manifest_delete_missing_args_limited_observed( + const Config* config, const DeleteManifest* manifest, size_t max_delete, size_t* deleted, + size_t* skipped, bool* limit_hit, DeletePathObserver observer, void* observer_context) { DeleteBudgetState budget = { .max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false}; bool ok = @@ -435,17 +434,18 @@ bool manifest_delete_missing_args_limited_observed(const Config* config, DeleteM removal fail). The ordinary extras walk then runs when --delete is active. Both draw from one --max-delete budget; the result reports a cap-stopped (partial) commit distinctly so the client can exit 25 like rsync. */ -DeleteCommitResult manifest_delete_all(const Config* config, DeleteManifest* manifest) { +DeleteCommitResult manifest_delete_all(const Config* config, const DeleteManifest* manifest) { return manifest_delete_all_counted(config, manifest, NULL); } -DeleteCommitResult manifest_delete_all_counted(const Config* config, DeleteManifest* manifest, +DeleteCommitResult manifest_delete_all_counted(const Config* config, const DeleteManifest* manifest, size_t* deleted) { return manifest_delete_all_observed(config, manifest, deleted, NULL, NULL); } -DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteManifest* manifest, - size_t* deleted, DeletePathObserver observer, +DeleteCommitResult manifest_delete_all_observed(const Config* config, + const DeleteManifest* manifest, size_t* deleted, + DeletePathObserver observer, void* observer_context) { if (deleted) *deleted = 0; diff --git a/src/shared/delete_commit.h b/src/shared/delete_commit.h index 16f00ae..ba760f1 100644 --- a/src/shared/delete_commit.h +++ b/src/shared/delete_commit.h @@ -44,7 +44,7 @@ DeleteManifest* receive_manifest_entries(int fd); protected-prefix skips). `--max-delete` and `--force` are honored here. The caller decides WHEN to run it based on the negotiated delete timing. Returns false (and the transfer fails) when the deletion cannot be committed. */ -bool manifest_delete_extras(const Config* config, DeleteManifest* manifest); +bool manifest_delete_extras(const Config* config, const DeleteManifest* manifest); /* --delete-missing-args exact-path deletions: remove each destination mirror in `manifest->missing` (never blocked by the protected prefixes, staging dir and basis dirs excluded). A regular file/symlink is unlinked; an empty @@ -53,22 +53,20 @@ bool manifest_delete_extras(const Config* config, DeleteManifest* manifest); parity). A missing path is a no-op. Returns false only on a genuine confinement or I/O error (the run then fails); tolerated per-path cases are reported and skipped. */ -bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest); +bool manifest_delete_missing_args(const Config* config, const DeleteManifest* manifest); /* Budgeted form of manifest_delete_missing_args for the per-directory delete session: each removed mirror draws from `max_delete` (SIZE_MAX = unlimited) and the tallies are accumulated into `*deleted`/`*skipped`. `*limit_hit` is set when the budget stopped the pass with entries left over. Returns false only on a genuine deletion error. */ -bool manifest_delete_missing_args_limited(const Config* config, DeleteManifest* manifest, +bool manifest_delete_missing_args_limited(const Config* config, const DeleteManifest* manifest, size_t max_delete, size_t* deleted, size_t* skipped, bool* limit_hit); /* Observer-aware form of manifest_delete_missing_args_limited: `observer` (may be NULL) is invoked for every destination-relative path truly removed. */ -bool manifest_delete_missing_args_limited_observed(const Config* config, DeleteManifest* manifest, - size_t max_delete, size_t* deleted, - size_t* skipped, bool* limit_hit, - DeletePathObserver observer, - void* observer_context); +bool manifest_delete_missing_args_limited_observed( + const Config* config, const DeleteManifest* manifest, size_t max_delete, size_t* deleted, + size_t* skipped, bool* limit_hit, DeletePathObserver observer, void* observer_context); /* Outcome of committing a delete manifest. LIMIT_REACHED reports rsync's partial --max-delete result: the budget allowed some deletions and the rest were skipped (the run still stores all file data but the client exits 25). */ @@ -84,15 +82,16 @@ typedef enum { share one --max-delete budget. Returns DELETE_COMMIT_OK when nothing was to do or everything committed, DELETE_COMMIT_LIMIT_REACHED when the budget stopped part of the work, or DELETE_COMMIT_ERROR on a genuine failure. */ -DeleteCommitResult manifest_delete_all(const Config* config, DeleteManifest* manifest); +DeleteCommitResult manifest_delete_all(const Config* config, const DeleteManifest* manifest); /* Like manifest_delete_all, but reports how many destination entries the commit removed (for the end-of-transfer wire stats). `deleted` may be NULL. */ -DeleteCommitResult manifest_delete_all_counted(const Config* config, DeleteManifest* manifest, +DeleteCommitResult manifest_delete_all_counted(const Config* config, const DeleteManifest* manifest, size_t* deleted); /* Observer-aware form of manifest_delete_all_counted: `observer` (may be NULL) is invoked for every destination-relative path truly removed. */ -DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteManifest* manifest, - size_t* deleted, DeletePathObserver observer, +DeleteCommitResult manifest_delete_all_observed(const Config* config, + const DeleteManifest* manifest, size_t* deleted, + DeletePathObserver observer, void* observer_context); /* -n/--dry-run --delete would-delete reporting: walk the destination exactly as @@ -100,7 +99,7 @@ DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteMani WOULD be removed to `out`, without touching disk. Uses the same staging-dir, basis-dir and protected-prefix skips as the real commit. Returns true on a clean walk; `*count_out` receives the number of paths appended. */ -bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, ArrayList* out, - size_t* count_out); +bool manifest_would_delete_list(const Config* config, const DeleteManifest* manifest, + ArrayList* out, size_t* count_out); #endif From 1167e7970b2a60f955cb809ee795c5e2feec7df2 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 14:52:36 +0200 Subject: [PATCH 10/10] refactor(delete): rename basis helper to delete_basis_relative --- src/shared/delete.c | 4 ++-- src/shared/delete.h | 2 +- tests/test_file.c | 12 ++++++------ 3 files changed, 9 insertions(+), 9 deletions(-) diff --git a/src/shared/delete.c b/src/shared/delete.c index 3e4fb62..a47cc3b 100644 --- a/src/shared/delete.c +++ b/src/shared/delete.c @@ -550,7 +550,7 @@ bool delete_extras(const char* dest_root, const ArrayList* manifest) { its root-relative form, and one outside the root returns NULL (the walk cannot reach it, and it is not protected data beneath the root). Exposed so tests can exercise the root-of-"/" child mapping directly. */ -char* file_receive_basis_delete_relative(const Config* config, const char* path) { +char* delete_basis_relative(const Config* config, const char* path) { if (!path) return NULL; if (path[0] != '/') @@ -616,7 +616,7 @@ bool delete_skips_build(const Config* config, const ArrayList* protected_paths, if (basis_root_relative) { /* An absolute basis outside the receive root is unreachable by this walk, so it contributes no protection prefix (and no slot). */ - char* relative = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); + char* relative = delete_basis_relative(config, config->basis_dirs[i].path); if (!relative) continue; out->owned_prefixes[i] = relative; diff --git a/src/shared/delete.h b/src/shared/delete.h index 235db8c..f0921a1 100644 --- a/src/shared/delete.h +++ b/src/shared/delete.h @@ -146,6 +146,6 @@ void delete_skips_free(DeleteSkipSet* set); /* Convert one basis-directory path to the receive-root-relative protection prefix the delete walker uses (NULL when it lies outside the root). Exposed for unit tests of the root-of-"/" and normalization edge cases. */ -char* file_receive_basis_delete_relative(const Config* config, const char* path); +char* delete_basis_relative(const Config* config, const char* path); #endif diff --git a/tests/test_file.c b/tests/test_file.c index 69912f0..efb87a7 100644 --- a/tests/test_file.c +++ b/tests/test_file.c @@ -2255,26 +2255,26 @@ static void test_basis_delete_relative_root_slash() { EXPECT_NOT_NULL(cfg); cfg->receive_root_directory = str_dup("/"); - char* rel = file_receive_basis_delete_relative(cfg, "/a"); + char* rel = delete_basis_relative(cfg, "/a"); EXPECT_NOT_NULL(rel); EXPECT_EQ_STR(rel, "a"); free(rel); - rel = file_receive_basis_delete_relative(cfg, "/a/b"); + rel = delete_basis_relative(cfg, "/a/b"); EXPECT_NOT_NULL(rel); EXPECT_EQ_STR(rel, "a/b"); free(rel); /* The root itself is not a child. */ - EXPECT_NULL(file_receive_basis_delete_relative(cfg, "/")); + EXPECT_NULL(delete_basis_relative(cfg, "/")); /* A relative entry is already root-relative. */ - rel = file_receive_basis_delete_relative(cfg, "x/y"); + rel = delete_basis_relative(cfg, "x/y"); EXPECT_NOT_NULL(rel); EXPECT_EQ_STR(rel, "x/y"); free(rel); /* An absolute path outside a non-"/" root is unreachable. */ free(cfg->receive_root_directory); cfg->receive_root_directory = str_dup("/root"); - EXPECT_NULL(file_receive_basis_delete_relative(cfg, "/other/a")); - rel = file_receive_basis_delete_relative(cfg, "/root/a"); + EXPECT_NULL(delete_basis_relative(cfg, "/other/a")); + rel = delete_basis_relative(cfg, "/root/a"); EXPECT_NOT_NULL(rel); EXPECT_EQ_STR(rel, "a"); free(rel);