Merge feat/p3-basis-dest: alternate basis dirs (--compare-dest/--copy-dest/--link-dest)

This commit is contained in:
2026-09-06 20:10:40 +02:00
18 changed files with 1300 additions and 57 deletions
+58
View File
@@ -114,6 +114,27 @@ static int set_nonneg_int_option(int* dest, const char* value, const char* optio
return 0;
}
/* Validate and append one --compare-dest/--copy-dest/--link-dest directory.
* The path is interpreted on the receiver relative to the destination root,
* so it must be a non-empty relative path with no "." / ".." components (an
* absolute or escaping path is rejected up front instead of failing on the
* server). Returns 0 on success, -1 on error. */
static int set_basis_dest_option(Config* config, BasisDestType type, const char* value,
const char* option_name) {
if (!value || !value[0]) {
log_message(LOG_LEVEL_ERROR, "missing argument for %s", option_name);
return -1;
}
if (config_basis_append(config, type, value) != 0) {
log_message(LOG_LEVEL_ERROR,
"%s requires a non-empty relative directory name with no '.', '..', or absolute "
"path (resolved below the destination root)",
option_name);
return -1;
}
return 0;
}
static int set_stderr_mode(const char* value) {
if (strcmp(value, "errors") == 0 || strcmp(value, "e") == 0)
log_set_stderr_mode(LOG_STDERR_ERRORS);
@@ -974,6 +995,36 @@ int parse_args(Config* config, int argc, char* argv[], int* positional_args,
}
log_message(LOG_LEVEL_ERROR, "%s is not supported yet (xxHash64 is used)", argv[i]);
return -1;
} else if (strncmp(argv[i], "--compare-dest=", 15) == 0) {
if (set_basis_dest_option(config, BASIS_DEST_COMPARE, argv[i] + 15, "--compare-dest") != 0)
return -1;
} else if (opt_is(argv[i], "--compare-dest", NULL)) {
if (i + 1 >= argc) {
log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]);
return -1;
}
if (set_basis_dest_option(config, BASIS_DEST_COMPARE, argv[++i], "--compare-dest") != 0)
return -1;
} else if (strncmp(argv[i], "--copy-dest=", 12) == 0) {
if (set_basis_dest_option(config, BASIS_DEST_COPY, argv[i] + 12, "--copy-dest") != 0)
return -1;
} else if (opt_is(argv[i], "--copy-dest", NULL)) {
if (i + 1 >= argc) {
log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]);
return -1;
}
if (set_basis_dest_option(config, BASIS_DEST_COPY, argv[++i], "--copy-dest") != 0)
return -1;
} else if (strncmp(argv[i], "--link-dest=", 12) == 0) {
if (set_basis_dest_option(config, BASIS_DEST_LINK, argv[i] + 12, "--link-dest") != 0)
return -1;
} else if (opt_is(argv[i], "--link-dest", NULL)) {
if (i + 1 >= argc) {
log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]);
return -1;
}
if (set_basis_dest_option(config, BASIS_DEST_LINK, argv[++i], "--link-dest") != 0)
return -1;
} else if (argv[i][0] == '-') {
char* escaped = output_escape(argv[i], false);
fprintf(stderr, "Unknown option: %s\n", escaped ? escaped : "<allocation failed>");
@@ -1010,6 +1061,13 @@ int parse_args(Config* config, int argc, char* argv[], int* positional_args,
config->files_from_set = set;
}
/* The "unchanged" decision for --compare-dest/--copy-dest/--link-dest must
* be made on the receiver against the basis directories, which requires the
* per-file STATUS_CHECK handshake: basis-dir options therefore imply
* --incremental (and, via the block below, metadata) on the sender. */
if (config_has_basis(config))
config->use_incremental = true;
/* Incremental and delta transfers need metadata unless the user disabled it. */
if ((config->use_incremental || config->use_delta) && !config->use_metadata &&
!config->metadata_explicitly_disabled) {
+51 -1
View File
@@ -218,6 +218,49 @@ static bool files_from_list_valid(const Config* config) {
return no_implied_dirs_files_from_valid(config);
}
/* Basis directories are honored by the receiver's per-file incremental check,
which (like every whole-file payload path in FastSync) is bounded by
MAX_RECEIVE_WHOLE_FILE_SIZE. rsync would apply basis dirs to files of any
size; FastSync cannot, so when basis dirs are requested this preflight scan
refuses the run up front with a clear diagnostic instead of letting the
receiver abort the whole transfer mid-stream with no client explanation.
Returns true when the tree can be transferred. */
static bool basis_oversize_preflight(const Config* config) {
PreparedScanner prepared;
if (!prepare_scanner(config, 0, &prepared))
return false;
DirectoryScanner* scanner =
directory_scanner_create_with_options(config->send_directory, &prepared.options);
prepared_scanner_destroy(&prepared);
if (!scanner)
return false;
bool ok = true;
Chunk* chunk;
while ((chunk = directory_scanner_next(scanner)) != NULL) {
for (int i = 0; i < chunk->element_count; i++) {
File* f = chunk->items[i];
if (f == NULL || f->is_dir || f->data == NULL || f->data->size <= MAX_RECEIVE_WHOLE_FILE_SIZE)
continue;
char* escaped = output_escape(file_wire_path(f), config->eight_bit_output);
log_message(LOG_LEVEL_ERROR,
"%s is %llu bytes, larger than the %llu-byte whole-file transfer limit; "
"--compare-dest/--copy-dest/--link-dest cannot sync files above this limit",
escaped ? escaped : "<allocation failed>", (unsigned long long)f->data->size,
(unsigned long long)MAX_RECEIVE_WHOLE_FILE_SIZE);
free(escaped);
ok = false;
break;
}
chunk_destroy(chunk);
if (!ok)
break;
}
if (directory_scanner_failed(scanner))
ok = false;
directory_scanner_destroy(scanner);
return ok;
}
/* Select the configured transport for both transfer execution paths. */
static Client* connect_transfer_client(const Config* config) {
if (config->transport == TRANSPORT_SSH) {
@@ -662,7 +705,10 @@ static int incremental_check(Client* client, File* file, const Config* config,
return -1;
if (!send_n_data(client->file_descriptor, &mtime_nsec, sizeof(mtime_nsec)))
return -1;
if (config->checksum) {
/* With alternate basis directories the receiver must be able to verify the
* content of every candidate basis file, so the sender supplies its xxHash64
* for every file even when --checksum was not requested. */
if (config->checksum || config_has_basis(config)) {
uint64_t checksum;
if (!file_checksum(file, &checksum) ||
!send_n_data(client->file_descriptor, &checksum, sizeof(checksum)))
@@ -1216,6 +1262,8 @@ int send_files(Config* config) {
return send_dry_run_manifest(config);
if (!files_from_list_valid(config))
return 1;
if (config_has_basis(config) && !basis_oversize_preflight(config))
return 1;
Client* client = connect_transfer_client(config);
if (!client) {
@@ -1389,6 +1437,8 @@ int send_files_multithreaded(Config** config_ptr) {
return send_dry_run_manifest(config);
if (!files_from_list_valid(config))
return 1;
if (config_has_basis(config) && !basis_oversize_preflight(config))
return 1;
long pages = sysconf(_SC_AVPHYS_PAGES);
long page_size = sysconf(_SC_PAGE_SIZE);
+6
View File
@@ -11,6 +11,12 @@ bool validate_config(const Config* config) {
print_usage();
return false;
}
if (config_has_basis(config) && config->use_chunk_serialization) {
log_message(LOG_LEVEL_ERROR,
"--compare-dest/--copy-dest/--link-dest require per-file incremental checks and "
"cannot be combined with -s (chunk serialization)");
return false;
}
if (config->use_sendfile && (config->use_chunk_serialization || config->use_compression)) {
log_message(LOG_LEVEL_ERROR, "-f/--sendfile cannot be combined with -c (compression) or -s "
"(chunk serialization)");
+7
View File
@@ -70,6 +70,13 @@ void print_usage(void) {
printf(" -@, --modify-window <sec> Modification time tolerance\n");
printf(" -u, --update Skip files newer than the source on receiver\n");
printf(" --existing Skip files not already present at destination\n");
printf(" --compare-dest <dir> Treat DIR (relative to destination root) as an extra\n");
printf(" comparison basis: unchanged files are not transferred\n");
printf(" (requires --incremental, which is implied)\n");
printf(" --copy-dest <dir> Like --compare-dest, but copies the unchanged file from DIR\n");
printf(" into the destination instead of transferring its data\n");
printf(" --link-dest <dir> Like --copy-dest, but hard-links the unchanged file from DIR\n");
printf(" into the destination (repeatable; earlier DIRs win)\n");
printf(" --checksum-choice, --cc <alg> Checksum algorithm (not supported yet; xxHash64 is "
"used)\n");
printf(" --delta Delta transfer for changed files (requires --incremental)\n");
+1 -1
View File
@@ -302,7 +302,7 @@ static bool receiver_save_file(File* file, void* context_pointer) {
result = file_save_to_disk_full(context->config->receive_root_directory, file, context->config);
}
if (result != FILE_SAVE_ERROR && context->config->remove_source_files && !file->is_dir &&
!receiver_outcomes_append(&context->outcomes, (unsigned char)result)) {
!file->skip && !receiver_outcomes_append(&context->outcomes, (unsigned char)result)) {
file_destroy(file);
return false;
}
+127 -8
View File
@@ -109,9 +109,8 @@ static void config_set_defaults(Config* config) {
config->rsync_path = NULL;
config->old_args = false;
config->temp_dir = NULL;
config->compare_dest = NULL;
config->copy_dest = NULL;
config->link_dest = NULL;
config->basis_dirs = NULL;
config->basis_count = 0;
config->partial_dir = NULL;
config->suffix = NULL;
config->delete_before = false;
@@ -212,6 +211,87 @@ bool config_has_valid_delete_timing(const Config* config) {
return timing_count <= 1;
}
bool config_has_basis(const Config* config) {
return config && config->basis_count > 0;
}
/* A basis-dir path travels from the client to the receiver and is resolved
* below the destination root, so it must be a non-empty relative path with no
* "." or ".." component and no traversal: an absolute or escaping path would
* make the receiver read or link files outside its authorized root.
*
* Returns a malloc'd CANONICAL copy of an accepted path, or NULL when the path
* is rejected. Canonicalization collapses interior empty components ("a//b" ->
* "a/b"), drops "." components and trailing "/"s, so validation, the delete
* walker prefix match and the receiver's basis lookup all agree on one form.
* The normalizer is the single source of truth for both config_basis_path_valid
* and config_basis_append. */
static char* basis_path_normalize(const char* path) {
if (!path || path[0] == '\0' || path[0] == '/' || has_path_traversal(path))
return NULL;
if (strcmp(path, ".") == 0)
return NULL;
char* dup = str_dup(path);
if (!dup)
return NULL;
size_t out_len = 0;
char* out = malloc(strlen(path) + 1);
if (!out) {
free(dup);
return NULL;
}
char* saveptr = NULL;
bool ok = true;
for (char* part = strtok_r(dup, "/", &saveptr); part; part = strtok_r(NULL, "/", &saveptr)) {
if (strcmp(part, "..") == 0) {
ok = false;
break;
}
if (strcmp(part, ".") == 0)
continue;
if (out_len > 0)
out[out_len++] = '/';
size_t len = strlen(part);
memcpy(out + out_len, part, len);
out_len += len;
}
free(dup);
if (!ok || out_len == 0) {
free(out);
return NULL;
}
out[out_len] = '\0';
return out;
}
bool config_basis_path_valid(const char* path) {
char* normalized = basis_path_normalize(path);
if (!normalized)
return false;
free(normalized);
return true;
}
int config_basis_append(Config* config, BasisDestType type, const char* path) {
if (!config ||
(type != BASIS_DEST_COMPARE && type != BASIS_DEST_COPY && type != BASIS_DEST_LINK) ||
config->basis_count >= MAX_BASIS_DIRS)
return -1;
char* normalized = basis_path_normalize(path);
if (!normalized)
return -1;
BasisDest* grown = realloc(config->basis_dirs, (config->basis_count + 1) * sizeof(BasisDest));
if (!grown) {
free(normalized);
return -1;
}
config->basis_dirs = grown;
config->basis_dirs[config->basis_count].type = type;
config->basis_dirs[config->basis_count].path = normalized;
config->basis_count++;
return 0;
}
bool config_is_remote_dest(const char* s) {
if (s == NULL)
return false;
@@ -268,9 +348,13 @@ void config_delete(Config* config) {
free(config->rsh_command);
free(config->rsync_path);
free(config->temp_dir);
free(config->compare_dest);
free(config->copy_dest);
free(config->link_dest);
for (int i = 0; i < config->basis_count; i++) {
free(config->basis_dirs[i].path);
config->basis_dirs[i].path = NULL;
}
free(config->basis_dirs);
config->basis_dirs = NULL;
config->basis_count = 0;
free(config->partial_dir);
free(config->suffix);
free(config->address);
@@ -358,6 +442,17 @@ static bool send_resume_options(int fd, const Config* c) {
send_str(fd, c->chmod_spec ? c->chmod_spec : "") && send_skip_compress_options(fd, c);
}
static bool send_basis_options(int fd, const Config* c) {
if (!send_int(fd, c->basis_count))
return false;
for (int i = 0; i < c->basis_count; i++) {
if (!send_int(fd, (int)c->basis_dirs[i].type) ||
!send_str(fd, c->basis_dirs[i].path ? c->basis_dirs[i].path : ""))
return false;
}
return true;
}
static bool receive_core_fields(int fd, Config* c) {
int value;
if (!receive_wire_bool(fd, &c->eight_bit_output))
@@ -506,12 +601,35 @@ static bool receive_resume_options(int fd, Config* c) {
return true;
}
static bool receive_basis_options(int fd, Config* c) {
int count;
if (!receive_int(fd, &count))
return false;
if (count < 0 || count > MAX_BASIS_DIRS)
return false;
for (int i = 0; i < count; i++) {
int type;
if (!receive_int(fd, &type) || type <= BASIS_DEST_NONE || type > BASIS_DEST_LINK)
return false;
char* path = receive_str(fd);
if (!path)
return false;
/* config_basis_append validates and canonicalizes the path; a rejected
path (absolute / traversal / empty) drops the whole connection. */
bool ok = config_basis_append(c, (BasisDestType)type, path) == 0;
free(path);
if (!ok)
return false;
}
return true;
}
bool config_send(int file_descriptor, const Config* config) {
protocol_session_set_max_alloc(NULL, config->max_alloc);
if (!send_core_fields(file_descriptor, config) || !send_delta_fields(file_descriptor, config) ||
!send_file_options(file_descriptor, config) ||
!send_selection_options(file_descriptor, config) ||
!send_resume_options(file_descriptor, config))
!send_resume_options(file_descriptor, config) || !send_basis_options(file_descriptor, config))
return false;
Status status;
if (!receive_status(file_descriptor, &status))
@@ -543,7 +661,8 @@ Config* config_receive(int file_descriptor) {
!receive_delta_fields(file_descriptor, config) ||
!receive_file_options(file_descriptor, config) ||
!receive_selection_options(file_descriptor, config) ||
!receive_resume_options(file_descriptor, config))
!receive_resume_options(file_descriptor, config) ||
!receive_basis_options(file_descriptor, config))
goto error;
if (config->compress_choice[0] != '\0' && strcmp(config->compress_choice, "zstd") != 0 &&
strcmp(config->compress_choice, "none") != 0) {
+30 -3
View File
@@ -12,6 +12,22 @@ typedef enum { TRANSPORT_TCP, TRANSPORT_SSH } TransportType;
Config can carry it; the concrete type lives in delay_updates.h. */
typedef struct DelayUpdatesContext DelayUpdatesContext;
/* Alternate basis-directory modes (--compare-dest / --copy-dest /
* --link-dest). Each flag adds one entry to the ordered Config->basis_dirs
* list; the receiver consults entries in command-line order and stops at the
* first exact match, mirroring rsync's basis-dir priority rules. */
typedef enum {
BASIS_DEST_NONE = 0,
BASIS_DEST_COMPARE, /* compare only: never copies, never materializes */
BASIS_DEST_COPY, /* local copy of the matched basis file */
BASIS_DEST_LINK /* hard link to the matched basis file */
} BasisDestType;
typedef struct BasisDest {
BasisDestType type;
char* path; /* relative to the destination root (receiver-confined) */
} BasisDest;
typedef struct Config {
char* version;
char* send_directory;
@@ -134,9 +150,12 @@ typedef struct Config {
char* rsync_path;
bool old_args;
char* temp_dir;
char* compare_dest;
char* copy_dest;
char* link_dest;
/* Alternate basis directories, ordered by command-line appearance. Each
* entry's type selects compare/copy/link behavior on an exact match. These
* cross the wire so the receiver can consult them; they are interpreted
* relative to the destination root and confined there. */
BasisDest* basis_dirs;
int basis_count;
// PR #174: Partial transfer resumption
char* partial_dir;
@@ -189,6 +208,8 @@ typedef struct Config {
#define PROTOCOL_VERSION "2.8.0"
#define DEFAULT_CHUNK_SIZE (10 * 1024 * 1024)
/* Upper bound on total basis-dir entries (rsync caps --link-dest at 20). */
#define MAX_BASIS_DIRS 64
Config* config_create(void);
void config_delete(Config* config);
@@ -208,5 +229,11 @@ bool config_delete_timing_early(const Config* config);
* set (none = the default delete-after commit timing); without deletion no
* timing flag may be set (each timing flag implies --delete). */
bool config_has_valid_delete_timing(const Config* config);
/* True when at least one --compare-dest/--copy-dest/--link-dest was set. */
bool config_has_basis(const Config* config);
/* Append one basis-dir entry. Returns 0 on success, -1 on allocation failure. */
int config_basis_append(Config* config, BasisDestType type, const char* path);
/* Validate a client-provided basis-dir path (relative, confined, non-empty). */
bool config_basis_path_valid(const char* path);
#endif
+103
View File
@@ -83,6 +83,7 @@ File* file_create(const char* path) {
file->metadata = NULL;
file->skip = false;
file->is_dir = false;
file->basis_link = NULL;
return file;
}
@@ -98,6 +99,8 @@ void file_destroy(void* item) {
file->path = NULL;
free(file->send_path);
file->send_path = NULL;
free(file->basis_link);
file->basis_link = NULL;
free(file);
}
@@ -630,6 +633,106 @@ bool file_to_disk_secure_no_replace(const char* path, const void* data,
preserve_executability, false, true, false, temp_dir);
}
/* Atomic --link-dest install. The destination is replaced (via a temporary
* name and a final rename) with a hard link to `basis_path`. When a hard
* link cannot be created (the basis lives on a different filesystem, the
* filesystem refuses hard links, ...) the install falls back to writing a
* local copy from `data`/`data_size`, which the caller has already verified is
* byte-identical to the basis file. `metadata` is only applied on that copy
* fallback; a successful hard link keeps the basis inode's own attributes
* (applying metadata through the shared inode would mutate the basis file).
* Returns false only when both the link and the copy fallback fail. */
bool file_to_disk_secure_link(const char* path, const char* basis_path, const void* data,
unsigned long long data_size, const FileMetadata* metadata,
bool preserve_executability, bool use_fsync, const char* temp_dir) {
if (!path || !basis_path)
return false;
char* leaf = NULL;
int dirfd = file_open_secure_parent(path, &leaf, true);
if (dirfd < 0)
return false;
int scratch_dirfd = -1;
if (temp_dir) {
scratch_dirfd = file_open_private_dir(temp_dir);
if (scratch_dirfd < 0) {
int saved_errno = errno;
log_message(LOG_LEVEL_ERROR, "could not open --temp-dir scratch directory '%s': %s", temp_dir,
strerror(saved_errno));
close(dirfd);
free(leaf);
return false;
}
}
char* basis_leaf = NULL;
int basis_dirfd = file_open_secure_parent(basis_path, &basis_leaf, false);
bool linked = false;
if (basis_dirfd >= 0 && basis_leaf != NULL) {
int tmp_size = snprintf(NULL, 0, ".%s.tmp.%ld.%llu", leaf, (long)getpid(), ~0ULL);
char* tmp = NULL;
if (tmp_size >= 0)
tmp = malloc((size_t)tmp_size + 1);
if (!tmp) {
log_message(LOG_LEVEL_ERROR, "memory allocation failed while hard-linking basis file");
} else {
for (unsigned int i = 0; i < 100 && !linked; ++i) {
if (scratch_dirfd >= 0)
snprintf(tmp, (size_t)tmp_size + 1, ".%s.tmp.%ld.%llu", leaf, (long)getpid(),
next_temp_sequence());
else
snprintf(tmp, (size_t)tmp_size + 1, ".%s.tmp.%ld.%u", leaf, (long)getpid(), i);
if (linkat(basis_dirfd, basis_leaf, scratch_dirfd >= 0 ? scratch_dirfd : dirfd, tmp, 0) ==
0) {
linked = true;
break;
}
if (errno != EEXIST)
break; /* EXDEV / EPERM / ...: give up and fall back to a copy */
}
if (linked) {
int target_dirfd = scratch_dirfd >= 0 ? scratch_dirfd : dirfd;
if (use_fsync) {
int tfd = openat(target_dirfd, tmp, O_RDONLY | O_NOFOLLOW | O_CLOEXEC);
if (tfd < 0 || fsync(tfd) != 0) {
linked = false;
if (tfd >= 0)
close(tfd);
} else {
close(tfd);
}
}
if (linked && renameat(target_dirfd, tmp, dirfd, leaf) != 0)
linked = false;
if (!linked)
unlinkat(target_dirfd, tmp, 0);
}
free(tmp);
}
}
if (basis_dirfd >= 0)
close(basis_dirfd);
free(basis_leaf);
basis_leaf = NULL;
if (!linked) {
if (scratch_dirfd >= 0)
close(scratch_dirfd);
close(dirfd);
free(leaf);
/* The basis file could not be linked in (missing, cross-device, refused
by the filesystem). Write a byte-identical local copy instead. */
return file_to_disk_secure_with_fsync(path, data, data_size, false, false, metadata,
preserve_executability, use_fsync, temp_dir);
}
if (scratch_dirfd >= 0)
close(scratch_dirfd);
close(dirfd);
free(leaf);
return true;
}
bool file_write_to_disk(const char* path, const void* data, unsigned long long data_size,
bool inplace, bool sparse) {
if (!path || (!data && data_size != 0) || has_path_traversal(path))
+7
View File
@@ -63,5 +63,12 @@ bool file_to_disk_secure_no_replace(const char* path, const void* data,
unsigned long long data_size, bool sparse,
const FileMetadata* metadata, bool preserve_executability,
const char* temp_dir);
/* Atomic --link-dest install: replace `path` with a hard link to `basis_path`
(via a temp name + rename); fall back to a byte-identical local copy from
`data` when the link is impossible (EXDEV/EPERM/unsupported filesystem).
`metadata` is applied only on the copy fallback. */
bool file_to_disk_secure_link(const char* path, const char* basis_path, const void* data,
unsigned long long data_size, const FileMetadata* metadata,
bool preserve_executability, bool use_fsync, const char* temp_dir);
#endif
+275 -18
View File
@@ -75,10 +75,17 @@ static FileSaveResult file_stage_delayed_update(const char* root_directory,
/* The staged location is brand new (stale leftovers from a prior crash were
wiped by prepare), so the plain atomic temp+rename engine installs the
complete file there. --temp-dir scratch is deliberately not layered on
top of the delay-updates staging tree. */
bool ok =
file_to_disk_secure_with_fsync(staged_path, file->data->data, file->data->size, false, sparse,
metadata, preserve_executability, config->use_fsync, NULL);
top of the delay-updates staging tree. A --link-dest basis file is hard
linked into the staging tree (so publication's rename keeps the link). */
bool ok;
if (file->basis_link) {
ok = file_to_disk_secure_link(staged_path, file->basis_link, file->data->data, file->data->size,
metadata, preserve_executability, config->use_fsync, NULL);
} else {
ok = file_to_disk_secure_with_fsync(staged_path, file->data->data, file->data->size, false,
sparse, metadata, preserve_executability, config->use_fsync,
NULL);
}
if (!ok) {
free(staged_path);
return FILE_SAVE_ERROR;
@@ -272,16 +279,27 @@ FileSaveResult file_save_to_disk_full(const char* root_directory, const File* fi
while (temp_len > 1 && confined_temp[temp_len - 1] == '/')
confined_temp[--temp_len] = '\0';
}
bool ok =
config && config->ignore_existing
? file_to_disk_secure_no_replace(disk_path, file->data->data, file->data->size, sparse,
metadata, preserve_executability, confined_temp)
: config && config->update
? file_to_disk_secure_update(disk_path, file->data->data, file->data->size, inplace,
sparse, metadata, preserve_executability, confined_temp)
: file_to_disk_secure_with_fsync(disk_path, file->data->data, file->data->size, inplace,
sparse, metadata, preserve_executability,
config && config->use_fsync, confined_temp);
/* A --link-dest basis hit installs an atomic hard link (with a byte-copy
fallback); --inplace and the update/no-replace write variants do not
apply to a fresh hard link, whose inode attributes already match. The
existing/ignore-existing/update/backup preamble above has already made the
policy decision. */
bool ok;
if (config && file->basis_link) {
ok = file_to_disk_secure_link(disk_path, file->basis_link, file->data->data, file->data->size,
metadata, preserve_executability, config->use_fsync,
confined_temp);
} else {
ok = config && config->ignore_existing
? file_to_disk_secure_no_replace(disk_path, file->data->data, file->data->size, sparse,
metadata, preserve_executability, confined_temp)
: config && config->update
? file_to_disk_secure_update(disk_path, file->data->data, file->data->size, inplace,
sparse, metadata, preserve_executability, confined_temp)
: file_to_disk_secure_with_fsync(disk_path, file->data->data, file->data->size,
inplace, sparse, metadata, preserve_executability,
config && config->use_fsync, confined_temp);
}
free(confined_temp);
confined_temp = NULL;
if (!ok)
@@ -505,6 +523,143 @@ static File* receive_delta_file(int fd, const Config* config, const char* check_
return NULL;
}
/* ---- Alternate basis directories (--compare-dest / --copy-dest / --link-dest) ----
* The receiver consults the ordered basis-dir list only when the destination
* entry is NOT already up to date. An "exact match" requires an equal size,
* an equal mtime (unless --size-only), and an equal content xxHash64, so a
* hard link / local copy is only ever made from byte-identical content. */
typedef struct BasisMatch {
bool hit;
BasisDestType type;
char* basis_path; /* owned absolute path of the matched basis file */
struct stat st; /* fstat() of the matched basis file */
Data* content; /* owned basis bytes (or empty Data), NULL when not loaded */
} BasisMatch;
static void basis_match_free(BasisMatch* match) {
if (!match)
return;
free(match->basis_path);
match->basis_path = NULL;
data_destroy(match->content);
match->content = NULL;
match->hit = false;
match->type = BASIS_DEST_NONE;
}
/* Open `path` (via the secure, root-confined primitives) and require it to be
a regular file of exactly `expected_size` bytes. Returns an open read-only
descriptor and its fstat on success. */
static bool basis_open_regular(const char* path, unsigned long long expected_size, int* out_fd,
struct stat* out_st) {
char* leaf = NULL;
int parent_fd = file_open_secure_parent(path, &leaf, false);
if (parent_fd < 0)
return false;
int fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW);
free(leaf);
close(parent_fd);
if (fd < 0)
return false;
struct stat st;
if (fstat(fd, &st) != 0 || !S_ISREG(st.st_mode) ||
(unsigned long long)st.st_size != expected_size) {
close(fd);
return false;
}
*out_fd = fd;
*out_st = st;
return true;
}
/* Read the whole remaining content of an open descriptor. A zero-length file
yields an empty Data (data pointer NULL). */
static Data* basis_read_content(int fd, unsigned long long size) {
if (size == 0)
return data_create_reserve(0);
if (size > MAX_RECEIVE_WHOLE_FILE_SIZE || size > SIZE_MAX)
return NULL;
void* buf = protocol_alloc((size_t)size);
if (!buf)
return NULL;
size_t got = 0;
while (got < (size_t)size) {
ssize_t n = read(fd, (char*)buf + got, (size_t)size - got);
if (n <= 0) {
free(buf);
return NULL;
}
got += (size_t)n;
}
return data_create(buf, (size_t)size);
}
/* --ignore-times forces every file to be updated, so no basis hit is ever
declared (matching rsync, where -I prevents link-dest from linking). */
static bool basis_quick_matches(const Config* config, const struct stat* st, time_t check_mtime,
long check_mtime_nsec) {
if (config->size_only)
return true;
long mtime_nsec = 0;
#ifdef __linux__
mtime_nsec = st->st_mtim.tv_nsec;
#endif
return metadata_mtime_matches(st->st_mtime, mtime_nsec, check_mtime, check_mtime_nsec,
config->modify_window);
}
/* Search the basis-dir list in command-line order and return the first exact
match. When load_content is true the matched bytes are kept in out->content
so the caller can materialize the file without re-reading it. */
static bool basis_match_find(const Config* config, const char* check_path,
unsigned long long check_size, time_t check_mtime,
long check_mtime_nsec, uint64_t check_checksum, bool load_content,
BasisMatch* out) {
memset(out, 0, sizeof(*out));
if (!config || !config_has_basis(config) || config->ignore_times)
return false;
for (int i = 0; i < config->basis_count; i++) {
const BasisDest* entry = &config->basis_dirs[i];
char* basis_dir = path_cat(config->receive_root_directory, entry->path);
if (!basis_dir)
continue;
char* candidate = path_cat(basis_dir, check_path);
free(basis_dir);
if (!candidate)
continue;
int fd;
struct stat st;
if (basis_open_regular(candidate, check_size, &fd, &st)) {
if (basis_quick_matches(config, &st, check_mtime, check_mtime_nsec)) {
Data* content = basis_read_content(fd, check_size);
if (content) {
uint64_t basis_hash = check_size == 0 ? delta_xxhash64("", 0)
: content->data ? delta_xxhash64(content->data, content->size)
: 0;
if (basis_hash == check_checksum) {
out->hit = true;
out->type = entry->type;
out->basis_path = candidate;
candidate = NULL; /* ownership transferred to out */
out->st = st;
out->content = load_content ? content : NULL;
if (!load_content)
data_destroy(content);
close(fd);
return true;
}
}
data_destroy(content);
}
close(fd);
}
free(candidate);
}
return false;
}
File* receive_incremental_check(int fd, const Config* config, bool* skipped) {
if (!config || !skipped) {
send_status(fd, STATUS_ERROR);
@@ -531,7 +686,8 @@ File* receive_incremental_check(int fd, const Config* config, bool* skipped) {
send_status(fd, STATUS_ERROR);
return NULL;
}
if (config->checksum && !receive_n_data(fd, &check_checksum, sizeof(check_checksum))) {
if ((config->checksum || config_has_basis(config)) &&
!receive_n_data(fd, &check_checksum, sizeof(check_checksum))) {
free(check_path);
return NULL;
}
@@ -642,6 +798,78 @@ File* receive_incremental_check(int fd, const Config* config, bool* skipped) {
return NULL;
}
/* ---- Alternate basis directories ---- */
if (config_has_basis(config)) {
BasisMatch basis;
basis_match_find(config, check_path, check_size, (time_t)check_mtime, (long)check_mtime_nsec,
check_checksum, true, &basis);
if (basis.hit) {
if (basis.type == BASIS_DEST_COMPARE) {
/* compare-dest never copies: an exact match only suppresses the data
for a file the destination does not already hold (sparse backup).
When the destination holds a DIFFERENT version FastSync falls back to
a normal transfer rather than deleting the stale entry the way rsync
does (see RSYNC_COMPAT.md). */
basis_match_free(&basis);
if (!has_old_file) {
if (!send_status(fd, STATUS_OK)) {
close(old_fd);
free(full_path);
free(check_path);
return NULL;
}
free(old_data);
close(old_fd);
free(full_path);
free(check_path);
*skipped = true;
return NULL;
}
} else {
/* copy-dest / link-dest: materialize the unchanged file locally so the
sender can skip the data. The store engine re-applies the normal
existing/ignore-existing/update/backup/delay-updates policy. */
File* materialized = file_create(check_path);
if (materialized && basis.content) {
materialized->data = basis.content;
basis.content = NULL;
materialized->metadata = file_metadata_create(&basis.st);
materialized->skip = true; /* receiver must not ack this as a data file */
if (basis.type == BASIS_DEST_LINK) {
materialized->basis_link = basis.basis_path;
basis.basis_path = NULL;
}
if (!materialized->metadata) {
file_destroy(materialized);
materialized = NULL;
}
} else {
file_destroy(materialized);
materialized = NULL;
}
if (materialized) {
if (!send_status(fd, STATUS_OK)) {
basis_match_free(&basis);
file_destroy(materialized);
close(old_fd);
free(full_path);
free(check_path);
return NULL;
}
basis_match_free(&basis);
free(old_data);
close(old_fd);
free(full_path);
free(check_path);
*skipped = false;
return materialized;
}
/* Materialization setup failed: fall through to the normal transfer. */
}
}
basis_match_free(&basis);
}
if (try_delta && old_data != NULL) {
bool delta_failed = false;
File* delta_file =
@@ -839,7 +1067,36 @@ bool manifest_delete_extras(const Config* config, ArrayList* manifest) {
if (!config || !manifest)
return false;
fprintf(stderr, "Deleting files not in manifest...\n");
const char* skip_staging = config->delay_updates ? DELAY_UPDATES_STAGING_DIR : NULL;
return delete_extras_limited(config->receive_root_directory, manifest, MAX_SERVER_DELETE_COUNT,
skip_staging);
/* With --delay-updates the staged (not yet published) files live directly
under the receive root in the staging directory; the delete walker must
not treat them as extras or it would remove every staged file before it
can be published. That staging name is protected only as a DIRECT child
of the receive root so a nested destination directory that happens to be
named .fastsync-stage is still ordinary content. Alternate basis
directories (--compare-dest / --copy-dest / --link-dest) are excluded at
any depth: they are extra comparison snapshots the user pointed at, not
destination content, and deleting them would destroy the very files a
--link-dest run just linked into place. */
int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count;
DeleteSkipEntry* skips = NULL;
if (skip_count > 0) {
skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry));
if (!skips)
return false;
int idx = 0;
if (config->delay_updates) {
skips[idx].prefix = DELAY_UPDATES_STAGING_DIR;
skips[idx].top_level_only = true;
idx++;
}
for (int i = 0; i < config->basis_count; i++) {
skips[idx].prefix = config->basis_dirs[i].path;
skips[idx].top_level_only = false;
idx++;
}
}
bool deletion_ok = delete_extras_limited(config->receive_root_directory, manifest,
MAX_SERVER_DELETE_COUNT, skips, skip_count);
free(skips);
return deletion_ok;
}
+6
View File
@@ -29,6 +29,12 @@ typedef struct {
/* True when this entry is an explicit directory entry (--dirs mode): the
* receiver creates the directory instead of writing a regular file. */
bool is_dir;
/* Receiver-only, --link-dest: when set, install the destination entry as a
* hard link to this absolute (root-confined) path instead of writing
* `data`. The matching code has already verified the link target's content
* equals the incoming file, and `data` is kept as the cross-filesystem
* fallback (a local copy) if the hard link cannot be created. */
char* basis_link;
} File;
/* The path that should be sent on the wire and used for the receiver-side
+1 -1
View File
@@ -302,7 +302,7 @@ int write_thread(void* pipeline_context) {
/* Record the per-file outcome so a --remove-source-files sender learns
which sources were actually written versus skipped on the receiver.
Explicit directory entries have no source and are never acknowledged. */
if (context->config->remove_source_files && !file->is_dir &&
if (context->config->remove_source_files && !file->is_dir && !file->skip &&
!receiver_outcomes_append(&context->outcomes, (unsigned char)result)) {
file_destroy(file);
pipeline_context_receiver_note_bytes_released(context, file_bytes);
+34 -13
View File
@@ -204,9 +204,26 @@ static bool is_dir_in_manifest(const char* rel_path, ArrayList* manifest) {
return false;
}
/* True when child_rel is, or lies below, a protected entry. A prefix "a"
therefore protects "a" and "a/b/c" but not "ab". Entries with top_level_only
set only protect DIRECT children of the receive root (at_root); nested
directories that share such a name stay ordinary destination content. */
static bool path_under_skip_prefix(const char* child_rel, bool at_root,
const DeleteSkipEntry* skips, int skip_count) {
for (int i = 0; i < skip_count; i++) {
if (skips[i].top_level_only && !at_root)
continue;
size_t prefix_len = strlen(skips[i].prefix);
if (strncmp(child_rel, skips[i].prefix, prefix_len) == 0 &&
(child_rel[prefix_len] == '\0' || child_rel[prefix_len] == '/'))
return true;
}
return false;
}
static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifest,
size_t max_delete, size_t* deleted_count,
const char* skip_root_child) {
size_t max_delete, size_t* deleted_count, const DeleteSkipEntry* skips,
int skip_count) {
int scanfd = dup(dirfd);
if (scanfd < 0)
return false;
@@ -220,18 +237,22 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifes
while ((entry = readdir(dir)) != NULL) {
if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0)
continue;
/* A --delay-updates run keeps its staging directory as a direct child of
the receive root. Its contents are not manifest entries yet (they are
published after deletion), so descending into it would delete every
staged file as an "extra". Skip only the top-level staging name; nested
directories with the same name are ordinary destination content. */
if (rel_path[0] == '\0' && skip_root_child && strcmp(entry->d_name, skip_root_child) == 0)
continue;
char* child_rel = path_cat((char*)rel_path, entry->d_name);
if (!child_rel) {
operation_ok = false;
continue;
}
/* A --delay-updates run keeps its staging directory as a direct child of
the receive root, and basis-dir snapshots live below it too. Their
contents are not manifest entries, so descending into them would delete
every staged / basis file as an "extra". Only the staging name (a
top-level-only prefix) and the basis prefixes are protected: a nested
destination directory that happens to be called .fastsync-stage is
ordinary content. */
if (path_under_skip_prefix(child_rel, rel_path[0] == '\0', skips, skip_count)) {
free(child_rel);
continue;
}
struct stat st;
if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) {
if (errno != ENOENT)
@@ -249,7 +270,7 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifes
bool child_removed = false;
if (childfd >= 0) {
child_removed = delete_extras_fd(childfd, child_rel, manifest, max_delete, deleted_count,
skip_root_child);
skips, skip_count);
if (!child_removed)
operation_ok = false;
close(childfd);
@@ -301,7 +322,7 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifes
}
bool delete_extras_limited(const char* dest_root, ArrayList* manifest, size_t max_delete,
const char* skip_root_child) {
const DeleteSkipEntry* skips, int skip_count) {
if (!manifest)
return false;
int rootfd;
@@ -318,14 +339,14 @@ bool delete_extras_limited(const char* dest_root, ArrayList* manifest, size_t ma
if (rootfd < 0)
return false;
size_t deleted_count = 0;
bool ok = delete_extras_fd(rootfd, "", manifest, max_delete, &deleted_count, skip_root_child);
bool ok = delete_extras_fd(rootfd, "", manifest, max_delete, &deleted_count, skips, skip_count);
if (close(rootfd) != 0)
ok = false;
return ok;
}
bool delete_extras(const char* dest_root, ArrayList* manifest) {
return delete_extras_limited(dest_root, manifest, SIZE_MAX, NULL);
return delete_extras_limited(dest_root, manifest, SIZE_MAX, NULL, 0);
}
bool has_path_traversal(const char* path) {
+14 -5
View File
@@ -10,12 +10,21 @@ char* output_escape(const char* string, bool eight_bit_output);
char* path_cat(const char* path1, const char* path2);
bool glob_match(const char* pattern, const char* str);
bool delete_extras(const char* dest_root, ArrayList* manifest);
/* Remove files/dirs under dest_root that are not listed in manifest. When
skip_root_child is non-NULL, a direct child of dest_root with that exact
name is left untouched (used to protect the --delay-updates staging
directory, which holds files that are still to be published). */
/* One protected entry for the delete walker. When top_level_only is true the
prefix is skipped only as a DIRECT child of dest_root (the --delay-updates
staging directory, which must not hide genuine extras inside a nested
destination directory that happens to share the staging name); otherwise the
prefix is skipped at any depth (the --compare-dest/--copy-dest/--link-dest
basis trees, which the transfer links from and are never destination
content). */
typedef struct {
const char* prefix;
bool top_level_only;
} DeleteSkipEntry;
/* Remove files/dirs under dest_root that are not listed in manifest without
ever descending into a protected prefix (see DeleteSkipEntry). */
bool delete_extras_limited(const char* dest_root, ArrayList* manifest, size_t max_delete,
const char* skip_root_child);
const DeleteSkipEntry* skips, int skip_count);
bool utils_set_authorized_root(int fd, const char* canonical_path);
/* The fd-only compatibility form is fail-closed for path-based operations;
* callers should use utils_set_authorized_root with the canonical identity. */