feat: receiver-side basis matching with link/copy materialization

The receiver's per-file incremental check now consults the ordered basis-dir
list whenever the destination is not already up to date.  An exact basis match
requires equal size, equal mtime (unless --size-only; --ignore-times disables
basis matching like rsync), and an equal content xxHash64 -- the sender sends
its xxHash for every file whenever basis dirs are configured (not only under
--checksum), so a hard link or local copy is only ever made from byte-identical
content.

On a match:
  - compare-dest: reply STATUS_OK and skip data only when the destination does
    not already hold the file (sparse, rsync parity).  A destination that holds
    a DIFFERENT version falls back to a normal transfer instead of rsync's
    delete, keeping the mirror complete.
  - copy-dest: reply STATUS_OK and hand a synthetic File (bytes read from the
    basis file, basis metadata) to the normal store sink, so the file is
    installed as a real local copy through the existing atomic temp+rename
    engine and honors --existing/--ignore-existing/--update/--backup/
    --delay-updates/--partial-dir unchanged.
  - link-dest: same, but File.basis_link records the basis path and the store
    engine calls the new file_to_disk_secure_link(): an atomic temp hard link +
    rename.  Cross-filesystem/refused links fall back to a byte-identical local
    copy (never a corrupt or partial file); the copy fallback applies metadata,
    while a successful link keeps the basis inode's own attributes so the basis
    file is never mutated.

Basis-materialized files carry File.skip so they are not acknowledged to a
--remove-source-files sender (the sender already saw STATUS_OK and keeps the
source).  The no-match path is byte-for-byte identical to the existing delta /
full-data transfer.
This commit is contained in:
2026-09-06 18:47:26 +02:00
parent df890f76ea
commit d38920c972
7 changed files with 386 additions and 22 deletions
+4 -1
View File
@@ -613,7 +613,10 @@ static int incremental_check(Client* client, File* file, const Config* config,
return -1; return -1;
if (!send_n_data(client->file_descriptor, &mtime_nsec, sizeof(mtime_nsec))) if (!send_n_data(client->file_descriptor, &mtime_nsec, sizeof(mtime_nsec)))
return -1; return -1;
if (config->checksum) { /* With alternate basis directories the receiver must be able to verify the
* content of every candidate basis file, so the sender supplies its xxHash64
* for every file even when --checksum was not requested. */
if (config->checksum || config_has_basis(config)) {
uint64_t checksum; uint64_t checksum;
if (!file_checksum(file, &checksum) || if (!file_checksum(file, &checksum) ||
!send_n_data(client->file_descriptor, &checksum, sizeof(checksum))) !send_n_data(client->file_descriptor, &checksum, sizeof(checksum)))
+1 -1
View File
@@ -214,7 +214,7 @@ static bool receiver_save_file(File* file, void* context_pointer) {
result = file_save_to_disk_full(context->config->receive_root_directory, file, context->config); result = file_save_to_disk_full(context->config->receive_root_directory, file, context->config);
} }
if (result != FILE_SAVE_ERROR && context->config->remove_source_files && !file->is_dir && if (result != FILE_SAVE_ERROR && context->config->remove_source_files && !file->is_dir &&
!receiver_outcomes_append(&context->outcomes, (unsigned char)result)) { !file->skip && !receiver_outcomes_append(&context->outcomes, (unsigned char)result)) {
file_destroy(file); file_destroy(file);
return false; return false;
} }
+103
View File
@@ -83,6 +83,7 @@ File* file_create(const char* path) {
file->metadata = NULL; file->metadata = NULL;
file->skip = false; file->skip = false;
file->is_dir = false; file->is_dir = false;
file->basis_link = NULL;
return file; return file;
} }
@@ -98,6 +99,8 @@ void file_destroy(void* item) {
file->path = NULL; file->path = NULL;
free(file->send_path); free(file->send_path);
file->send_path = NULL; file->send_path = NULL;
free(file->basis_link);
file->basis_link = NULL;
free(file); free(file);
} }
@@ -630,6 +633,106 @@ bool file_to_disk_secure_no_replace(const char* path, const void* data,
preserve_executability, false, true, false, temp_dir); preserve_executability, false, true, false, temp_dir);
} }
/* Atomic --link-dest install. The destination is replaced (via a temporary
* name and a final rename) with a hard link to `basis_path`. When a hard
* link cannot be created (the basis lives on a different filesystem, the
* filesystem refuses hard links, ...) the install falls back to writing a
* local copy from `data`/`data_size`, which the caller has already verified is
* byte-identical to the basis file. `metadata` is only applied on that copy
* fallback; a successful hard link keeps the basis inode's own attributes
* (applying metadata through the shared inode would mutate the basis file).
* Returns false only when both the link and the copy fallback fail. */
bool file_to_disk_secure_link(const char* path, const char* basis_path, const void* data,
unsigned long long data_size, const FileMetadata* metadata,
bool preserve_executability, bool use_fsync, const char* temp_dir) {
if (!path || !basis_path)
return false;
char* leaf = NULL;
int dirfd = file_open_secure_parent(path, &leaf, true);
if (dirfd < 0)
return false;
int scratch_dirfd = -1;
if (temp_dir) {
scratch_dirfd = file_open_private_dir(temp_dir);
if (scratch_dirfd < 0) {
int saved_errno = errno;
log_message(LOG_LEVEL_ERROR, "could not open --temp-dir scratch directory '%s': %s", temp_dir,
strerror(saved_errno));
close(dirfd);
free(leaf);
return false;
}
}
char* basis_leaf = NULL;
int basis_dirfd = file_open_secure_parent(basis_path, &basis_leaf, false);
bool linked = false;
if (basis_dirfd >= 0 && basis_leaf != NULL) {
int tmp_size = snprintf(NULL, 0, ".%s.tmp.%ld.%llu", leaf, (long)getpid(), ~0ULL);
char* tmp = NULL;
if (tmp_size >= 0)
tmp = malloc((size_t)tmp_size + 1);
if (!tmp) {
log_message(LOG_LEVEL_ERROR, "memory allocation failed while hard-linking basis file");
} else {
for (unsigned int i = 0; i < 100 && !linked; ++i) {
if (scratch_dirfd >= 0)
snprintf(tmp, (size_t)tmp_size + 1, ".%s.tmp.%ld.%llu", leaf, (long)getpid(),
next_temp_sequence());
else
snprintf(tmp, (size_t)tmp_size + 1, ".%s.tmp.%ld.%u", leaf, (long)getpid(), i);
if (linkat(basis_dirfd, basis_leaf, scratch_dirfd >= 0 ? scratch_dirfd : dirfd, tmp, 0) ==
0) {
linked = true;
break;
}
if (errno != EEXIST)
break; /* EXDEV / EPERM / ...: give up and fall back to a copy */
}
if (linked) {
int target_dirfd = scratch_dirfd >= 0 ? scratch_dirfd : dirfd;
if (use_fsync) {
int tfd = openat(target_dirfd, tmp, O_RDONLY | O_NOFOLLOW | O_CLOEXEC);
if (tfd < 0 || fsync(tfd) != 0) {
linked = false;
if (tfd >= 0)
close(tfd);
} else {
close(tfd);
}
}
if (linked && renameat(target_dirfd, tmp, dirfd, leaf) != 0)
linked = false;
if (!linked)
unlinkat(target_dirfd, tmp, 0);
}
free(tmp);
}
}
if (basis_dirfd >= 0)
close(basis_dirfd);
free(basis_leaf);
basis_leaf = NULL;
if (!linked) {
if (scratch_dirfd >= 0)
close(scratch_dirfd);
close(dirfd);
free(leaf);
/* The basis file could not be linked in (missing, cross-device, refused
by the filesystem). Write a byte-identical local copy instead. */
return file_to_disk_secure_with_fsync(path, data, data_size, false, false, metadata,
preserve_executability, use_fsync, temp_dir);
}
if (scratch_dirfd >= 0)
close(scratch_dirfd);
close(dirfd);
free(leaf);
return true;
}
bool file_write_to_disk(const char* path, const void* data, unsigned long long data_size, bool file_write_to_disk(const char* path, const void* data, unsigned long long data_size,
bool inplace, bool sparse) { bool inplace, bool sparse) {
if (!path || (!data && data_size != 0) || has_path_traversal(path)) if (!path || (!data && data_size != 0) || has_path_traversal(path))
+7
View File
@@ -63,5 +63,12 @@ bool file_to_disk_secure_no_replace(const char* path, const void* data,
unsigned long long data_size, bool sparse, unsigned long long data_size, bool sparse,
const FileMetadata* metadata, bool preserve_executability, const FileMetadata* metadata, bool preserve_executability,
const char* temp_dir); const char* temp_dir);
/* Atomic --link-dest install: replace `path` with a hard link to `basis_path`
(via a temp name + rename); fall back to a byte-identical local copy from
`data` when the link is impossible (EXDEV/EPERM/unsupported filesystem).
`metadata` is applied only on the copy fallback. */
bool file_to_disk_secure_link(const char* path, const char* basis_path, const void* data,
unsigned long long data_size, const FileMetadata* metadata,
bool preserve_executability, bool use_fsync, const char* temp_dir);
#endif #endif
+264 -19
View File
@@ -75,10 +75,17 @@ static FileSaveResult file_stage_delayed_update(const char* root_directory,
/* The staged location is brand new (stale leftovers from a prior crash were /* The staged location is brand new (stale leftovers from a prior crash were
wiped by prepare), so the plain atomic temp+rename engine installs the wiped by prepare), so the plain atomic temp+rename engine installs the
complete file there. --temp-dir scratch is deliberately not layered on complete file there. --temp-dir scratch is deliberately not layered on
top of the delay-updates staging tree. */ top of the delay-updates staging tree. A --link-dest basis file is hard
bool ok = linked into the staging tree (so publication's rename keeps the link). */
file_to_disk_secure_with_fsync(staged_path, file->data->data, file->data->size, false, sparse, bool ok;
metadata, preserve_executability, config->use_fsync, NULL); if (file->basis_link) {
ok = file_to_disk_secure_link(staged_path, file->basis_link, file->data->data, file->data->size,
metadata, preserve_executability, config->use_fsync, NULL);
} else {
ok = file_to_disk_secure_with_fsync(staged_path, file->data->data, file->data->size, false,
sparse, metadata, preserve_executability, config->use_fsync,
NULL);
}
if (!ok) { if (!ok) {
free(staged_path); free(staged_path);
return FILE_SAVE_ERROR; return FILE_SAVE_ERROR;
@@ -272,16 +279,27 @@ FileSaveResult file_save_to_disk_full(const char* root_directory, const File* fi
while (temp_len > 1 && confined_temp[temp_len - 1] == '/') while (temp_len > 1 && confined_temp[temp_len - 1] == '/')
confined_temp[--temp_len] = '\0'; confined_temp[--temp_len] = '\0';
} }
bool ok = /* A --link-dest basis hit installs an atomic hard link (with a byte-copy
config && config->ignore_existing fallback); --inplace and the update/no-replace write variants do not
? file_to_disk_secure_no_replace(disk_path, file->data->data, file->data->size, sparse, apply to a fresh hard link, whose inode attributes already match. The
metadata, preserve_executability, confined_temp) existing/ignore-existing/update/backup preamble above has already made the
: config && config->update policy decision. */
? file_to_disk_secure_update(disk_path, file->data->data, file->data->size, inplace, bool ok;
sparse, metadata, preserve_executability, confined_temp) if (config && file->basis_link) {
: file_to_disk_secure_with_fsync(disk_path, file->data->data, file->data->size, inplace, ok = file_to_disk_secure_link(disk_path, file->basis_link, file->data->data, file->data->size,
sparse, metadata, preserve_executability, metadata, preserve_executability, config->use_fsync,
config && config->use_fsync, confined_temp); confined_temp);
} else {
ok = config && config->ignore_existing
? file_to_disk_secure_no_replace(disk_path, file->data->data, file->data->size, sparse,
metadata, preserve_executability, confined_temp)
: config && config->update
? file_to_disk_secure_update(disk_path, file->data->data, file->data->size, inplace,
sparse, metadata, preserve_executability, confined_temp)
: file_to_disk_secure_with_fsync(disk_path, file->data->data, file->data->size,
inplace, sparse, metadata, preserve_executability,
config && config->use_fsync, confined_temp);
}
free(confined_temp); free(confined_temp);
confined_temp = NULL; confined_temp = NULL;
if (!ok) if (!ok)
@@ -505,6 +523,143 @@ static File* receive_delta_file(int fd, const Config* config, const char* check_
return NULL; return NULL;
} }
/* ---- Alternate basis directories (--compare-dest / --copy-dest / --link-dest) ----
* The receiver consults the ordered basis-dir list only when the destination
* entry is NOT already up to date. An "exact match" requires an equal size,
* an equal mtime (unless --size-only), and an equal content xxHash64, so a
* hard link / local copy is only ever made from byte-identical content. */
typedef struct BasisMatch {
bool hit;
BasisDestType type;
char* basis_path; /* owned absolute path of the matched basis file */
struct stat st; /* fstat() of the matched basis file */
Data* content; /* owned basis bytes (or empty Data), NULL when not loaded */
} BasisMatch;
static void basis_match_free(BasisMatch* match) {
if (!match)
return;
free(match->basis_path);
match->basis_path = NULL;
data_destroy(match->content);
match->content = NULL;
match->hit = false;
match->type = BASIS_DEST_NONE;
}
/* Open `path` (via the secure, root-confined primitives) and require it to be
a regular file of exactly `expected_size` bytes. Returns an open read-only
descriptor and its fstat on success. */
static bool basis_open_regular(const char* path, unsigned long long expected_size, int* out_fd,
struct stat* out_st) {
char* leaf = NULL;
int parent_fd = file_open_secure_parent(path, &leaf, false);
if (parent_fd < 0)
return false;
int fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW);
free(leaf);
close(parent_fd);
if (fd < 0)
return false;
struct stat st;
if (fstat(fd, &st) != 0 || !S_ISREG(st.st_mode) ||
(unsigned long long)st.st_size != expected_size) {
close(fd);
return false;
}
*out_fd = fd;
*out_st = st;
return true;
}
/* Read the whole remaining content of an open descriptor. A zero-length file
yields an empty Data (data pointer NULL). */
static Data* basis_read_content(int fd, unsigned long long size) {
if (size == 0)
return data_create_reserve(0);
if (size > MAX_RECEIVE_WHOLE_FILE_SIZE || size > SIZE_MAX)
return NULL;
void* buf = protocol_alloc((size_t)size);
if (!buf)
return NULL;
size_t got = 0;
while (got < (size_t)size) {
ssize_t n = read(fd, (char*)buf + got, (size_t)size - got);
if (n <= 0) {
free(buf);
return NULL;
}
got += (size_t)n;
}
return data_create(buf, (size_t)size);
}
/* --ignore-times forces every file to be updated, so no basis hit is ever
declared (matching rsync, where -I prevents link-dest from linking). */
static bool basis_quick_matches(const Config* config, const struct stat* st, time_t check_mtime,
long check_mtime_nsec) {
if (config->size_only)
return true;
long mtime_nsec = 0;
#ifdef __linux__
mtime_nsec = st->st_mtim.tv_nsec;
#endif
return metadata_mtime_matches(st->st_mtime, mtime_nsec, check_mtime, check_mtime_nsec,
config->modify_window);
}
/* Search the basis-dir list in command-line order and return the first exact
match. When load_content is true the matched bytes are kept in out->content
so the caller can materialize the file without re-reading it. */
static bool basis_match_find(const Config* config, const char* check_path,
unsigned long long check_size, time_t check_mtime,
long check_mtime_nsec, uint64_t check_checksum, bool load_content,
BasisMatch* out) {
memset(out, 0, sizeof(*out));
if (!config || !config_has_basis(config) || config->ignore_times)
return false;
for (int i = 0; i < config->basis_count; i++) {
const BasisDest* entry = &config->basis_dirs[i];
char* basis_dir = path_cat(config->receive_root_directory, entry->path);
if (!basis_dir)
continue;
char* candidate = path_cat(basis_dir, check_path);
free(basis_dir);
if (!candidate)
continue;
int fd;
struct stat st;
if (basis_open_regular(candidate, check_size, &fd, &st)) {
if (basis_quick_matches(config, &st, check_mtime, check_mtime_nsec)) {
Data* content = basis_read_content(fd, check_size);
if (content) {
uint64_t basis_hash = check_size == 0 ? delta_xxhash64("", 0)
: content->data ? delta_xxhash64(content->data, content->size)
: 0;
if (basis_hash == check_checksum) {
out->hit = true;
out->type = entry->type;
out->basis_path = candidate;
candidate = NULL; /* ownership transferred to out */
out->st = st;
out->content = load_content ? content : NULL;
if (!load_content)
data_destroy(content);
close(fd);
return true;
}
}
data_destroy(content);
}
close(fd);
}
free(candidate);
}
return false;
}
File* receive_incremental_check(int fd, const Config* config, bool* skipped) { File* receive_incremental_check(int fd, const Config* config, bool* skipped) {
if (!config || !skipped) { if (!config || !skipped) {
send_status(fd, STATUS_ERROR); send_status(fd, STATUS_ERROR);
@@ -531,7 +686,8 @@ File* receive_incremental_check(int fd, const Config* config, bool* skipped) {
send_status(fd, STATUS_ERROR); send_status(fd, STATUS_ERROR);
return NULL; return NULL;
} }
if (config->checksum && !receive_n_data(fd, &check_checksum, sizeof(check_checksum))) { if ((config->checksum || config_has_basis(config)) &&
!receive_n_data(fd, &check_checksum, sizeof(check_checksum))) {
free(check_path); free(check_path);
return NULL; return NULL;
} }
@@ -642,6 +798,76 @@ File* receive_incremental_check(int fd, const Config* config, bool* skipped) {
return NULL; return NULL;
} }
/* ---- Alternate basis directories ---- */
if (config_has_basis(config)) {
BasisMatch basis;
basis_match_find(config, check_path, check_size, (time_t)check_mtime, (long)check_mtime_nsec,
check_checksum, true, &basis);
if (basis.hit) {
if (basis.type == BASIS_DEST_COMPARE) {
/* compare-dest never copies: an exact match only suppresses the data
for a file the destination does not already hold (sparse backup).
When the destination holds a DIFFERENT version FastSync falls back to
a normal transfer rather than deleting the stale entry the way rsync
does (see RSYNC_COMPAT.md). */
basis_match_free(&basis);
if (!has_old_file) {
if (!send_status(fd, STATUS_OK)) {
close(old_fd);
free(full_path);
free(check_path);
return NULL;
}
free(old_data);
close(old_fd);
free(full_path);
free(check_path);
*skipped = true;
return NULL;
}
} else {
/* copy-dest / link-dest: materialize the unchanged file locally so the
sender can skip the data. The store engine re-applies the normal
existing/ignore-existing/update/backup/delay-updates policy. */
File* materialized = file_create(check_path);
if (materialized && basis.content) {
materialized->data = basis.content;
basis.content = NULL;
materialized->metadata = file_metadata_create(&basis.st);
materialized->skip = true; /* receiver must not ack this as a data file */
if (basis.type == BASIS_DEST_LINK) {
materialized->basis_link = basis.basis_path;
basis.basis_path = NULL;
}
if (!materialized->metadata) {
file_destroy(materialized);
materialized = NULL;
}
} else {
file_destroy(materialized);
materialized = NULL;
}
if (materialized) {
if (!send_status(fd, STATUS_OK)) {
file_destroy(materialized);
close(old_fd);
free(full_path);
free(check_path);
return NULL;
}
free(old_data);
close(old_fd);
free(full_path);
free(check_path);
*skipped = false;
return materialized;
}
/* Materialization setup failed: fall through to the normal transfer. */
}
}
basis_match_free(&basis);
}
if (try_delta && old_data != NULL) { if (try_delta && old_data != NULL) {
bool delta_failed = false; bool delta_failed = false;
File* delta_file = File* delta_file =
@@ -843,10 +1069,29 @@ int receive_manifest(int fd, const Config* config, int* next_status) {
/* With --delay-updates the staged (not yet published) files live directly /* With --delay-updates the staged (not yet published) files live directly
under the receive root in the staging directory; the delete walker must under the receive root in the staging directory; the delete walker must
not treat them as extras or it would remove every staged file before it not treat them as extras or it would remove every staged file before it
can be published. */ can be published. Alternate basis directories (--compare-dest /
const char* skip_staging = config->delay_updates ? DELAY_UPDATES_STAGING_DIR : NULL; --copy-dest / --link-dest) are also excluded: they are extra comparison
bool deletion_ok = delete_extras_limited(config->receive_root_directory, manifest, snapshots the user pointed at, not destination content, and deleting them
MAX_SERVER_DELETE_COUNT, skip_staging); would destroy the very files a --link-dest run just linked into place. */
int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count;
const char** skip_prefixes = NULL;
bool deletion_ok = false;
if (skip_count > 0) {
skip_prefixes = calloc((size_t)skip_count, sizeof(char*));
if (!skip_prefixes) {
array_list_delete(manifest);
send_status(fd, STATUS_ERROR);
return -1;
}
int idx = 0;
if (config->delay_updates)
skip_prefixes[idx++] = DELAY_UPDATES_STAGING_DIR;
for (int i = 0; i < config->basis_count; i++)
skip_prefixes[idx++] = config->basis_dirs[i].path;
}
deletion_ok = delete_extras_limited(config->receive_root_directory, manifest,
MAX_SERVER_DELETE_COUNT, skip_prefixes, skip_count);
free(skip_prefixes);
array_list_delete(manifest); array_list_delete(manifest);
if (!deletion_ok) if (!deletion_ok)
send_status(fd, STATUS_ERROR); send_status(fd, STATUS_ERROR);
+6
View File
@@ -29,6 +29,12 @@ typedef struct {
/* True when this entry is an explicit directory entry (--dirs mode): the /* True when this entry is an explicit directory entry (--dirs mode): the
* receiver creates the directory instead of writing a regular file. */ * receiver creates the directory instead of writing a regular file. */
bool is_dir; bool is_dir;
/* Receiver-only, --link-dest: when set, install the destination entry as a
* hard link to this absolute (root-confined) path instead of writing
* `data`. The matching code has already verified the link target's content
* equals the incoming file, and `data` is kept as the cross-filesystem
* fallback (a local copy) if the hard link cannot be created. */
char* basis_link;
} File; } File;
/* The path that should be sent on the wire and used for the receiver-side /* The path that should be sent on the wire and used for the receiver-side
+1 -1
View File
@@ -297,7 +297,7 @@ int write_thread(void* pipeline_context) {
/* Record the per-file outcome so a --remove-source-files sender learns /* Record the per-file outcome so a --remove-source-files sender learns
which sources were actually written versus skipped on the receiver. which sources were actually written versus skipped on the receiver.
Explicit directory entries have no source and are never acknowledged. */ Explicit directory entries have no source and are never acknowledged. */
if (context->config->remove_source_files && !file->is_dir && if (context->config->remove_source_files && !file->is_dir && !file->skip &&
!receiver_outcomes_append(&context->outcomes, (unsigned char)result)) { !receiver_outcomes_append(&context->outcomes, (unsigned char)result)) {
file_destroy(file); file_destroy(file);
pipeline_context_receiver_note_bytes_released(context, file_bytes); pipeline_context_receiver_note_bytes_released(context, file_bytes);