hard-links: -H/--hard-links preserves inode relationships
CI / lint (pull_request) Successful in 50s
CI / sanitizers (address) (pull_request) Skipped
CI / sanitizers (undefined) (pull_request) Skipped
CI / fuzz-build (pull_request) Skipped
CI / coverage (pull_request) Skipped
CI / valgrind (pull_request) Skipped
CI / build-and-test (pull_request) Successful in 1m19s

Source files sharing (st_dev,st_ino) are recreated as hard links on the
destination; only the first member's data crosses the wire (siblings ride a
payload-less STATUS_HARDLINK frame). Ordering requires the single-FIFO-writer
receiver + forced sequential scan (documented). link()-failure falls back to a
byte-identical local copy. Rejects -s/--append. PROTOCOL_VERSION 2.11.0->2.12.0.
Review fixes: delete the dead HardLinkRegistry (ordering holds by FIFO writer),
and --existing no longer aborts when the first member is absent but the sibling
exists (leaves the sibling in place).
This commit is contained in:
2026-09-08 20:58:56 +02:00
parent cbb09e41ab
commit f891cd0a6a
21 changed files with 792 additions and 13 deletions
+2
View File
@@ -195,6 +195,8 @@ static bool validate_received_config(const Config* config) {
which chunk serialization -s disables: reject on the receiver too
so a -s sender cannot negotiate an inert append mode. */
!((config->append || config->append_verify) && config->use_chunk_serialization) &&
!(config->preserve_hard_links && config->use_chunk_serialization) &&
!(config->preserve_hard_links && (config->append || config->append_verify)) &&
(!config->use_compression ||
(config->compression_level >= 1 && config->compression_level <= 22)) &&
config->chunk_size > 0 && config->chunk_size <= MAX_CHUNK_SIZE &&
+1 -1
View File
@@ -293,7 +293,7 @@ typedef struct Config {
DelayUpdatesContext* delay_context;
} Config;
#define PROTOCOL_VERSION "2.11.0"
#define PROTOCOL_VERSION "2.12.0"
#define DEFAULT_CHUNK_SIZE (10 * 1024 * 1024)
/* Upper bound on total basis-dir entries (rsync caps --link-dest at 20). */
#define MAX_BASIS_DIRS 64
+5
View File
@@ -108,6 +108,9 @@ File* file_create(const char* path) {
file->skip = false;
file->is_dir = false;
file->basis_link = NULL;
file->link_group = 0;
file->link_first = false;
file->hardlink_target = NULL;
return file;
}
@@ -125,6 +128,8 @@ void file_destroy(void* item) {
file->send_path = NULL;
free(file->basis_link);
file->basis_link = NULL;
free(file->hardlink_target);
file->hardlink_target = NULL;
free(file);
}
+246
View File
@@ -102,6 +102,192 @@ static FileSaveResult file_stage_delayed_update(const char* root_directory,
return FILE_SAVE_WRITTEN;
}
/* Read the whole content of a confined regular file (used to fall back to a
byte-identical copy when a hard-link sibling's link() fails). Symlink-safe
(parent resolved via file_open_secure_parent + O_NOFOLLOW). A zero-length
file yields *out_size 0 and *out_buf NULL as a SUCCESS. Returns false only
on a real error/read failure, setting *source_absent to true when the reason
was that the path does not exist (ENOENT/ENOTDIR), so the caller can decide
between an abort and a graceful skip. */
static bool hardlink_read_source(const char* path, void** out_buf, unsigned long long* out_size,
bool* source_absent) {
*out_buf = NULL;
*out_size = 0;
*source_absent = false;
if (!path)
return false;
char* leaf = NULL;
int parent_fd = file_open_secure_parent(path, &leaf, false);
if (parent_fd < 0) {
*source_absent = errno == ENOENT || errno == ENOTDIR;
return false;
}
int fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW);
int saved_errno = errno;
free(leaf);
close(parent_fd);
if (fd < 0) {
*source_absent = saved_errno == ENOENT || saved_errno == ENOTDIR;
return false;
}
struct stat st;
if (fstat(fd, &st) != 0 || !S_ISREG(st.st_mode)) {
close(fd);
return false;
}
unsigned long long size = (unsigned long long)st.st_size;
if (size > MAX_RECEIVE_WHOLE_FILE_SIZE || size > SIZE_MAX) {
close(fd);
return false;
}
if (size == 0) {
close(fd);
return true;
}
void* buf = protocol_alloc((size_t)size);
if (!buf) {
close(fd);
return false;
}
size_t got = 0;
while (got < (size_t)size) {
ssize_t n = read(fd, (char*)buf + got, (size_t)size - got);
if (n <= 0) {
free(buf);
close(fd);
return false;
}
got += (size_t)n;
}
close(fd);
*out_buf = buf;
*out_size = size;
return true;
}
/* The group's first member's installed file is absent, but its destination
path was validated (a sibling is only ever processed after its group's first
member). When the sibling's OWN destination already exists it should be
left alone -- a clean skip -- rather than aborting the whole transfer (the
asymmetric --existing case: the first member was skipped because its
destination was missing, while the sibling already has one). Only when the
sibling's destination is missing too is this a genuine failure to
link/copy, which aborts. */
static FileSaveResult hardlink_sibling_absent_first(const char* destination_path) {
if (destination_path && file_path_exists_secure(destination_path))
return FILE_SAVE_SKIPPED;
return FILE_SAVE_ERROR;
}
/* Install a --hard-links/-H sibling: the destination entry is atomically
replaced (temp + rename) with a hard link to the group's first member. The
first member is guaranteed already installed at `hardlink_target` under the
root because -H relies on the receiver's single-FIFO-writer pipeline (one
receive thread, one write thread, FIFO queue => wire order == write order)
plus the sender's forced sequential scan, so a sibling is always processed
after its group's first member. When link() fails (different filesystem,
filesystem refuses links) a byte-identical copy of the first member is
written instead, so the result is never partial or corrupt. With
--delay-updates the sibling is staged as a hard link to the first member's
STAGED file (publication's renames preserve the shared inode). The final
--existing/--ignore-existing/--update policies are decided against the final
destination like every normal write. */
static FileSaveResult file_save_hardlink_sibling(const char* root_directory, const File* file,
const Config* config) {
Config* cfg = (Config*)config;
if (!root_directory || !file || !file->path || !file->hardlink_target)
return FILE_SAVE_ERROR;
char* destination_path = path_cat(root_directory, file->path);
if (!destination_path)
return FILE_SAVE_ERROR;
if (cfg->existing && !file_path_exists_secure(destination_path)) {
free(destination_path);
return FILE_SAVE_SKIPPED;
}
if (cfg->ignore_existing && file_path_exists_secure(destination_path)) {
free(destination_path);
return FILE_SAVE_SKIPPED;
}
if (cfg->update && file_destination_is_newer_secure(destination_path, file->metadata)) {
free(destination_path);
return FILE_SAVE_SKIPPED;
}
bool preallocate = cfg && cfg->preallocate;
bool preserve_executability = cfg && cfg->use_executability;
bool use_fsync = cfg && cfg->use_fsync;
if (cfg->delay_updates) {
if (!cfg->delay_context) {
cfg->delay_context = delay_updates_context_create(root_directory);
if (!cfg->delay_context) {
free(destination_path);
return FILE_SAVE_ERROR;
}
}
if (!delay_updates_prepare(cfg->delay_context)) {
free(destination_path);
return FILE_SAVE_ERROR;
}
char* staged_first = path_cat(cfg->delay_context->staging_root, file->hardlink_target);
char* staged_sibling = path_cat(cfg->delay_context->staging_root, file->path);
if (!staged_first || !staged_sibling) {
free(staged_first);
free(staged_sibling);
free(destination_path);
return FILE_SAVE_ERROR;
}
void* content = NULL;
unsigned long long content_size = 0;
bool source_absent = false;
if (!hardlink_read_source(staged_first, &content, &content_size, &source_absent)) {
FileSaveResult absent_result =
source_absent ? hardlink_sibling_absent_first(destination_path) : FILE_SAVE_ERROR;
free(staged_first);
free(staged_sibling);
free(destination_path);
return absent_result;
}
bool ok =
file_to_disk_secure_link(staged_sibling, staged_first, content, content_size, preallocate,
file->metadata, preserve_executability, use_fsync, NULL);
free(content);
if (ok)
ok = delay_updates_record(cfg->delay_context, staged_sibling, destination_path, file->path);
if (!ok)
unlink(staged_sibling);
free(staged_first);
free(staged_sibling);
free(destination_path);
return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR;
}
char* first_disk = path_cat(root_directory, file->hardlink_target);
if (!first_disk) {
free(destination_path);
return FILE_SAVE_ERROR;
}
void* content = NULL;
unsigned long long content_size = 0;
bool source_absent = false;
if (!hardlink_read_source(first_disk, &content, &content_size, &source_absent)) {
FileSaveResult absent_result =
source_absent ? hardlink_sibling_absent_first(destination_path) : FILE_SAVE_ERROR;
free(first_disk);
free(destination_path);
return absent_result;
}
const char* temp_dir = (cfg && cfg->temp_dir) ? cfg->temp_dir : NULL;
bool ok =
file_to_disk_secure_link(destination_path, first_disk, content, content_size, preallocate,
file->metadata, preserve_executability, use_fsync, temp_dir);
free(content);
free(first_disk);
free(destination_path);
return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR;
}
FileSaveResult file_save_to_disk_full(const char* root_directory, const File* file,
const Config* config) {
/* Backups are incompatible with ignore-existing: moving the entry first
@@ -146,6 +332,14 @@ FileSaveResult file_save_to_disk_full(const char* root_directory, const File* fi
return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR;
}
/* --hard-links/-H sibling: a later member of a link group arrives with no
payload and is installed as a hard link to (or, on link() failure, a
byte-identical copy of) the group's first member. Handled entirely here,
before the normal data-write paths (which would create an empty file). */
if (file->link_group != 0 && !file->link_first && file->hardlink_target != NULL) {
return file_save_hardlink_sibling(root_directory, file, config);
}
/* These options arrive from the client. They are names below the server
root, never independent filesystem roots. --temp-dir is confined exactly
like --backup-dir/--partial-dir: an absolute or `..`-escaping scratch
@@ -1622,6 +1816,58 @@ File* file_receive_directory(int file_descriptor) {
return file;
}
/* Receive a --hard-links/-H sibling frame (the leading STATUS_HARDLINK code has
already been consumed): the destination path, the run-local link-group id,
and the first (data-carrying) member's destination-relative wire path. The
created File carries no payload; it is installed beneath the receive root as
a hard link to (or, on link failure, a byte-identical copy of) the first
member. All paths are validated like every other received path (non-empty,
relative, no traversal). */
File* file_receive_hardlink(int file_descriptor) {
char* path = receive_str(file_descriptor);
if (path == NULL)
return NULL;
if (path[0] == '\0' || has_path_traversal(path)) {
char* escaped_path = output_escape(path, log_get_8_bit_output());
log_message(LOG_LEVEL_ERROR, "Invalid received hard-link path: %s",
escaped_path ? escaped_path : "<allocation failed>");
free(escaped_path);
free(path);
send_status(file_descriptor, STATUS_ERROR);
return NULL;
}
int gid;
if (!receive_int(file_descriptor, &gid) || gid <= 0) {
free(path);
return NULL;
}
char* target = receive_str(file_descriptor);
if (!target) {
free(path);
return NULL;
}
if (target[0] == '\0' || has_path_traversal(target)) {
char* escaped = output_escape(target, log_get_8_bit_output());
log_message(LOG_LEVEL_ERROR, "Invalid hard-link target path: %s",
escaped ? escaped : "<allocation failed>");
free(escaped);
free(target);
free(path);
send_status(file_descriptor, STATUS_ERROR);
return NULL;
}
File* file = file_create(path);
free(path);
if (file == NULL) {
free(target);
return NULL;
}
file->link_group = gid;
file->link_first = false;
file->hardlink_target = target;
return file;
}
/* Read a delete-manifest frame (the STATUS_MANIFEST leading code has already
been consumed): a keep-set entry count followed by that many
destination-relative paths, then a protected-prefix count followed by that
+1
View File
@@ -9,6 +9,7 @@
File* file_receive(const Config* config, int file_descriptor);
File* file_receive_directory(int file_descriptor);
File* file_receive_hardlink(int file_descriptor);
File* receive_incremental_check(int fd, const Config* config, bool* skipped);
/* A received delete-manifest frame: the keep-set (`keeps`, destination-relative
+9
View File
@@ -35,6 +35,15 @@ typedef struct {
* equals the incoming file, and `data` is kept as the cross-filesystem
* fallback (a local copy) if the hard link cannot be created. */
char* basis_link;
/* --hard-links (-H), sender + receiver wire state. link_group is a run-local
* id shared by every member of one source inode (0 = not part of a group).
* The FIRST member (link_first == true) carries its data on the wire and is
* written normally; every sibling (link_first == false) carries NO data and
* hardlink_target holds the first member's wire path so the receiver can link
* to (or copy from) the already-installed first member. */
int link_group;
bool link_first;
char* hardlink_target;
} File;
/* The path that should be sent on the wire and used for the receiver-side
+127
View File
@@ -0,0 +1,127 @@
#include "hardlink.h"
#include <stdint.h>
#include <stdlib.h>
#include <string.h>
#include "log.h"
#include "utils.h"
/* ---- Sender-side detection table ---- */
HardLinkTable* hardlink_table_create(void) {
HardLinkTable* table = calloc(1, sizeof(HardLinkTable));
if (!table)
return NULL;
if (mtx_init(&table->mutex, mtx_plain) != thrd_success) {
free(table);
return NULL;
}
table->next_gid = 1;
return table;
}
static void hardlink_item_destroy(HardLinkItem* item) {
if (!item)
return;
free(item->first_path);
item->first_path = NULL;
}
void hardlink_table_destroy(HardLinkTable* table) {
if (!table)
return;
for (size_t i = 0; i < table->count; i++)
hardlink_item_destroy(&table->items[i]);
free(table->items);
table->items = NULL;
table->count = 0;
table->capacity = 0;
mtx_destroy(&table->mutex);
free(table);
}
static HardLinkItem* hardlink_table_find_locked(HardLinkTable* table, dev_t dev, ino_t ino) {
for (size_t i = 0; i < table->count; i++) {
if (table->items[i].dev == dev && table->items[i].ino == ino)
return &table->items[i];
}
return NULL;
}
static bool hardlink_table_add_locked(HardLinkTable* table, dev_t dev, ino_t ino, const char* path,
int gid, HardLinkItem** out) {
if (table->count == table->capacity) {
size_t new_capacity = table->capacity == 0 ? 8 : table->capacity * 2;
if (new_capacity < table->capacity)
return false;
HardLinkItem* grown = realloc(table->items, new_capacity * sizeof(HardLinkItem));
if (!grown)
return false;
table->items = grown;
table->capacity = new_capacity;
}
HardLinkItem* item = &table->items[table->count];
char* dup = str_dup(path);
if (!dup)
return false;
memset(item, 0, sizeof(*item));
item->dev = dev;
item->ino = ino;
item->gid = gid;
item->first_path = dup;
table->count++;
*out = item;
return true;
}
bool hardlink_table_assign(HardLinkTable* table, const char* wire_path, dev_t dev, ino_t ino,
int* gid, bool* is_first, char** first_path_out) {
if (!table || !wire_path || !gid || !is_first || !first_path_out)
return false;
if (mtx_lock(&table->mutex) != thrd_success)
return false;
bool ok = true;
const HardLinkItem* item = hardlink_table_find_locked(table, dev, ino);
int next_gid;
if (item) {
*is_first = false;
char* dup = str_dup(item->first_path);
if (!dup) {
ok = false;
} else {
*gid = item->gid;
*first_path_out = dup;
}
next_gid = -1;
} else {
if (table->next_gid <= 0) {
ok = false;
next_gid = -1;
} else {
next_gid = table->next_gid;
HardLinkItem* created = NULL;
if (!hardlink_table_add_locked(table, dev, ino, wire_path, next_gid, &created)) {
ok = false;
} else {
char* dup = str_dup(wire_path);
if (!dup) {
hardlink_item_destroy(created);
table->count--;
ok = false;
} else {
*is_first = true;
*gid = next_gid;
*first_path_out = dup;
}
}
}
}
if (ok && next_gid > 0)
table->next_gid++;
mtx_unlock(&table->mutex);
if (!ok) {
log_message(LOG_LEVEL_ERROR, "memory allocation failed while detecting hard links");
}
return ok;
}
+66
View File
@@ -0,0 +1,66 @@
#ifndef HARDLINK_H
#define HARDLINK_H
#include <stdbool.h>
#include <stddef.h>
#include <sys/types.h>
#include <threads.h>
/*
* --hard-links / -H support.
*
* Sender side: a HardLinkTable detects regular files on the source that share
* an (st_dev, st_ino) identity (a `cp -al`-style hard-linked tree) and assigns
* each distinct inode a stable, run-local link-group id. The first member
* encountered carries the file data; every later member is marked as a sibling
* (no data payload) that the receiver creates as a hard link to the first
* member's destination file. Grouping is scoped by st_dev so inode reuse
* across different filesystems is never conflated. The table is mutex-guarded
* so the parallel (multi-threaded) scanner COULD share one instance across its
* worker threads; the first-thread-to-call designates the data-carrying member,
* which is safe because a hard-link group's members are byte-identical. (In
* practice the sender forces the sequential scanner whenever -H is on; the
* mutex guards the shared table for any path that supplies one.)
*
* ORDERING (why there is no receiver-side handshake): the receiver stores every
* file - including a hard-link group's first member - through a SINGLE writer
* thread draining a single FIFO queue driven by a single receive thread, so
* wire order == write order and every sibling is processed AFTER its group's
* first member. The sender additionally forces the sequential scanner with -H
* so the first-member frame always precedes its siblings on the wire. Sibling
* install therefore needs no present/wait registry: it hard-links to the first
* member (or copies it) knowing that path is already installed - or that, if
* the first member was skipped (already up to date), its destination still
* exists. This guarantee is REQUIRED; do not introduce a concurrent
* multi-writer receiver for -H without re-adding an ordering mechanism.
*/
typedef struct HardLinkItem {
dev_t dev;
ino_t ino;
int gid;
char* first_path; /* wire path of the group's data-carrying first member */
} HardLinkItem;
typedef struct HardLinkTable {
mtx_t mutex;
HardLinkItem* items;
size_t count;
size_t capacity;
int next_gid;
} HardLinkTable;
HardLinkTable* hardlink_table_create(void);
void hardlink_table_destroy(HardLinkTable* table);
/* Assign a link-group id to the regular file at `wire_path` with (dev, ino).
* On the first encounter the file becomes the group's first (data-carrying)
* member (*is_first = true) and a fresh gid is allocated. On a later member
* *is_first = false and *first_path_out is set to a malloc'd copy of the first
* member's wire path (the caller stores it and owns it; on the first member
* path the returned *first_path_out is a malloc'd copy of its own wire path).
* Returns false on allocation failure (transfer should abort). */
bool hardlink_table_assign(HardLinkTable* table, const char* wire_path, dev_t dev, ino_t ino,
int* gid, bool* is_first, char** first_path_out);
#endif
+2
View File
@@ -405,6 +405,8 @@ static const char* status_to_string(Status status) {
return "APPEND_OK";
case STATUS_APPEND_DATA:
return "APPEND_DATA";
case STATUS_HARDLINK:
return "HARDLINK";
default:
return "UNKNOWN";
}
+7 -1
View File
@@ -84,7 +84,13 @@ enum NET_STATUS {
STATUS_APPEND,
STATUS_APPEND_SIG,
STATUS_APPEND_OK,
STATUS_APPEND_DATA
STATUS_APPEND_DATA,
/* --hard-links/-H: a sibling (later member) of a source hard-link group.
* The sender transmits only the path, the run-local link-group id, and the
* first (data-carrying) member's destination-relative wire path; the receiver
* creates this entry as a hard link to the first member's installed file
* (falling back to a byte-identical copy if link() fails). Protocol 2.12.0. */
STATUS_HARDLINK
};
void io_set_fds(int read_fd, int write_fd);