Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
116 changes: 88 additions & 28 deletions fileio.c
Original file line number Diff line number Diff line change
Expand Up @@ -75,11 +75,60 @@ int sparse_end(int f, OFF_T size, int updating_basis_or_equiv)
/* Note that the offset is just the caller letting us know where
* the current file position is in the file. The use_seek arg tells
* us that we should seek over matching data instead of writing it. */
/* Flush any deferred run of zero bytes as a hole, advancing the file
* position past it (both do_lseek() and do_punch_hole() move the offset). */
static int flush_sparse_hole(int f)
{
if (!sparse_seek)
return 0;
if (sparse_past_write >= preallocated_len) {
if (do_lseek(f, sparse_seek, SEEK_CUR) < 0) {
sparse_seek = 0;
return -1;
}
} else if (do_punch_hole(f, sparse_past_write, sparse_seek) < 0) {
sparse_seek = 0;
return -1;
}
sparse_seek = 0;
return 0;
}

static int full_sparse_write(int f, const char *buf, int len)
{
while (len > 0) {
int ret = write(f, buf, len);
if (ret <= 0) {
if (ret < 0 && errno == EINTR)
continue;
sparse_seek = 0;
return -1;
}
buf += ret;
len -= ret;
}
return 0;
}

/* Emit one span of data that is not being turned into a hole. For an in-place
* update (use_seek) the bytes on disk already match, so we only need to move
* past them; otherwise we write them out. Either way a deferred hole is
* flushed first so that the span lands at the right offset. */
static int emit_sparse_span(int f, int use_seek, const char *buf, int len)
{
if (flush_sparse_hole(f) < 0)
return -1;
if (use_seek)
return do_lseek(f, len, SEEK_CUR) < 0 ? -1 : 0;
return full_sparse_write(f, buf, len);
}

static int write_sparse(int f, int use_seek, OFF_T offset, const char *buf, int len)
{
int l1 = 0, l2 = 0;
int ret;
int l1, l2, i, start, end;

/* Always treat a leading and trailing run of zeros as a (deferred)
* hole, since they may merge with holes in the adjacent write calls. */
for (l1 = 0; l1 < len && buf[l1] == 0; l1++) {}
for (l2 = 0; l2 < len-l1 && buf[len-(l2+1)] == 0; l2++) {}

Expand All @@ -88,36 +137,45 @@ static int write_sparse(int f, int use_seek, OFF_T offset, const char *buf, int
if (l1 == len)
return len;

if (sparse_seek) {
if (sparse_past_write >= preallocated_len) {
if (do_lseek(f, sparse_seek, SEEK_CUR) < 0)
/* Scan the middle [l1, len-l2) for interior runs of zeros that are at
* least SPARSE_WRITE_SIZE long (the hole granularity rsync has always
* used) and defer those as holes. Everything in between -- which may
* include shorter zero runs not worth a hole -- is emitted in one go,
* rather than being chopped into SPARSE_WRITE_SIZE-byte pieces, which
* made copying a large non-sparse file cost ~one write() per KiB.
*
* The matched (use_seek) case runs through the same scan: its interior
* zero runs still have to be punched out, which is what --inplace
* --sparse relies on to keep a hole-y basis file sparse. */
start = l1;
end = len - l2;
for (i = l1; i < end; ) {
int z;
if (buf[i] != 0) {
i++;
continue;
}
for (z = 1; i + z < end && buf[i+z] == 0; z++) {}
if (z < SPARSE_WRITE_SIZE) {
i += z;
continue;
}
if (i > start) {
if (emit_sparse_span(f, use_seek, buf + start, i - start) < 0)
return -1;
} else if (do_punch_hole(f, sparse_past_write, sparse_seek) < 0) {
sparse_seek = 0;
return -1;
sparse_past_write = offset + i;
}
sparse_seek += z;
i += z;
start = i;
}
sparse_seek = l2;
sparse_past_write = offset + len - l2;

if (use_seek) {
/* The in-place data already matches. */
if (do_lseek(f, len - (l1+l2), SEEK_CUR) < 0)
if (end > start) {
if (emit_sparse_span(f, use_seek, buf + start, end - start) < 0)
return -1;
return len;
}

while ((ret = write(f, buf + l1, len - (l1+l2))) <= 0) {
if (ret < 0 && errno == EINTR)
continue;
sparse_seek = 0;
return ret;
}

if (ret != (int)(len - (l1+l2))) {
sparse_seek = 0;
return l1+ret;
}
sparse_seek = l2;
sparse_past_write = offset + len - l2;

return len;
}
Expand Down Expand Up @@ -153,8 +211,10 @@ int write_file(int f, int use_seek, OFF_T offset, const char *buf, int len)
while (len > 0) {
int r1;
if (sparse_files > 0) {
int len1 = MIN(len, SPARSE_WRITE_SIZE);
r1 = write_sparse(f, use_seek, offset, buf, len1);
/* write_sparse() handles the whole span itself, scanning
* for holes and coalescing the non-zero data into large
* write()s instead of SPARSE_WRITE_SIZE-byte dribbles. */
r1 = write_sparse(f, use_seek, offset, buf, len);
offset += r1;
} else {
if (!wf_writeBuf) {
Expand Down
28 changes: 28 additions & 0 deletions testsuite/preallocate_test.py
Original file line number Diff line number Diff line change
Expand Up @@ -124,5 +124,33 @@ def seed_holey(head=4096, hole=2 * 1024 * 1024, tail=4096):
f"{allocated(TODIR / deep)} for a {os.path.getsize(TODIR / deep)}"
"-byte file")

# --- --inplace --sparse must punch interior holes in matched blocks ----------
# Make source and destination byte-identical and densely allocated. With
# --block-size matching the pattern size, every block match is at the same
# offset, so skip_matched() sends it through write_sparse(use_seek=1). The
# interior zero run must still be scanned and punched rather than merely
# seeked over with the rest of the matching block.
rmtree(src)
rmtree(TODIR)
makepath(src / 'd1' / 'd2' / 'd3', TODIR / 'd1' / 'd2' / 'd3')
with open(src / deep, 'wb') as source, open(TODIR / deep, 'wb') as dest:
for _ in range(256):
block = os.urandom(4096) + b'\0' * 24576 + os.urandom(4096)
source.write(block)
dest.write(block)

matched_size = os.path.getsize(TODIR / deep)
matched_before = allocated(TODIR / deep)
run_rsync('-a', '--ignore-times', '--inplace', '--sparse', '--no-whole-file',
'--block-size=32768', f'{src}/', f'{TODIR}/')
assert_same(TODIR / deep, src / deep,
label='--inplace --sparse matched-block content')
matched_after = allocated(TODIR / deep)
if (can_punch and matched_before >= matched_size
and matched_after * 2 >= matched_before):
test_fail(f"--inplace --sparse left matching interior zero runs allocated: "
f"{matched_after} of {matched_before} bytes remain allocated "
f"after a {matched_size}-byte matched-block update")

print("preallocate: --preallocate (do_fallocate) + sparse hole-punching "
"(do_punch_hole) verified at depth")
Loading