From 70238bf6e6c80ea99a32eb6407bf622968922604 Mon Sep 17 00:00:00 2001 From: folbrich Date: Sat, 26 Sep 2026 18:29:32 +0200 Subject: [PATCH] Leave zeros out of blank targets when reflinks are available The null seed skipped zero ranges on a blank target, a new or truncated file, only when reflinks weren't available. With them, it cloned the zero block file into the target one filesystem block at a time, filling in holes that read as zeros already. On Btrfs, extracting a Debian 12 cloud image, 3.2 GB of which 2.2 GB are zeros, into a new file with the previous build as seed took 77s and left the file in 494k extents. Skipping the zeros, it takes 8.4s and leaves 6.7k extents, with the same result. With that, extracting into a new file with the old image as seed is the quickest way to update an image kept on a filesystem with reflinks, and the cookbook says so. --- docs/cookbook.md | 2 ++ nullseed.go | 28 ++++++++++------------------ nullseed_test.go | 14 ++++++++++---- 3 files changed, 22 insertions(+), 22 deletions(-) diff --git a/docs/cookbook.md b/docs/cookbook.md index a0f1f3e..b333395 100644 --- a/docs/cookbook.md +++ b/docs/cookbook.md @@ -23,6 +23,8 @@ desync extract -s /local/store \ image-v3.qcow2.caibx image-v3.qcow2 ``` +On filesystems with reflinks, like Btrfs and XFS, what the new image has in common with a seed is cloned rather than copied. The new file shares those blocks with the seed and only takes up space for what changed, and zero-filled ranges stay holes. This is usually the fastest way to update an image kept as a file: extract the new version next to the old one, then replace or remove the old one. + Extract an image using several seeds present in a directory. Each of the `.caibx` files in the directory needs to have a matching blob of the same name. It is possible for the source index file to be in the same directory also (it'll be skipped automatically). ```text diff --git a/nullseed.go b/nullseed.go index 974aa68..d1ceecb 100644 --- a/nullseed.go +++ b/nullseed.go @@ -108,16 +108,16 @@ func (s *nullChunkSection) WriteInto(dst *os.File, offset, length, blocksize uin return 0, 0, fmt.Errorf("unable to copy %d bytes to %s : wrong size", length, dst.Name()) } - // When cloning isn't available we'd normally have to copy the 0 bytes into - // the target range. But if that's already blank (because it's a new/truncated - // file) there's no need to copy 0 bytes. + // A blank target, a new or truncated file, reads as zeros already. Cloning + // or copying zeros into it would only fill in what's a hole now, block by + // block when cloning. + if isBlank { + return 0, 0, nil + } if !s.canReflink { - if isBlank { - return 0, 0, nil - } return s.copy(dst, offset, s.Size()) } - return s.clone(dst, offset, length, blocksize, isBlank) + return s.clone(dst, offset, length, blocksize) } func (s *nullChunkSection) copy(dst *os.File, offset, length uint64) (uint64, uint64, error) { @@ -130,18 +130,14 @@ func (s *nullChunkSection) copy(dst *os.File, offset, length uint64) (uint64, ui return uint64(copied), 0, err } -func (s *nullChunkSection) clone(dst *os.File, offset, length, blocksize uint64, isBlank bool) (uint64, uint64, error) { +func (s *nullChunkSection) clone(dst *os.File, offset, length, blocksize uint64) (uint64, uint64, error) { dstAlignStart := (offset/blocksize + 1) * blocksize dstAlignEnd := (offset + length) / blocksize * blocksize // If the range is too small to contain a full aligned block, there is // nothing that can be cloned, and the copies below would write outside - // the range. Write zeros over the whole range instead, or nothing if - // it's still blank. + // the range. Write zeros over the whole range instead. if dstAlignEnd <= dstAlignStart { - if isBlank { - return 0, 0, nil - } return s.copy(dst, offset, length) } @@ -163,11 +159,7 @@ func (s *nullChunkSection) clone(dst *os.File, offset, length, blocksize uint64, if err := cloneRange(dst, s.blockfile, 0, blocksize, blkOffset); err != nil { // Not every filesystem that passes the CanClone probe can clone // every range. ZFS for example refuses to clone from the blockfile - // before it has been committed to disk. Fall back to writing zeros, - // or to doing nothing if the target range is still blank. - if isBlank { - return copied, cloned, nil - } + // before it has been committed to disk. Fall back to writing zeros. c3, _, err := s.copy(dst, blkOffset, dstAlignEnd-blkOffset) return copied + c3, cloned, err } diff --git a/nullseed_test.go b/nullseed_test.go index 231f03c..eac8692 100644 --- a/nullseed_test.go +++ b/nullseed_test.go @@ -14,7 +14,7 @@ import ( // Simulates filesystems like ZFS where CanClone succeeds but the actual // cloning of blocks fails, e.g. because the zero blockfile hasn't been // committed to disk yet. The null seed is expected to fall back to writing -// zeros, or to leave a blank target untouched. +// zeros. func TestNullChunkSectionCloneFallback(t *testing.T) { defer func() { cloneRange = CloneRange }() @@ -52,9 +52,13 @@ func TestNullChunkSectionCloneFallback(t *testing.T) { assert.Equal(t, make([]byte, length), got) }) - t.Run("leaves blank target untouched when cloning fails", func(t *testing.T) { + t.Run("leaves a blank target alone", func(t *testing.T) { + // A new or truncated file reads as zeros already, nothing is cloned + // or copied into it even though cloning works + var cloneCalls int cloneRange = func(dst, src *os.File, srcOffset, srcLength, dstOffset uint64) error { - return errors.New("simulated clone failure") + cloneCalls++ + return nil } dstName := filepath.Join(dir, "out2") dst, err := os.Create(dstName) @@ -62,9 +66,11 @@ func TestNullChunkSectionCloneFallback(t *testing.T) { defer dst.Close() require.NoError(t, dst.Truncate(int64(length))) - _, cloned, err := newSection().WriteInto(dst, 0, length, blocksize, true) + copied, cloned, err := newSection().WriteInto(dst, 0, length, blocksize, true) require.NoError(t, err) + assert.Equal(t, uint64(0), copied) assert.Equal(t, uint64(0), cloned) + assert.Equal(t, 0, cloneCalls) got, err := os.ReadFile(dstName) require.NoError(t, err)