From da0c4173124db568c33c7a5d40bb77d7c1ad7cb0 Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Fri, 11 Sep 2026 20:46:19 +0200 Subject: [PATCH] diff: align the chunk lists to count added/removed bytes The byte counts of a "modified" file were derived from the difference of the two chunk id SETS, which loses everything about order and multiplicity: a file whose chunks were only reordered was reported as modified with 0 B added and 0 B removed, and duplicating a chunk added no bytes at all. Align the two chunk lists as sequences instead (difflib.SequenceMatcher, the way a text diff aligns lines) and count the chunks that are not part of the alignment: the ones of archive1 as removed, the ones of archive2 as added. Insertions, removals, moves and duplicated chunks are now all reflected by the byte counts. The common prefix and suffix of the two lists are stripped first. That is what makes the usual cases cheap (e.g. a file that was appended to) and it also keeps the matcher away from the long runs of identical chunks it is slow on. SequenceMatcher still degrades to quadratic runtime on chunk lists that repeat the same chunk id very often (sparse files, VM images with big all-zero ranges), so the alignment is skipped above MAX_ALIGN_CHUNKS / MAX_ALIGN_WORK and the chunk ids are only counted then, which is what borg did before. Co-Authored-By: Claude Opus 5 --- docs/internals/frontends.rst | 13 ++-- src/borg/archiver/diff_cmd.py | 12 ++-- src/borg/item.pyi | 5 ++ src/borg/item.pyx | 63 +++++++++++++++++--- src/borg/testsuite/archiver/diff_cmd_test.py | 42 +++++++++++-- src/borg/testsuite/item_test.py | 48 ++++++++++++++- 6 files changed, 158 insertions(+), 25 deletions(-) diff --git a/docs/internals/frontends.rst b/docs/internals/frontends.rst index f95dba8a26..fc46dea525 100644 --- a/docs/internals/frontends.rst +++ b/docs/internals/frontends.rst @@ -642,11 +642,14 @@ type: added: If **type** is '*modified*', '*added*' or '*removed*', **added** and **removed** give the amount of data (in bytes) added and removed. For '*added*', **removed** is 0; for '*removed*', - **added** is 0. For '*modified*', **added** / **removed** is the total size of the chunks only - present in the ARCHIVE2 / ARCHIVE1 version of the file, so both are 0 for a file whose chunks - were merely reordered or duplicated. If the chunk ids can not be compared (the archives were - created with different ``--chunker-params``), a '*modified*' change has neither property and - the only information available is that the file contents were modified. + **added** is 0. For '*modified*', the two chunk lists of the file are aligned as sequences and + **added** / **removed** is the total size of the chunks of the ARCHIVE2 / ARCHIVE1 version that + are not part of that alignment, so inserted, removed, moved and duplicated content is accounted + for. For files with very long or very repetitive chunk lists, only the number of occurrences of + each chunk id is compared (aligning them would be too slow), so moved content does not show up + in their byte counts. If the chunk ids can not be compared (the archives were created with + different ``--chunker-params``), a '*modified*' change has neither property and the only + information available is that the file contents were modified. removed: See **added** property. diff --git a/src/borg/archiver/diff_cmd.py b/src/borg/archiver/diff_cmd.py index 8e26a1faa4..a0b83d8bba 100644 --- a/src/borg/archiver/diff_cmd.py +++ b/src/borg/archiver/diff_cmd.py @@ -275,10 +275,14 @@ def build_parser_diff(self, subparsers, common_parser, mid_common_parser): For each matching item in both archives, Borg reports: - Content changes: total added/removed bytes within files. If chunker parameters are comparable, - Borg compares chunk IDs quickly: the byte counts are the total sizes of the chunks only present - in one of the two versions of a file, so a file whose chunks were merely reordered or duplicated - is reported as modified with 0 B added and 0 B removed. Otherwise, Borg compares the content. In - the latter case, borg can only tell that a file was modified, not by how much: no byte counts + Borg compares chunk IDs quickly: it aligns the two chunk lists of a file like a text diff + aligns lines, and the byte counts are the total sizes of the chunks that are not part of that + alignment. Inserted, removed, moved and duplicated content is therefore accounted for - a + chunk that only moved within the file shows up as removed and added again. Files with very + long or very repetitive chunk lists are not aligned (this would be too slow); for these, only + the number of occurrences of each chunk ID is compared, so moved content does not show up in + their byte counts. If chunker parameters are not comparable, Borg compares the content. In + that case, borg can only tell that a file was modified, not by how much: no byte counts are given for it, the text output shows "modified: (can't get size)" instead. - Metadata changes: user, group, mode, and other metadata shown inline as "[old -> new]", like "[-rw-r--r-- -> -rwxr-xr-x]" for a mode change. Use ``--content-only`` to suppress metadata changes. diff --git a/src/borg/item.pyi b/src/borg/item.pyi index f3c917733a..f4f5ee3a06 100644 --- a/src/borg/item.pyi +++ b/src/borg/item.pyi @@ -6,6 +6,11 @@ from .helpers import StableDict def want_bytes(v: Any, *, errors: str = ...) -> bytes: ... def chunks_contents_equal(chunks1: Iterator, chunks2: Iterator) -> bool: ... +MAX_ALIGN_CHUNKS: int +MAX_ALIGN_WORK: int + +def chunks_diff_size(chunks1: list, chunks2: list) -> tuple[int, int]: ... + class PropDict: VALID_KEYS: set[str] = ... def __init__(self, data_dict: dict = None, internal_dict: dict = None, **kw) -> None: ... diff --git a/src/borg/item.pyx b/src/borg/item.pyx index 50e82ed5ca..3740eb28af 100644 --- a/src/borg/item.pyx +++ b/src/borg/item.pyx @@ -1,5 +1,6 @@ import stat -from collections import namedtuple +from collections import Counter, namedtuple +from difflib import SequenceMatcher from libc.string cimport memcmp from cpython.bytes cimport PyBytes_AsStringAndSize @@ -610,6 +611,57 @@ cpdef _init_names(): _init_names() +# Cost limits for the chunk list alignment done by chunks_diff_size(). difflib.SequenceMatcher +# degrades to quadratic runtime on chunk lists that contain the same chunk id over and over again, +# like the ones of a sparse file or a VM image with big all-zero ranges. +MAX_ALIGN_CHUNKS = 1 << 16 # max. length of a chunk list (the common prefix/suffix is not counted) +MAX_ALIGN_WORK = 1 << 20 # max. estimated matcher work: for each chunk of list 1, its count in list 2 + + +def chunks_diff_size(chunks1, chunks2): + """ + Determine how many content bytes chunks2 added and how many chunks1 removed. + + Both chunk lists are aligned as sequences, like a text diff aligns lines: the chunks that are + part of the alignment are the unchanged content, all others are counted - the ones of chunks1 + as removed bytes, the ones of chunks2 as added bytes. Insertions, removals, moves and + duplicated chunks are therefore all reflected by the byte counts. + + For chunk lists that are too long or too repetitive to align within MAX_ALIGN_CHUNKS / + MAX_ALIGN_WORK, the chunks are only counted per chunk id and just the surplus occurrences of an + id are counted as added/removed, so moved chunks do not show up in the byte counts then. + """ + # The common prefix and suffix align trivially. Stripping them is what makes the usual cases + # cheap (e.g. a file that was appended to) and it also keeps the matcher away from the long + # runs of identical chunks it is slow on. + start, end1, end2 = 0, len(chunks1), len(chunks2) + while start < end1 and start < end2 and chunks1[start].id == chunks2[start].id: + start += 1 + while end1 > start and end2 > start and chunks1[end1 - 1].id == chunks2[end2 - 1].id: + end1 -= 1 + end2 -= 1 + mid1, mid2 = chunks1[start:end1], chunks2[start:end2] + ids1 = [chunk.id for chunk in mid1] + ids2 = [chunk.id for chunk in mid2] + counts2 = Counter(ids2) + work = sum(counts2[cid] for cid in ids1) + if max(len(ids1), len(ids2)) > MAX_ALIGN_CHUNKS or work > MAX_ALIGN_WORK: + counts1 = Counter(ids1) + # a chunk id always refers to the same content, thus also always to the same size. + sizes = {chunk.id: chunk.size for chunk in mid1} + sizes.update((chunk.id, chunk.size) for chunk in mid2) + added = sum((counts2[cid] - counts1[cid]) * sizes[cid] for cid in counts2 if counts2[cid] > counts1[cid]) + removed = sum((counts1[cid] - counts2[cid]) * sizes[cid] for cid in counts1 if counts1[cid] > counts2[cid]) + return added, removed + added = removed = 0 + matcher = SequenceMatcher(a=ids1, b=ids2, autojunk=False) # autojunk would skip popular chunks + for tag, i1, i2, j1, j2 in matcher.get_opcodes(): + if tag != 'equal': + removed += sum(chunk.size for chunk in mid1[i1:i2]) + added += sum(chunk.size for chunk in mid2[j1:j2]) + return added, removed + + class DiffChange: """ Stores a change in a diff. @@ -732,14 +784,7 @@ class ItemDiff: if self._item1.chunks == self._item2.chunks: # same chunk lists, same content (e.g. a file that was only touched): no content change. return False - # the byte counts sum up the chunks only present in one of the items, so both are 0 if the content - # only changed by reordering or duplicating chunks - it is a content change nevertheless. - chunk_ids1 = {c.id for c in self._item1.chunks} - chunk_ids2 = {c.id for c in self._item2.chunks} - added_ids = chunk_ids2 - chunk_ids1 - removed_ids = chunk_ids1 - chunk_ids2 - added = self._item2.get_size(consider_ids=added_ids) - removed = self._item1.get_size(consider_ids=removed_ids) + added, removed = chunks_diff_size(self._item1.chunks, self._item2.chunks) self._changes['content'] = DiffChange("modified", {"added": added, "removed": removed}) return True diff --git a/src/borg/testsuite/archiver/diff_cmd_test.py b/src/borg/testsuite/archiver/diff_cmd_test.py index 18e2b47144..29b29c4cb2 100644 --- a/src/borg/testsuite/archiver/diff_cmd_test.py +++ b/src/borg/testsuite/archiver/diff_cmd_test.py @@ -756,16 +756,46 @@ def test_reordered_chunks(archivers, request): granularity_sleep() # the same-size rewrite must get a new ctime, or the files cache would reuse the old chunks create_regular_file(archiver.input_path, "file_swapped", contents=chunk_b + chunk_a) cmd(archiver, "create", "--chunker-params", "fixed,1024", "test1", "input") - # the same chunks in a different order: the content changed, but no bytes were added or removed. + # aligning the chunk lists keeps one of the two chunks, the other one is removed and added again. output = cmd(archiver, "diff", "--content-only", "test0", "test1") - assert_line_exists(output.splitlines(), r"^modified:\s+0 B\s+0 B input/file_swapped$") + assert_line_exists(output.splitlines(), r"^modified:\s+\+1.0 kB\s+-1.0 kB input/file_swapped$") output = cmd(archiver, "diff", "--content-only", "--json-lines", "test0", "test1") joutput = [json.loads(line) for line in output.splitlines() if line.startswith("{")] - assert joutput == [{"changes": [{"added": 0, "removed": 0, "type": "modified"}], "path": "input/file_swapped"}] - # such a change is counted, although it contributes no bytes. + assert joutput == [ + {"changes": [{"added": 1024, "removed": 1024, "type": "modified"}], "path": "input/file_swapped"} + ] output = cmd(archiver, "diff", "--stats", "--content-only", "test0", "test1") lines = output.splitlines() assert "Changed items: 1" in lines - assert_line_exists(lines, r"^Added size: 0 B$") - assert_line_exists(lines, r"^Removed size: 0 B$") + assert_line_exists(lines, r"^Added size: 1.02 kB$") + assert_line_exists(lines, r"^Removed size: 1.02 kB$") assert_line_not_exists(lines, r"^Items with unknown size changes:") + + +def test_duplicated_chunks(archivers, request): + """Duplicating the chunks of a file adds content, although it does not add any new chunk id.""" + archiver = request.getfixturevalue(archivers) + cmd(archiver, "repo-create", RK_ENCRYPTION) + chunk_a = b"a" * 1024 + create_regular_file(archiver.input_path, "file_repeated", contents=chunk_a) + cmd(archiver, "create", "--chunker-params", "fixed,1024", "test0", "input") + create_regular_file(archiver.input_path, "file_repeated", contents=chunk_a * 3) + cmd(archiver, "create", "--chunker-params", "fixed,1024", "test1", "input") + # the file grew by two chunks, even though both versions only use the one chunk id. + output = cmd(archiver, "diff", "--content-only", "--json-lines", "test0", "test1") + joutput = [json.loads(line) for line in output.splitlines() if line.startswith("{")] + assert joutput == [{"changes": [{"added": 2048, "removed": 0, "type": "modified"}], "path": "input/file_repeated"}] + + +def test_inserted_chunk(archivers, request): + """A chunk inserted into a file counts as added bytes only, the chunks behind it are just moved.""" + archiver = request.getfixturevalue(archivers) + cmd(archiver, "repo-create", RK_ENCRYPTION) + chunk_a, chunk_b, chunk_c = b"a" * 1024, b"b" * 1024, b"c" * 1024 + create_regular_file(archiver.input_path, "file_grown", contents=chunk_a + chunk_b) + cmd(archiver, "create", "--chunker-params", "fixed,1024", "test0", "input") + create_regular_file(archiver.input_path, "file_grown", contents=chunk_a + chunk_c + chunk_b) + cmd(archiver, "create", "--chunker-params", "fixed,1024", "test1", "input") + output = cmd(archiver, "diff", "--content-only", "--json-lines", "test0", "test1") + joutput = [json.loads(line) for line in output.splitlines() if line.startswith("{")] + assert joutput == [{"changes": [{"added": 1024, "removed": 0, "type": "modified"}], "path": "input/file_grown"}] diff --git a/src/borg/testsuite/item_test.py b/src/borg/testsuite/item_test.py index 1cfd3a254a..044796a72c 100644 --- a/src/borg/testsuite/item_test.py +++ b/src/borg/testsuite/item_test.py @@ -1,7 +1,7 @@ import pytest from ..cache import ChunkListEntry -from ..item import Item, ItemDiff, chunks_contents_equal +from ..item import MAX_ALIGN_CHUNKS, MAX_ALIGN_WORK, Item, ItemDiff, chunks_contents_equal, chunks_diff_size from ..helpers import StableDict from ..helpers.msgpack import Timestamp from ..platformflags import is_pypy @@ -188,3 +188,49 @@ def test_item_diff_time_ns_resolution(ctime1_ns, ctime2_ns, change_expected): diff = ItemDiff("p", item1, item2, iter([]), iter([]), can_compare_chunk_ids=True) assert (diff.ctime() is not None) == change_expected assert diff.mtime() is None + + +# chunk ids for the chunks_diff_size tests, all chunks are 10 bytes long. +CA, CB, CC, CD = (ChunkListEntry(bytes([n]) * 32, 10) for n in range(4)) + + +@pytest.mark.parametrize( + "chunks1, chunks2, expected", + [ + ([], [], (0, 0)), + ([CA, CB], [CA, CB], (0, 0)), # identical + ([CA, CB], [CA, CB, CC], (10, 0)), # appended + ([CA, CB, CC], [CA, CB], (0, 10)), # truncated + ([CA, CB], [CC, CA, CB], (10, 0)), # prepended + ([CA, CB], [CA, CC, CB], (10, 0)), # inserted in the middle + ([CA, CB, CC], [CA, CD, CC], (10, 10)), # replaced in the middle + ([CA, CB], [CB, CA], (10, 10)), # swapped: one of the two chunks aligns, the other one moved + ([CA, CB, CC], [CC, CB, CA], (20, 20)), # reversed: only one chunk aligns + ([CA], [CA, CA, CA], (20, 0)), # duplicated: no new chunk id, but the content grew + ([CA, CA, CA], [CA], (0, 20)), # de-duplicated + ([CA, CB], [CC, CD], (20, 20)), # nothing in common + ], +) +def test_chunks_diff_size(chunks1, chunks2, expected): + assert chunks_diff_size(chunks1, chunks2) == expected + + +def test_chunks_diff_size_over_length_limit(): + """Above MAX_ALIGN_CHUNKS the chunk lists are not aligned, the chunk ids are only counted.""" + chunks1 = [ChunkListEntry((n + 1).to_bytes(32, "big"), 10) for n in range(MAX_ALIGN_CHUNKS + 1)] + # the first and the last chunk differ, so neither a common prefix nor a common suffix is stripped. + chunks2 = [CA] + chunks1[1:-1] + [CB] + assert chunks_diff_size(chunks1, chunks2) == (20, 20) + # a pure reordering is not detected on this code path, thus no bytes are reported. + assert chunks_diff_size(chunks1, chunks1[::-1]) == (0, 0) + + +def test_chunks_diff_size_over_work_limit(): + """Chunk lists that repeat the same chunk id too often are not aligned either.""" + n = int(MAX_ALIGN_WORK**0.5) + 1 # n * n occurrences of the same id exceed the work limit + chunks1 = [CA] * n + [CB] + chunks2 = [CB] + [CA] * n + # the same multiset of chunks, only reordered: not detected without aligning the lists. + assert chunks_diff_size(chunks1, chunks2) == (0, 0) + # a chunk that really was added is still counted correctly. + assert chunks_diff_size(chunks1, chunks2 + [CC]) == (10, 0)