From 2037ccae8d1b31175386c74dae857de782528960 Mon Sep 17 00:00:00 2001 From: Parman Mohammadalizadeh Date: Sat, 1 Aug 2026 22:49:11 +0200 Subject: [PATCH 001/144] analyze: add --json output, fixes #9992 Every other status-type command (info, repo-info, repo-list, list, diff, prune) can emit JSON, so tooling does not have to scrape their text. borg analyze was the odd one out, although its numbers are exactly what monitoring wants. --json emits the numbers the text report is rendered from, as raw byte values, for the default mode (dedup_size, hotspots) as well as for --by-name (by_name). The compression factor is left out: it is stored_size / source_size, and "n/a" is not a useful JSON value. To keep one source of truth, the analysis methods now return their numbers and the printing moved into report_*() methods that format them. The text output is unchanged, byte for byte. hotspots is null rather than empty when fewer than two archives matched: the hot spots were then not computed at all, which is different from having computed them and found nothing. --- docs/internals/frontends.rst | 78 ++++++++++ docs/usage/analyze.rst.inc | 16 +- src/borg/archiver/_common.py | 1 + src/borg/archiver/analyze_cmd.py | 147 ++++++++++++++---- .../testsuite/archiver/analyze_cmd_test.py | 127 ++++++++++++++- 5 files changed, 339 insertions(+), 30 deletions(-) diff --git a/docs/internals/frontends.rst b/docs/internals/frontends.rst index 9b9bbe8fb3..c42702bae2 100644 --- a/docs/internals/frontends.rst +++ b/docs/internals/frontends.rst @@ -561,6 +561,84 @@ Example (excerpt) of ``borg diff --json-lines``:: {"path": "file3", "changes": [{"type": "removed", "size": 0}]} +Archive Analysis +++++++++++++++++ + +:ref:`borg_analyze` ``--json`` emits the numbers of its text report as one object. All sizes are +byte values; the compression factor the text report shows is ``stored_size / source_size``. + +Without ``--by-name``, the *dedup_size* and *hotspots* keys are present. + +*dedup_size* describes the considered set of archives: + +considered_archives + Number of archives matching the archive filters +total_archives + Number of non-deleted archives in the repository +whole_repository + True if no archive was left over by the filters, so the considered set is the whole + repository. Every referenced chunk is then trivially exclusive to the set, and the + *exclusive* key is absent. +deduplicated + Object with *source_size* and *stored_size*: the summed size of the union of chunks the + considered archives reference, chunks shared within the set counted once +exclusive + Object with *source_size* and *stored_size*: the chunks referenced only by the considered + set, i.e. what deleting the whole set would free. Absent if *whole_repository* is true. +unreferenced + Object with *stored_size* and *chunks*: the chunks no non-deleted archive references, which + ``borg compact`` could free. Their source size is not known, as it is only recorded in the + archives referencing a chunk. +total_chunks + Number of chunks in the repository chunk index +missing_chunks + Number of chunks referenced by an archive but absent from the repository chunk index + +*hotspots* is a list of objects with *path* (directory path) and *size* (bytes of chunks added or +removed in that directory between consecutive archives), busiest directory first. It is ``null`` +if fewer than two archives matched, as hot spots need at least two archives to compare. + +With ``--by-name``, the *by_name* key is present instead, decomposing the whole repository: + +archives + Number of non-deleted archives in the repository +names + List of objects with *name*, *archives* (number of archives with that name), *source_size* + and *stored_size*. The sizes are what is exclusive to that name: no archive of another name + references those chunks. Biggest *stored_size* first. +shared + Object with *source_size* and *stored_size*: the chunks referenced by two or more names +unreferenced + As above +total + Object with *archives*, *source_size* and *stored_size*. Each chunk is counted in exactly one + of *names*, *shared* and *unreferenced*, so the *names* and *shared* sizes add up to *total*. +total_chunks, missing_chunks + As above + +Example of ``borg analyze -a 'sh:userA-*' --json``:: + + { + "dedup_size": { + "considered_archives": 2, + "deduplicated": {"source_size": 3000, "stored_size": 3536}, + "exclusive": {"source_size": 2000, "stored_size": 3338}, + "missing_chunks": 0, + "total_archives": 3, + "total_chunks": 13, + "unreferenced": {"chunks": 0, "stored_size": 0}, + "whole_repository": false + }, + "encryption": {"encryption": "aes256-ocb", "id_hash": "sha256"}, + "hotspots": [{"path": "home/user/src", "size": 1000}], + "repository": { + "id": "06e4027d32f8eae8333f8fe06b1c2c46bf12f22ad10bd4d04a0f30751a26d77b", + "last_modified": "2026-08-01T22:46:05.886533", + "location": "/home/user/repository" + } + } + + .. _msgid: Message IDs diff --git a/docs/usage/analyze.rst.inc b/docs/usage/analyze.rst.inc index dd652d549b..91f2cfdeff 100644 --- a/docs/usage/analyze.rst.inc +++ b/docs/usage/analyze.rst.inc @@ -17,6 +17,8 @@ borg analyze +-----------------------------------------------------------------------------+----------------------------------------------+-----------------------------------------------------------------------------------------------------------------------------+ | | ``--by-name`` | decompose the whole repository by archive name (not combinable with archive filters) | +-----------------------------------------------------------------------------+----------------------------------------------+-----------------------------------------------------------------------------------------------------------------------------+ + | | ``--json`` | format output as JSON | + +-----------------------------------------------------------------------------+----------------------------------------------+-----------------------------------------------------------------------------------------------------------------------------+ | .. class:: borg-common-opt-ref | | | | :ref:`common_options` | @@ -53,7 +55,8 @@ borg analyze options - --by-name decompose the whole repository by archive name (not combinable with archive filters) + --by-name decompose the whole repository by archive name (not combinable with archive filters) + --json format output as JSON :ref:`common_options` @@ -131,4 +134,13 @@ the sizes of added and removed chunks per direct parent directory, and outputs a You can use that list to find directories with a lot of "activity" — maybe some of these are temporary or cache directories you forgot to exclude. To avoid including these unwanted directories in your backups, you can carefully exclude them in ``borg create`` (for future -backups) or use ``borg recreate`` to recreate existing archives without them. \ No newline at end of file +backups) or use ``borg recreate`` to recreate existing archives without them. + +**JSON output** + +With ``--json``, the same numbers are emitted as a single JSON object instead of the text +report, with raw byte values rather than formatted sizes. The default mode fills the +*dedup_size* and *hotspots* keys, ``--by-name`` fills the *by_name* key. The compression +factor is not included: it is ``stored_size / source_size``. + +See :ref:`json_output` for the object's structure. \ No newline at end of file diff --git a/src/borg/archiver/_common.py b/src/borg/archiver/_common.py index 77787c26a5..6d3cee1089 100644 --- a/src/borg/archiver/_common.py +++ b/src/borg/archiver/_common.py @@ -280,6 +280,7 @@ def wrapper(self, args, repository, manifest, **kwargs): "key_files": "Internals -> Data structures and file formats -> Key files", "borg_key_export": "borg key export --help", "internals_hashindex": "Internals -> Data structures and file formats -> HashIndex", + "json_output": "Internals -> All about JSON: How to develop frontends", } diff --git a/src/borg/archiver/analyze_cmd.py b/src/borg/archiver/analyze_cmd.py index 55330d447b..639fb8a79b 100644 --- a/src/borg/archiver/analyze_cmd.py +++ b/src/borg/archiver/analyze_cmd.py @@ -5,7 +5,7 @@ from ..archive import Archive from ..cache import get_archive_references, list_archive_reference_caches from ..constants import * # NOQA -from ..helpers import bin_to_hex, Error, format_file_size +from ..helpers import basic_json_data, bin_to_hex, Error, format_file_size, json_print from ..helpers import ProgressIndicatorPercent from ..helpers.argparsing import ArgumentParser from ..manifest import Manifest @@ -42,23 +42,40 @@ def __init__(self, args, repository, manifest): def analyze(self): logger.info("Starting archives analysis...") + json_data = {} if self.args.json else None if self.args.by_name: # the decomposition is inherently repository-wide: "shared" and "unreferenced" can only # be determined by looking at every archive, so archive filters must not be applied. filters = ["match_archives", "first", "last", "older", "newer", "oldest", "newest"] if any(getattr(self.args, name, None) for name in filters): raise Error("--by-name analyzes the whole repository and cannot be combined with archive filters.") - self.analyze_by_name() + by_name = self.analyze_by_name() + if json_data is None: + self.report_by_name(by_name) + else: + json_data["by_name"] = by_name else: considered_infos = self.manifest.archives.list_considering(self.args) if not considered_infos: raise Error("No archives match the given selection criteria.") - self.analyze_dedup_size(considered_infos) + dedup_size = self.analyze_dedup_size(considered_infos) + if json_data is None: + self.report_dedup_size(dedup_size) + else: + json_data["dedup_size"] = dedup_size if len(considered_infos) >= 2: self.analyze_hotspots(considered_infos) - self.report_hotspots() + hotspots = self.hotspots() else: logger.info("Skipping hot-spot analysis (needs at least 2 matching archives).") + hotspots = None # not computed, as opposed to computed and empty + if json_data is None: + if hotspots is not None: + self.report_hotspots(hotspots) + else: + json_data["hotspots"] = hotspots + if json_data is not None: + json_print(basic_json_data(self.manifest, extra=json_data)) logger.info("Finished archives analysis.") def fmt(self, value): @@ -101,7 +118,7 @@ def mark_references(self, archive_infos, update_flags): pi.finish() return missing - def analyze_by_name(self) -> None: + def analyze_by_name(self) -> dict: """Decompose the whole repository by archive name. Archives sharing a name form a series, so a name usually groups all backups of one source; @@ -114,6 +131,8 @@ def analyze_by_name(self) -> None: This needs only a single pass over all archives: each chunk records the name that first referenced it, and a reference from a different name sets the F_MULTI bit. + + Returns the decomposition as raw byte values, for report_by_name() or --json. """ all_infos = self.manifest.archives.list() # non-deleted archives if not all_infos: @@ -159,6 +178,29 @@ def update_flags(flags, owner=owner): total_plaintext = shared[0] + sum(v[0] for v in exclusive.values()) total_stored = shared[1] + sum(v[1] for v in exclusive.values()) + + return { + "archives": len(all_infos), + # biggest exclusive consumer first - that is what one would act on + "names": [ + { + "name": name, + "archives": archives_per_name[name], + "source_size": exclusive[name][0], + "stored_size": exclusive[name][1], + } + for name in sorted(names, key=lambda n: exclusive[n][1], reverse=True) + ], + "shared": {"source_size": shared[0], "stored_size": shared[1]}, + "unreferenced": {"stored_size": unref_stored, "chunks": unref_count}, + "total": {"archives": len(all_infos), "source_size": total_plaintext, "stored_size": total_stored}, + "total_chunks": total_count, + "missing_chunks": missing, + } + + def report_by_name(self, data) -> None: + """Print the --by-name decomposition computed by analyze_by_name().""" + names = [entry["name"] for entry in data["names"]] width = min(max([30] + [len(name) for name in names]), 60) def row(label, archives, source, stored, *, source_known=True): @@ -169,19 +211,24 @@ def row(label, archives, source, stored, *, source_known=True): print() print("Repository decomposition by archive name") print("=" * (width + 51)) - print(f"{len(all_infos)} archive(s) with {len(names)} distinct name(s)") + print(f"{data['archives']} archive(s) with {len(names)} distinct name(s)") print() print(f"{'name':<{width}}{'archives':>10}{'source':>14}{'stored':>14}{'compression':>13}") - # biggest exclusive consumer first - that is what one would act on - for name in sorted(names, key=lambda n: exclusive[n][1], reverse=True): - source, stored = exclusive[name] - row(name, archives_per_name[name], source, stored) - row("(shared by 2+ names)", "", shared[0], shared[1]) - row("(unreferenced)", "", 0, unref_stored, source_known=False) + for entry in data["names"]: + row(entry["name"], entry["archives"], entry["source_size"], entry["stored_size"]) + row("(shared by 2+ names)", "", data["shared"]["source_size"], data["shared"]["stored_size"]) + row("(unreferenced)", "", 0, data["unreferenced"]["stored_size"], source_known=False) print("-" * (width + 51)) - row("total (deduplicated)", len(all_infos), total_plaintext, total_stored) + row( + "total (deduplicated)", + data["total"]["archives"], + data["total"]["source_size"], + data["total"]["stored_size"], + ) print() - print(f"Unreferenced: {unref_count} of {total_count} chunks in the repository index.") + print( + f"Unreferenced: {data['unreferenced']['chunks']} of {data['total_chunks']} chunks in the repository index." + ) print() print("Each chunk is counted in exactly one row, so the rows add up to the total.") print("A name row shows what is exclusive to it: no archive of another name references these") @@ -197,8 +244,8 @@ def row(label, archives, source, stored, *, source_known=True): print(f"{'':<12} archives count as unreferenced (borg compact keeps them if the") print(f"{'':<12} repository is damaged).") - def analyze_dedup_size(self, considered_infos) -> None: - """Compute and report the deduplicated size of the considered set of archives. + def analyze_dedup_size(self, considered_infos) -> dict: + """Compute the deduplicated size of the considered set of archives. For both the plaintext (uncompressed source) size and the stored (compressed, as stored in the repository) size, two figures are reported: @@ -221,6 +268,8 @@ def analyze_dedup_size(self, considered_infos) -> None: plaintext size (which is 0 in the repo index) from the per-archive references cache. These in-memory mutations are never persisted: write_chunkindex_to_repo() zeroes flags and size, and close() only serializes F_NEW entries (there are none here). + + Returns the sizes as raw byte values, for report_dedup_size() or --json. """ considered_ids = {info.id for info in considered_infos} all_infos = self.manifest.archives.list() # non-deleted archives; the rest = all - considered @@ -252,28 +301,54 @@ def analyze_dedup_size(self, considered_infos) -> None: # chunk is trivially exclusive to it, so that line would just repeat the deduplicated size. whole_repo = not rest_infos + data = { + "considered_archives": len(considered_infos), + "total_archives": len(all_infos), + "whole_repository": whole_repo, + "deduplicated": {"source_size": set_plaintext, "stored_size": set_stored}, + "unreferenced": {"stored_size": unref_stored, "chunks": unref_count}, + "total_chunks": total_count, + "missing_chunks": missing, + } + if not whole_repo: + data["exclusive"] = {"source_size": excl_plaintext, "stored_size": excl_stored} + return data + + def report_dedup_size(self, data) -> None: + """Print the deduplicated sizes computed by analyze_dedup_size().""" + whole_repo = data["whole_repository"] + def row(label, source, stored, *, source_known=True): sizes = f"{self.fmt(source) if source_known else 'n/a':>14}{self.fmt(stored):>14}" ratio = self.factor(stored, source) if source_known else "n/a" print(f"{label:<26}{sizes}{ratio:>13}") + considered = data["considered_archives"] + total_archives = data["total_archives"] + deduplicated = data["deduplicated"] + unreferenced = data["unreferenced"] + print() if whole_repo: print("Deduplicated size of the whole repository") print("=" * 67) - print(f"Archives: {len(all_infos)} (all archives in the repository)") + print(f"Archives: {total_archives} (all archives in the repository)") else: - print(f"Deduplicated size of the {len(considered_infos)} considered archive(s)") + print(f"Deduplicated size of the {considered} considered archive(s)") print("=" * 67) - print(f"Considered archives: {len(considered_infos)} (of {len(all_infos)} in the repository)") + print(f"Considered archives: {considered} (of {total_archives} in the repository)") print() print(f"{'':26}{'source':>14}{'stored':>14}{'compression':>13}") - row("Deduplicated size:" if whole_repo else "Deduplicated size of set:", set_plaintext, set_stored) + row( + "Deduplicated size:" if whole_repo else "Deduplicated size of set:", + deduplicated["source_size"], + deduplicated["stored_size"], + ) if not whole_repo: - row("Exclusive size of set:", excl_plaintext, excl_stored) - row("Unreferenced chunks:", 0, unref_stored, source_known=False) + row("Exclusive size of set:", data["exclusive"]["source_size"], data["exclusive"]["stored_size"]) + row("Unreferenced chunks:", 0, unreferenced["stored_size"], source_known=False) print() - print(f"Unreferenced: {unref_count} of {total_count} chunks in the repository index.") + print(f"Unreferenced: {unreferenced['chunks']} of {data['total_chunks']} chunks in the repository index.") print() if whole_repo: print(f"{'source':<12} = uncompressed source data size (each chunk counted once)") @@ -341,13 +416,21 @@ def analyze_path_change(path): if directory_path not in base: analyze_path_change(directory_path) - def report_hotspots(self): + def hotspots(self) -> list: + """The hot spots collected by analyze_hotspots(), busiest directory first.""" + return [ + {"path": directory_path, "size": self.difference_by_path[directory_path]} + for directory_path in sorted( + self.difference_by_path, key=lambda p: self.difference_by_path[p], reverse=True + ) + ] + + def report_hotspots(self, hotspots): print() print("chunks added or removed by directory path") print("=========================================") - for directory_path in sorted(self.difference_by_path, key=lambda p: self.difference_by_path[p], reverse=True): - difference = self.difference_by_path[directory_path] - print(f"{directory_path}: {difference}") + for hotspot in hotspots: + print(f"{hotspot['path']}: {hotspot['size']}") class AnalyzeMixIn: @@ -420,6 +503,15 @@ def build_parser_analyze(self, subparsers, common_parser, mid_common_parser): are temporary or cache directories you forgot to exclude. To avoid including these unwanted directories in your backups, you can carefully exclude them in ``borg create`` (for future backups) or use ``borg recreate`` to recreate existing archives without them. + + **JSON output** + + With ``--json``, the same numbers are emitted as a single JSON object instead of the text + report, with raw byte values rather than formatted sizes. The default mode fills the + *dedup_size* and *hotspots* keys, ``--by-name`` fills the *by_name* key. The compression + factor is not included: it is ``stored_size / source_size``. + + See :ref:`json_output` for the object's structure. """ ) subparser = ArgumentParser(parents=[common_parser], description=self.do_analyze.__doc__, epilog=analyze_epilog) @@ -430,4 +522,5 @@ def build_parser_analyze(self, subparsers, common_parser, mid_common_parser): action="store_true", help="decompose the whole repository by archive name (not combinable with archive filters)", ) + subparser.add_argument("--json", action="store_true", help="format output as JSON") define_archive_filters_group(subparser) diff --git a/src/borg/testsuite/archiver/analyze_cmd_test.py b/src/borg/testsuite/archiver/analyze_cmd_test.py index 7afd3f3797..38d4967057 100644 --- a/src/borg/testsuite/archiver/analyze_cmd_test.py +++ b/src/borg/testsuite/archiver/analyze_cmd_test.py @@ -1,10 +1,11 @@ +import json import pathlib import re import pytest from ...constants import * # NOQA -from ...helpers import Error +from ...helpers import Error, format_file_size from . import cmd, generate_archiver_tests, RK_ENCRYPTION pytest_generate_tests = lambda metafunc: generate_archiver_tests(metafunc, kinds="local") # NOQA @@ -192,3 +193,127 @@ def test_analyze_dedup_size_single_archive(archivers, request): assert re.search(r"Deduplicated size of set:\s*1\.00 kB", output) # file1 is also in other-1, so nothing is exclusive to only-1 assert re.search(r"Exclusive size of set:\s*0 B", output) + + +def test_analyze_json(archivers, request): + """--json reports the same numbers as the text report, as raw byte values.""" + archiver = request.getfixturevalue(archivers) + + cmd(archiver, "repo-create", RK_ENCRYPTION) + input_path = pathlib.Path(archiver.input_path) + + # same layout as test_analyze_dedup_size: each file is one 1000 byte chunk. + (input_path / "shared").write_text("s" * 1000) + (input_path / "a_only").write_text("a" * 1000) + cmd(archiver, "create", "userA-1", archiver.input_path) + cmd(archiver, "create", "userA-2", archiver.input_path) + + (input_path / "a_only").unlink() + (input_path / "b_only").write_text("b" * 1000) + cmd(archiver, "create", "userB-1", archiver.input_path) + + json_output = cmd(archiver, "analyze", "-a", "sh:userA-*", "--json") + dedup_size = json.loads(json_output)["dedup_size"] + assert dedup_size["considered_archives"] == 2 + assert dedup_size["total_archives"] == 3 + assert dedup_size["whole_repository"] is False + assert dedup_size["deduplicated"]["source_size"] == 2000 # {shared, a_only} + assert dedup_size["exclusive"]["source_size"] == 1000 # {a_only}, "shared" is also in userB-1 + assert dedup_size["unreferenced"] == {"stored_size": 0, "chunks": 0} + assert dedup_size["missing_chunks"] == 0 + assert dedup_size["total_chunks"] > 0 + + # the stored sizes are compression dependent, so relate them to the text report instead of + # hardcoding them: both columns of a row are what the text report formats from these values. + text = cmd(archiver, "analyze", "-a", "sh:userA-*") + for key, label in [("deduplicated", "Deduplicated size of set:"), ("exclusive", "Exclusive size of set:")]: + source, stored = dedup_size[key]["source_size"], dedup_size[key]["stored_size"] + assert stored > 0 + row = rf"^{re.escape(label)}\s+{re.escape(format_file_size(source))}\s+{re.escape(format_file_size(stored))}\s" + assert re.search(row, text, re.MULTILINE) + + # the text report itself is not printed in JSON mode + assert "Deduplicated size of set:" not in json_output + + +def test_analyze_json_whole_repository(archivers, request): + """Without an archive filter every chunk is trivially exclusive, so no exclusive size is given.""" + archiver = request.getfixturevalue(archivers) + + cmd(archiver, "repo-create", RK_ENCRYPTION) + input_path = pathlib.Path(archiver.input_path) + (input_path / "file1").write_text("x" * 1000) + cmd(archiver, "create", "one", archiver.input_path) + (input_path / "file2").write_text("y" * 1000) + cmd(archiver, "create", "two", archiver.input_path) + + dedup_size = json.loads(cmd(archiver, "analyze", "--json"))["dedup_size"] + assert dedup_size["whole_repository"] is True + assert dedup_size["considered_archives"] == 2 + assert dedup_size["total_archives"] == 2 + assert dedup_size["deduplicated"]["source_size"] == 2000 + assert "exclusive" not in dedup_size + + +def test_analyze_json_hotspots(archivers, request): + """The hot spots are a list of path/size objects, busiest first; null if they were not computed.""" + archiver = request.getfixturevalue(archivers) + + cmd(archiver, "repo-create", RK_ENCRYPTION) + input_path = pathlib.Path(archiver.input_path) + + (input_path / "file1").write_text("1") + cmd(archiver, "create", "archive", archiver.input_path) + + # only one matching archive: nothing to compare against, so hot spots are not computed + assert json.loads(cmd(archiver, "analyze", "-a", "archive", "--json"))["hotspots"] is None + + (input_path / "file2").write_text("22") + cmd(archiver, "create", "archive", archiver.input_path) + + # the 2nd archive added one chunk of 2 bytes below the input directory + hotspots = json.loads(cmd(archiver, "analyze", "-a", "archive", "--json"))["hotspots"] + assert [hotspot for hotspot in hotspots if hotspot["path"].endswith("/input")] == [ + {"path": str(input_path).removeprefix("/"), "size": 2} + ] + # busiest directory first, as in the text report + assert [hotspot["size"] for hotspot in hotspots] == sorted((hotspot["size"] for hotspot in hotspots), reverse=True) + + +def test_analyze_json_by_name(archivers, request): + """--by-name --json decomposes the repository into per-name exclusive, shared and unreferenced.""" + archiver = request.getfixturevalue(archivers) + + cmd(archiver, "repo-create", RK_ENCRYPTION) + input_path = pathlib.Path(archiver.input_path) + + # same layout as test_analyze_by_name + (input_path / "shared").write_text("s" * 1000) + (input_path / "a_only").write_text("a" * 1000) + cmd(archiver, "create", "alpha", archiver.input_path) + cmd(archiver, "create", "alpha", archiver.input_path) + + (input_path / "a_only").unlink() + (input_path / "b_only").write_text("b" * 1000) + cmd(archiver, "create", "beta", archiver.input_path) + + result = json.loads(cmd(archiver, "analyze", "--by-name", "--json")) + assert "dedup_size" not in result and "hotspots" not in result + by_name = result["by_name"] + + assert by_name["archives"] == 3 + assert {entry["name"]: entry["archives"] for entry in by_name["names"]} == {"alpha": 2, "beta": 1} + assert {entry["name"]: entry["source_size"] for entry in by_name["names"]} == {"alpha": 1000, "beta": 1000} + assert by_name["shared"]["source_size"] == 1000 + assert by_name["total"]["archives"] == 3 + assert by_name["total"]["source_size"] == 3000 # 1000 alpha + 1000 beta + 1000 shared + assert by_name["total"]["stored_size"] > 0 + assert by_name["missing_chunks"] == 0 + + # every chunk is counted in exactly one row, so the rows add up to the total + for size in ("source_size", "stored_size"): + assert sum(entry[size] for entry in by_name["names"]) + by_name["shared"][size] == by_name["total"][size] + # biggest exclusive consumer first + assert [entry["stored_size"] for entry in by_name["names"]] == sorted( + (entry["stored_size"] for entry in by_name["names"]), reverse=True + ) From 05509440228ea84a5e7b6059a41dae4a65b50462 Mon Sep 17 00:00:00 2001 From: Parman Mohammadalizadeh Date: Mon, 3 Aug 2026 08:53:04 +0200 Subject: [PATCH 002/144] analyze: fix the hot-spot JSON test on Windows The test rebuilt the expected hot-spot path from the input directory, stripping a leading slash. Archived paths are normalized, and on Windows that also drops the drive colon (C:\Users -> C/Users), so the expectation read D:/a/... where borg had stored D/a/.... Assert the size of the input directory's hot spot by path suffix, like the text-report test above already does, and check the paths against what the text report prints instead of rebuilding them. --- src/borg/testsuite/archiver/analyze_cmd_test.py | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/src/borg/testsuite/archiver/analyze_cmd_test.py b/src/borg/testsuite/archiver/analyze_cmd_test.py index 38d4967057..df49bfc124 100644 --- a/src/borg/testsuite/archiver/analyze_cmd_test.py +++ b/src/borg/testsuite/archiver/analyze_cmd_test.py @@ -273,9 +273,13 @@ def test_analyze_json_hotspots(archivers, request): # the 2nd archive added one chunk of 2 bytes below the input directory hotspots = json.loads(cmd(archiver, "analyze", "-a", "archive", "--json"))["hotspots"] - assert [hotspot for hotspot in hotspots if hotspot["path"].endswith("/input")] == [ - {"path": str(input_path).removeprefix("/"), "size": 2} - ] + assert [hotspot["size"] for hotspot in hotspots if hotspot["path"].endswith("/input")] == [2] + # paths and sizes are the ones the text report prints. Archived paths are normalized + # (a Windows "D:/x" is stored as "D/x"), so compare against the report rather than + # rebuilding the path here. + text = cmd(archiver, "analyze", "-a", "archive") + for hotspot in hotspots: + assert f"{hotspot['path']}: {hotspot['size']}" in text # busiest directory first, as in the text report assert [hotspot["size"] for hotspot in hotspots] == sorted((hotspot["size"] for hotspot in hotspots), reverse=True) From afedfb682b1b44777a05ff62d99316461e0009e3 Mon Sep 17 00:00:00 2001 From: Mrityunjay Raj Date: Tue, 21 Jul 2026 19:48:26 +0530 Subject: [PATCH 003/144] check: keep the records of corrupt packs across check cycles Only the intact-pack records are cycle progress; the corrupt ones are kept for repair until the pack verifies intact or is gone from packs/, and their ids are now reported in the check summary. Refs #9696. --- docs/internals/data-structures.rst | 8 ++- src/borg/repository.py | 57 +++++++++++++++---- src/borg/testsuite/repository_test.py | 79 ++++++++++++++++++++++++++- 3 files changed, 128 insertions(+), 16 deletions(-) diff --git a/docs/internals/data-structures.rst b/docs/internals/data-structures.rst index 53bb619efd..39b6294787 100644 --- a/docs/internals/data-structures.rst +++ b/docs/internals/data-structures.rst @@ -36,9 +36,11 @@ config/ cache/ checked-packs - repository check progress (partial checks, full checks' checkpointing), - the set of packs checked so far this cycle (pack id -> timestamp, result), - as a hashtable with an appended integrity hash + repository check results (pack id -> timestamp, result), as a hashtable with an + appended integrity hash. Records of intact packs hold the check progress (partial + checks, full checks' checkpointing) and are dropped when a check cycle completes. + Records of corrupt packs are kept for repair until the pack verifies intact or is + no longer listed in packs/. There is a list of pointers to archive objects in this directory: diff --git a/src/borg/repository.py b/src/borg/repository.py index 254c094589..2a126d4545 100644 --- a/src/borg/repository.py +++ b/src/borg/repository.py @@ -313,11 +313,14 @@ def superseded_gap_ranges(reader, chunks, pack_id, obj_ranges, pack_size): class PackTracker: - """Packs verified in the current check cycle, mapping pack_id -> (timestamp, result). + """Pack verification results, mapping pack_id -> (timestamp, result). A cycle is one full pass over packs/; --max-duration may spread it over several partial checks. + Intact records (result=1) hold the cycle progress and are dropped when the cycle completes. + Corrupt records (result=0) are kept across cycles until the pack verifies intact or is no longer + listed in packs/. Stored at cache/checked-packs as the serialized table with a sha256 over it appended. - new() starts a cycle, load() resumes the stored one. + new() starts an empty tracker, load() reads the stored one. """ NAME = "cache/checked-packs" @@ -377,6 +380,31 @@ def get(self, pack_id): def record(self, pack_id, ok): self.table[pack_id] = self.Entry(timestamp=int(time.time()), result=int(ok)) + def corrupt_ids(self): + """Return the ids of the packs recorded corrupt, sorted.""" + return sorted(pack_id for pack_id, entry in self.table.items() if not entry.result) + + def drop_ok(self): + """Remove the intact-pack records, keeping the corrupt ones.""" + # the keys are collected first because the table must not be mutated while iterating it. + for pack_id in [pack_id for pack_id, entry in self.table.items() if entry.result]: + del self.table[pack_id] + + def finish_cycle(self, pack_ids): + """End a completed cycle: drop the intact records and the corrupt records of packs that are + gone, then store the remaining records (or delete the stored object if none remain). + + pack_ids is the set of pack ids listed in packs/ during this cycle; a record for an id not in + it refers to a pack that was deleted or compacted away. + """ + self.drop_ok() + for pack_id in [pack_id for pack_id, _ in self.table.items() if pack_id not in pack_ids]: + del self.table[pack_id] + if len(self.table): + self.save() + else: + self.clear() + def save(self): with io.BytesIO() as f: self.table.write(f) @@ -815,7 +843,8 @@ def check(self, repair=False, max_duration=0): rebuild re-reads every pack anyway - so a read-only check just stops and reports it instead of continuing. The index is never rebuilt here in any case: reading every pack to do so would be far too slow and expensive for a routine (e.g. cron) check. Salvaging good objects out of - corrupt packs and dropping those packs is left to repair, refs #8572. + corrupt packs and dropping those packs is left to repair, refs #8572. The ids of the packs + found corrupt are kept in cache/checked-packs for repair, refs #9696. """ def verify(namespace, name): @@ -839,15 +868,15 @@ def store_list(namespace): assert not (repair and partial) mode = "partial" if partial else "full" logger.info(f"Starting {mode} repository check") - if partial: - tracker = PackTracker.load(self.store) - else: - tracker = PackTracker.new(self.store) - tracker.clear() # a full check verifies every pack, so discard the stored cycle - if len(tracker): + tracker = PackTracker.load(self.store) + if not partial: + tracker.drop_ok() # a full check verifies every pack, so it starts a new cycle + if not len(tracker): + logger.info("Starting from beginning.") + elif partial: logger.info(f"Continuing check cycle, {len(tracker)} packs already checked.") else: - logger.info("Starting from beginning.") + logger.info(f"Starting from beginning, re-verifying {len(tracker)} packs recorded corrupt.") t_start = time.monotonic() t_last_checkpoint = t_start index_files = index_errors = 0 @@ -899,11 +928,11 @@ def store_list(namespace): tracker.save() break else: - # scanned all packs without hitting the time limit: the cycle is done, drop the set. + # scanned all packs without hitting the time limit: the cycle is done, drop its progress. if pack_infos: pack_pi.show(current=len(pack_infos)) # finish at 100% logger.info("Finished checking packs.") - tracker.clear() + tracker.finish_cycle({hex_to_bin(info.name) for info in pack_infos}) pack_pi.finish() else: # TODO: --repair will rebuild the index from the packs here instead of stopping (refs #8572). @@ -912,6 +941,10 @@ def store_list(namespace): logger.info( f"Checked {index_files} index files ({index_errors} errors) and {pack_files} packs ({pack_errors} errors)." ) + if index_errors == 0: # the packs were checked, so the corrupt records are from this check + corrupt_ids = tracker.corrupt_ids() + if corrupt_ids: + logger.error("Corrupt packs: " + ", ".join(bin_to_hex(pack_id) for pack_id in corrupt_ids)) if objs_errors == 0: logger.info(f"Finished {mode} repository check, no problems found.") elif repair: diff --git a/src/borg/testsuite/repository_test.py b/src/borg/testsuite/repository_test.py index 1e6d7cfd1e..217ceb95ec 100644 --- a/src/borg/testsuite/repository_test.py +++ b/src/borg/testsuite/repository_test.py @@ -1,4 +1,5 @@ import io +import logging import os import sys from collections import namedtuple @@ -1133,7 +1134,83 @@ def test_check_full_ignores_recorded_set(tmp_path, monkeypatch): assert pack_key in hashed_keys # verified after = PackTracker.load(repository.store) - assert len(after) == 0 # cycle complete, set dropped + assert len(after) == 0 # cycle complete, intact records dropped + + +def _store_corrupt_pack(repository, pack_id): + # the stored content does not hash to pack_id, so verifying it fails. + repository.store_store("packs/" + bin_to_hex(pack_id), b"CORRUPT-does-not-match-name") + return pack_id + + +def test_check_full_keeps_corrupt_record_after_cycle(tmp_path): + # a completed cycle keeps the records of corrupt packs and drops the records of intact ones. + with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: + intact_id, _ = _store_intact_pack(repository) + corrupt_id = _store_corrupt_pack(repository, H(2)) + + assert repository.check(repair=False) is False + + after = PackTracker.load(repository.store) + assert after.corrupt_ids() == [corrupt_id] + assert intact_id not in after.table + + +def test_check_full_reports_corrupt_pack_ids(tmp_path, caplog): + # the summary names the corrupt packs. + with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: + corrupt_id = _store_corrupt_pack(repository, H(1)) + + with caplog.at_level(logging.ERROR, logger="borg.repository"): + assert repository.check(repair=False) is False + + assert f"Corrupt packs: {bin_to_hex(corrupt_id)}" in caplog.text + + +def test_check_full_reverifies_carried_over_corrupt_record(tmp_path, monkeypatch): + # a corrupt record from an earlier cycle is re-verified and dropped once the pack is intact again. + with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: + intact_id, pack_key = _store_intact_pack(repository) + + tracker = PackTracker.new(repository.store) + tracker.record(intact_id, ok=False) # recorded corrupt in an earlier cycle + tracker.save() + + hashed_keys = _spy_hash(repository, monkeypatch) + + assert repository.check(repair=False) is True + assert pack_key in hashed_keys # re-verified + + after = PackTracker.load(repository.store) + assert len(after) == 0 # verified intact, so the corrupt record is dropped + + +def test_check_full_prunes_corrupt_record_of_vanished_pack(tmp_path): + # a corrupt record for a pack that is no longer listed is dropped at cycle end. + with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: + _store_intact_pack(repository) + + tracker = PackTracker.new(repository.store) + tracker.record(H(9), ok=False) # no such pack in packs/ + tracker.save() + + assert repository.check(repair=False) is True + + after = PackTracker.load(repository.store) + assert len(after) == 0 + + +def test_check_partial_keeps_corrupt_record_across_runs(tmp_path): + # corrupt records survive a partial check that completes its cycle, too. + with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: + corrupt_id = _store_corrupt_pack(repository, H(1)) + + assert repository.check(repair=False, max_duration=3600) is False + assert PackTracker.load(repository.store).corrupt_ids() == [corrupt_id] + + # a second run re-verifies it and keeps reporting it. + assert repository.check(repair=False, max_duration=3600) is False + assert PackTracker.load(repository.store).corrupt_ids() == [corrupt_id] def test_check_checked_packs_ignores_foreign_entry_layout(tmp_path): From 4e3372e6b13f1a7a1a5a9dd08f1a23186507ca19 Mon Sep 17 00:00:00 2001 From: Mrityunjay Raj Date: Thu, 23 Jul 2026 02:14:27 +0530 Subject: [PATCH 004/144] check: add --max-age to reuse intact-pack check results across cycles --- docs/internals/data-structures.rst | 6 +- src/borg/archiver/check_cmd.py | 22 ++++- src/borg/repository.py | 35 ++++--- src/borg/testsuite/archiver/check_cmd_test.py | 19 ++++ src/borg/testsuite/repository_test.py | 94 +++++++++++++++++++ 5 files changed, 158 insertions(+), 18 deletions(-) diff --git a/docs/internals/data-structures.rst b/docs/internals/data-structures.rst index 39b6294787..e62a04c4da 100644 --- a/docs/internals/data-structures.rst +++ b/docs/internals/data-structures.rst @@ -39,8 +39,10 @@ cache/ repository check results (pack id -> timestamp, result), as a hashtable with an appended integrity hash. Records of intact packs hold the check progress (partial checks, full checks' checkpointing) and are dropped when a check cycle completes. - Records of corrupt packs are kept for repair until the pack verifies intact or is - no longer listed in packs/. + With ``check --max-age`` they are kept across cycles instead and reused while + younger than the given age. Records of corrupt packs are kept for repair until + the pack verifies intact or is no longer listed in packs/. Records of packs no + longer listed in packs/ are pruned when a cycle completes. There is a list of pointers to archive objects in this directory: diff --git a/src/borg/archiver/check_cmd.py b/src/borg/archiver/check_cmd.py index 682f3a9279..c0d1ea1b7e 100644 --- a/src/borg/archiver/check_cmd.py +++ b/src/borg/archiver/check_cmd.py @@ -4,7 +4,7 @@ from ..archive import ArchiveChecker from ..constants import * # NOQA from ..helpers import set_ec, EXIT_WARNING, CancelledByUser, CommandError, IntegrityError -from ..helpers import yes, ArchiveFormatter +from ..helpers import interval, yes, ArchiveFormatter from ..helpers.argparsing import ArgumentParser from ..logger import create_logger @@ -44,6 +44,8 @@ def do_check(self, args, repository): raise CommandError("--repository-only contradicts the --find-lost-archives option.") if args.repair and args.max_duration: raise CommandError("--repair does not allow --max-duration argument.") + if args.repair and args.max_age: + raise CommandError("--repair does not allow the --max-age option.") if args.max_duration and not args.repo_only: # when doing a partial repo check, we can only do a low-level check of the repository files. # archives check requires that a full repo check was done before and has built/cached a ChunkIndex. @@ -64,7 +66,8 @@ def do_check(self, args, repository): # the repository check has finished, which can take hours. ArchiveFormatter.validate_format(format) if not args.archives_only: - if not repository.check(repair=args.repair, max_duration=args.max_duration): + max_age = int(args.max_age.total_seconds()) if args.max_age else 0 + if not repository.check(repair=args.repair, max_duration=args.max_duration, max_age=max_age): set_ec(EXIT_WARNING) if not args.repo_only and not archive_checker.check( repository, @@ -130,6 +133,12 @@ def build_parser_check(self, subparsers, common_parser, mid_common_parser): ``--max-duration`` you must also pass ``--repository-only``, and must not pass ``--archives-only``, nor ``--repair``. + The ``--max-age`` option keeps the results of previous repository checks and + skips packs whose intact result is younger than the given interval (e.g. + ``--max-age=4w``), spreading the verification cost over repeated checks. + Packs recorded corrupt are always re-verified. ``--max-age`` cannot be + combined with ``--repair``. + **Warning:** Please note that partial repository checks (i.e., running with ``--max-duration``) can only perform non-cryptographic checksum checks on the repository files. Enabling partial repository checks excludes archive checks @@ -222,6 +231,15 @@ def build_parser_check(self, subparsers, common_parser, mid_common_parser): subparser.add_argument( "--find-lost-archives", dest="find_lost_archives", action="store_true", help="attempt to find lost archives" ) + subparser.add_argument( + "--max-age", + metavar="INTERVAL", + dest="max_age", + type=interval, + default=None, + action=Highlander, + help="reuse intact-pack check results younger than INTERVAL (e.g. 4w)", + ) subparser.add_argument( "--max-duration", metavar="SECONDS", diff --git a/src/borg/repository.py b/src/borg/repository.py index 2a126d4545..c3011c9eba 100644 --- a/src/borg/repository.py +++ b/src/borg/repository.py @@ -316,9 +316,9 @@ class PackTracker: """Pack verification results, mapping pack_id -> (timestamp, result). A cycle is one full pass over packs/; --max-duration may spread it over several partial checks. - Intact records (result=1) hold the cycle progress and are dropped when the cycle completes. - Corrupt records (result=0) are kept across cycles until the pack verifies intact or is no longer - listed in packs/. + Intact records (result=1) hold the cycle progress and are dropped when the cycle completes, + or kept across cycles when the cycle is finished with keep_ok. Corrupt records (result=0) + are kept across cycles until the pack verifies intact or is no longer listed in packs/. Stored at cache/checked-packs as the serialized table with a sha256 over it appended. new() starts an empty tracker, load() reads the stored one. """ @@ -390,14 +390,15 @@ def drop_ok(self): for pack_id in [pack_id for pack_id, entry in self.table.items() if entry.result]: del self.table[pack_id] - def finish_cycle(self, pack_ids): - """End a completed cycle: drop the intact records and the corrupt records of packs that are - gone, then store the remaining records (or delete the stored object if none remain). + def finish_cycle(self, pack_ids, keep_ok=False): + """End a completed cycle: drop the intact records (unless keep_ok) and the records of packs + that are gone, then store the remaining records (or delete the stored object if none remain). pack_ids is the set of pack ids listed in packs/ during this cycle; a record for an id not in it refers to a pack that was deleted or compacted away. """ - self.drop_ok() + if not keep_ok: + self.drop_ok() for pack_id in [pack_id for pack_id, _ in self.table.items() if pack_id not in pack_ids]: del self.table[pack_id] if len(self.table): @@ -831,7 +832,7 @@ def info(self): info = dict(id=self.id, version=self.version) return info - def check(self, repair=False, max_duration=0): + def check(self, repair=False, max_duration=0, max_age=0): """Check repository consistency. packs/ and index/ objects are named by the sha256 of their content, so a pack or index file @@ -845,6 +846,9 @@ def check(self, repair=False, max_duration=0): far too slow and expensive for a routine (e.g. cron) check. Salvaging good objects out of corrupt packs and dropping those packs is left to repair, refs #8572. The ids of the packs found corrupt are kept in cache/checked-packs for repair, refs #9696. + + max_age (seconds, 0 = off): keep the intact-pack records across cycles and skip packs whose + intact record is younger than max_age. """ def verify(namespace, name): @@ -869,12 +873,14 @@ def store_list(namespace): mode = "partial" if partial else "full" logger.info(f"Starting {mode} repository check") tracker = PackTracker.load(self.store) - if not partial: - tracker.drop_ok() # a full check verifies every pack, so it starts a new cycle + if not partial and not max_age: + tracker.drop_ok() # without max_age, a full check verifies every pack: start a new cycle if not len(tracker): logger.info("Starting from beginning.") elif partial: - logger.info(f"Continuing check cycle, {len(tracker)} packs already checked.") + logger.info(f"Continuing check cycle, {len(tracker)} pack check results on record.") + elif max_age: + logger.info(f"{len(tracker)} pack check results on record, reusing those younger than max_age.") else: logger.info(f"Starting from beginning, re-verifying {len(tracker)} packs recorded corrupt.") t_start = time.monotonic() @@ -910,8 +916,9 @@ def store_list(namespace): pack_pi.show(increase=1) # advance for skipped packs too, so the bar tracks packs/, not work done pack_id = hex_to_bin(info.name) entry = tracker.get(pack_id) - if entry is not None and entry.result: # intact in this cycle; a corrupt one is verified again - continue + if entry is not None and entry.result: # recorded intact; a corrupt one is verified again + if not max_age or time.time() - entry.timestamp < max_age: + continue pack_files += 1 ok = verify("packs", info.name) if not ok: @@ -932,7 +939,7 @@ def store_list(namespace): if pack_infos: pack_pi.show(current=len(pack_infos)) # finish at 100% logger.info("Finished checking packs.") - tracker.finish_cycle({hex_to_bin(info.name) for info in pack_infos}) + tracker.finish_cycle({hex_to_bin(info.name) for info in pack_infos}, keep_ok=bool(max_age)) pack_pi.finish() else: # TODO: --repair will rebuild the index from the packs here instead of stopping (refs #8572). diff --git a/src/borg/testsuite/archiver/check_cmd_test.py b/src/borg/testsuite/archiver/check_cmd_test.py index ef42ad776b..102ebbca15 100644 --- a/src/borg/testsuite/archiver/check_cmd_test.py +++ b/src/borg/testsuite/archiver/check_cmd_test.py @@ -74,6 +74,25 @@ def test_check_usage(archivers, request): assert "archive2" in output +def test_check_max_age(archivers, request): + archiver = request.getfixturevalue(archivers) + check_cmd_setup(archiver) + + # --repair does not allow --max-age. + if archiver.FORK_DEFAULT: + cmd(archiver, "check", "--repair", "--max-age=1d", exit_code=CommandError().exit_code) + else: + with pytest.raises(CommandError): + cmd(archiver, "check", "--repair", "--max-age=1d") + + # a first check with --max-age records the results, a second one reuses them. + output = cmd(archiver, "check", "-v", "--repository-only", "--max-age=4w", exit_code=0) + assert "Starting full repository check" in output + output = cmd(archiver, "check", "-v", "--repository-only", "--max-age=4w", exit_code=0) + assert "reusing those younger than max_age" in output + assert "no problems found" in output + + def test_date_matching(archivers, request): archiver = request.getfixturevalue(archivers) check_cmd_setup(archiver) diff --git a/src/borg/testsuite/repository_test.py b/src/borg/testsuite/repository_test.py index 217ceb95ec..590763a40d 100644 --- a/src/borg/testsuite/repository_test.py +++ b/src/borg/testsuite/repository_test.py @@ -2,6 +2,7 @@ import logging import os import sys +import time from collections import namedtuple from hashlib import sha256 @@ -1213,6 +1214,99 @@ def test_check_partial_keeps_corrupt_record_across_runs(tmp_path): assert PackTracker.load(repository.store).corrupt_ids() == [corrupt_id] +def test_check_max_age_skips_fresh_ok(tmp_path, monkeypatch): + # with max_age, a pack recorded intact recently is not re-verified, and its record is kept. + with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: + intact_id, pack_key = _store_intact_pack(repository) + + tracker = PackTracker.new(repository.store) + tracker.record(intact_id, ok=True) # fresh timestamp + tracker.save() + + hashed_keys = _spy_hash(repository, monkeypatch) + + assert repository.check(repair=False, max_age=3600) is True + assert pack_key not in hashed_keys # skipped, its record is fresh + + after = PackTracker.load(repository.store) + assert after.table[intact_id].result == 1 # kept across the cycle + + +def test_check_max_age_reverifies_stale_ok(tmp_path, monkeypatch): + # with max_age, a pack whose intact record is older than max_age is re-verified. + with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: + intact_id, pack_key = _store_intact_pack(repository) + + tracker = PackTracker.new(repository.store) + old_ts = int(time.time()) - 100 + tracker.table[intact_id] = PackTracker.Entry(timestamp=old_ts, result=1) + tracker.save() + + hashed_keys = _spy_hash(repository, monkeypatch) + + assert repository.check(repair=False, max_age=50) is True + assert pack_key in hashed_keys # stale, re-verified + + after = PackTracker.load(repository.store) + assert after.table[intact_id].timestamp > old_ts # record refreshed + + +def test_check_max_age_reverifies_corrupt_even_when_fresh(tmp_path, monkeypatch): + # the age window only applies to intact records: a fresh corrupt record is still re-verified. + with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: + intact_id, pack_key = _store_intact_pack(repository) + + tracker = PackTracker.new(repository.store) + tracker.record(intact_id, ok=False) # fresh, but corrupt + tracker.save() + + hashed_keys = _spy_hash(repository, monkeypatch) + + assert repository.check(repair=False, max_age=3600) is True + assert pack_key in hashed_keys # re-verified despite the fresh record + + +def test_check_max_age_prunes_vanished_ok_record(tmp_path): + # with max_age, an intact record for a pack no longer listed is pruned at cycle end. + with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: + intact_id, _ = _store_intact_pack(repository) + + tracker = PackTracker.new(repository.store) + tracker.record(intact_id, ok=True) + tracker.record(H(9), ok=True) # no such pack in packs/ + tracker.save() + + assert repository.check(repair=False, max_age=3600) is True + + after = PackTracker.load(repository.store) + assert intact_id in after.table + assert H(9) not in after.table + + +def test_check_max_age_partial_progress(tmp_path, monkeypatch): + # a partial check with max_age skips packs with a fresh intact record and verifies the rest. + pack_a = fchunk(b"A", chunk_id=H(1)) + pack_a_id = sha256(pack_a).digest() + pack_b = fchunk(b"BB", chunk_id=H(2)) + pack_b_id = sha256(pack_b).digest() + with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: + repository.store_store("packs/" + bin_to_hex(pack_a_id), pack_a) + repository.store_store("packs/" + bin_to_hex(pack_b_id), pack_b) + + tracker = PackTracker.new(repository.store) + tracker.record(pack_a_id, ok=True) # fresh + tracker.save() + + hashed_keys = _spy_hash(repository, monkeypatch) + + assert repository.check(repair=False, max_duration=3600, max_age=3600) is True + assert "packs/" + bin_to_hex(pack_a_id) not in hashed_keys + assert "packs/" + bin_to_hex(pack_b_id) in hashed_keys + + after = PackTracker.load(repository.store) + assert pack_a_id in after.table and pack_b_id in after.table + + def test_check_checked_packs_ignores_foreign_entry_layout(tmp_path): # load() drops a set whose entries have a different layout than Entry, even though its sha256 matches. OtherEntry = namedtuple("OtherEntry", "timestamp result extra") From cf00471de4d9b830cdc13886fae4092d3846141a Mon Sep 17 00:00:00 2001 From: Mrityunjay Raj Date: Fri, 24 Jul 2026 04:12:44 +0530 Subject: [PATCH 005/144] check: keep pack check results unconditionally, --max-age only controls their reuse Records are pruned only for packs no longer listed in packs/, so --max-duration now requires --max-age to make progress. --- docs/internals/data-structures.rst | 12 ++-- src/borg/archiver/check_cmd.py | 37 ++++++----- src/borg/repository.py | 43 ++++--------- src/borg/testsuite/archiver/check_cmd_test.py | 9 ++- src/borg/testsuite/repository_test.py | 62 ++++++++++++------- 5 files changed, 84 insertions(+), 79 deletions(-) diff --git a/docs/internals/data-structures.rst b/docs/internals/data-structures.rst index e62a04c4da..3637448311 100644 --- a/docs/internals/data-structures.rst +++ b/docs/internals/data-structures.rst @@ -37,12 +37,12 @@ config/ cache/ checked-packs repository check results (pack id -> timestamp, result), as a hashtable with an - appended integrity hash. Records of intact packs hold the check progress (partial - checks, full checks' checkpointing) and are dropped when a check cycle completes. - With ``check --max-age`` they are kept across cycles instead and reused while - younger than the given age. Records of corrupt packs are kept for repair until - the pack verifies intact or is no longer listed in packs/. Records of packs no - longer listed in packs/ are pruned when a cycle completes. + appended integrity hash. Records are kept across checks: ``check --max-age`` + skips packs whose intact record is younger than the given age, which also lets + partial checks (``--max-duration``) continue where a previous one stopped. + Records of corrupt packs are kept for repair and always re-verified. Records of + packs no longer listed in packs/ are pruned when a check finishes scanning + packs/. There is a list of pointers to archive objects in this directory: diff --git a/src/borg/archiver/check_cmd.py b/src/borg/archiver/check_cmd.py index c0d1ea1b7e..d0ce135417 100644 --- a/src/borg/archiver/check_cmd.py +++ b/src/borg/archiver/check_cmd.py @@ -46,6 +46,9 @@ def do_check(self, args, repository): raise CommandError("--repair does not allow --max-duration argument.") if args.repair and args.max_age: raise CommandError("--repair does not allow the --max-age option.") + if args.max_duration and not args.max_age: + # partial checks progress by skipping packs whose record is younger than max_age. + raise CommandError("--max-duration requires the --max-age option.") if args.max_duration and not args.repo_only: # when doing a partial repo check, we can only do a low-level check of the repository files. # archives check requires that a full repo check was done before and has built/cached a ChunkIndex. @@ -121,23 +124,25 @@ def build_parser_check(self, subparsers, common_parser, mid_common_parser): repository checks only, or pass ``--archives-only`` to run the archive checks only. - The ``--max-duration`` option can be used to split a long-running repository - check into multiple partial checks. After the given number of seconds, the check - is interrupted. The next partial check will continue where the previous one - stopped, until the full repository has been checked. Assuming a complete check - would take 7 hours, then running a daily check with ``--max-duration=3600`` - (1 hour) would result in one full repository check per week. Doing a full - repository check aborts any previous partial check; the next partial check will - restart from the beginning. With partial repository checks you can run neither - archive checks, nor enable repair mode. Consequently, if you want to use - ``--max-duration`` you must also pass ``--repository-only``, and must not pass - ``--archives-only``, nor ``--repair``. + The ``--max-age`` option makes the check reuse the results of previous + repository checks: packs whose intact result is younger than the given + interval (e.g. ``--max-age=4w``) are skipped, spreading the verification + cost over repeated checks. Check results are recorded in any case; + ``--max-age`` only controls their reuse. Packs recorded corrupt are always + re-verified. ``--max-age`` cannot be combined with ``--repair``. - The ``--max-age`` option keeps the results of previous repository checks and - skips packs whose intact result is younger than the given interval (e.g. - ``--max-age=4w``), spreading the verification cost over repeated checks. - Packs recorded corrupt are always re-verified. ``--max-age`` cannot be - combined with ``--repair``. + The ``--max-duration`` option can be used to split a long-running repository + check into multiple partial checks. After the given number of seconds, the + check is interrupted. Because a verified pack's result is recorded and + reused, ``--max-duration`` requires ``--max-age``: the next partial check + skips the recently verified packs and continues with the rest, until every + pack has a result younger than ``--max-age``. Assuming a complete check + would take 7 hours, then running a daily check with ``--max-duration=3600 + --max-age=1w`` (1 hour) would result in one full repository verification + per week. With partial repository checks you can run neither archive + checks, nor enable repair mode. Consequently, if you want to use + ``--max-duration`` you must also pass ``--repository-only``, and must not + pass ``--archives-only``, nor ``--repair``. **Warning:** Please note that partial repository checks (i.e., running with ``--max-duration``) can only perform non-cryptographic checksum checks on the diff --git a/src/borg/repository.py b/src/borg/repository.py index c3011c9eba..8ab3035e87 100644 --- a/src/borg/repository.py +++ b/src/borg/repository.py @@ -315,10 +315,9 @@ def superseded_gap_ranges(reader, chunks, pack_id, obj_ranges, pack_size): class PackTracker: """Pack verification results, mapping pack_id -> (timestamp, result). - A cycle is one full pass over packs/; --max-duration may spread it over several partial checks. - Intact records (result=1) hold the cycle progress and are dropped when the cycle completes, - or kept across cycles when the cycle is finished with keep_ok. Corrupt records (result=0) - are kept across cycles until the pack verifies intact or is no longer listed in packs/. + Records are kept across checks: intact records (result=1) are reused by checks run with + max_age, corrupt records (result=0) are kept for repair and always re-verified. Records of + packs no longer listed in packs/ are pruned when a check finishes scanning packs/. Stored at cache/checked-packs as the serialized table with a sha256 over it appended. new() starts an empty tracker, load() reads the stored one. """ @@ -384,21 +383,11 @@ def corrupt_ids(self): """Return the ids of the packs recorded corrupt, sorted.""" return sorted(pack_id for pack_id, entry in self.table.items() if not entry.result) - def drop_ok(self): - """Remove the intact-pack records, keeping the corrupt ones.""" - # the keys are collected first because the table must not be mutated while iterating it. - for pack_id in [pack_id for pack_id, entry in self.table.items() if entry.result]: - del self.table[pack_id] - - def finish_cycle(self, pack_ids, keep_ok=False): - """End a completed cycle: drop the intact records (unless keep_ok) and the records of packs - that are gone, then store the remaining records (or delete the stored object if none remain). - - pack_ids is the set of pack ids listed in packs/ during this cycle; a record for an id not in - it refers to a pack that was deleted or compacted away. + def prune(self, pack_ids): + """Drop the records whose pack id is not in pack_ids (the set of pack ids listed in packs/), + then store the remaining records (or delete the stored object if none remain). """ - if not keep_ok: - self.drop_ok() + # the keys are collected first because the table must not be mutated while iterating it. for pack_id in [pack_id for pack_id, _ in self.table.items() if pack_id not in pack_ids]: del self.table[pack_id] if len(self.table): @@ -847,8 +836,8 @@ def check(self, repair=False, max_duration=0, max_age=0): corrupt packs and dropping those packs is left to repair, refs #8572. The ids of the packs found corrupt are kept in cache/checked-packs for repair, refs #9696. - max_age (seconds, 0 = off): keep the intact-pack records across cycles and skip packs whose - intact record is younger than max_age. + max_age (seconds, 0 = verify every pack): skip packs whose intact record is younger than + max_age. Check results are always recorded and kept. """ def verify(namespace, name): @@ -873,16 +862,12 @@ def store_list(namespace): mode = "partial" if partial else "full" logger.info(f"Starting {mode} repository check") tracker = PackTracker.load(self.store) - if not partial and not max_age: - tracker.drop_ok() # without max_age, a full check verifies every pack: start a new cycle if not len(tracker): logger.info("Starting from beginning.") - elif partial: - logger.info(f"Continuing check cycle, {len(tracker)} pack check results on record.") elif max_age: logger.info(f"{len(tracker)} pack check results on record, reusing those younger than max_age.") else: - logger.info(f"Starting from beginning, re-verifying {len(tracker)} packs recorded corrupt.") + logger.info(f"{len(tracker)} pack check results on record, verifying every pack.") t_start = time.monotonic() t_last_checkpoint = t_start index_files = index_errors = 0 @@ -917,7 +902,7 @@ def store_list(namespace): pack_id = hex_to_bin(info.name) entry = tracker.get(pack_id) if entry is not None and entry.result: # recorded intact; a corrupt one is verified again - if not max_age or time.time() - entry.timestamp < max_age: + if max_age and time.time() - entry.timestamp < max_age: continue pack_files += 1 ok = verify("packs", info.name) @@ -931,15 +916,13 @@ def store_list(namespace): logger.info(f"Checkpointing at pack {info.name}.") tracker.save() if partial and now > t_start + max_duration: - logger.info(f"Finished partial repository check, {len(tracker)} packs checked so far.") - tracker.save() + logger.info(f"Finished partial repository check, {len(tracker)} pack check results on record.") break else: - # scanned all packs without hitting the time limit: the cycle is done, drop its progress. if pack_infos: pack_pi.show(current=len(pack_infos)) # finish at 100% logger.info("Finished checking packs.") - tracker.finish_cycle({hex_to_bin(info.name) for info in pack_infos}, keep_ok=bool(max_age)) + tracker.prune({hex_to_bin(info.name) for info in pack_infos}) pack_pi.finish() else: # TODO: --repair will rebuild the index from the packs here instead of stopping (refs #8572). diff --git a/src/borg/testsuite/archiver/check_cmd_test.py b/src/borg/testsuite/archiver/check_cmd_test.py index 102ebbca15..fd82333764 100644 --- a/src/borg/testsuite/archiver/check_cmd_test.py +++ b/src/borg/testsuite/archiver/check_cmd_test.py @@ -78,15 +78,18 @@ def test_check_max_age(archivers, request): archiver = request.getfixturevalue(archivers) check_cmd_setup(archiver) - # --repair does not allow --max-age. + # --repair does not allow --max-age, and --max-duration requires --max-age. if archiver.FORK_DEFAULT: cmd(archiver, "check", "--repair", "--max-age=1d", exit_code=CommandError().exit_code) + cmd(archiver, "check", "--repository-only", "--max-duration=3600", exit_code=CommandError().exit_code) else: with pytest.raises(CommandError): cmd(archiver, "check", "--repair", "--max-age=1d") + with pytest.raises(CommandError): + cmd(archiver, "check", "--repository-only", "--max-duration=3600") - # a first check with --max-age records the results, a second one reuses them. - output = cmd(archiver, "check", "-v", "--repository-only", "--max-age=4w", exit_code=0) + # a check records its results, a later one with --max-age reuses them. + output = cmd(archiver, "check", "-v", "--repository-only", exit_code=0) assert "Starting full repository check" in output output = cmd(archiver, "check", "-v", "--repository-only", "--max-age=4w", exit_code=0) assert "reusing those younger than max_age" in output diff --git a/src/borg/testsuite/repository_test.py b/src/borg/testsuite/repository_test.py index 590763a40d..c192e12249 100644 --- a/src/borg/testsuite/repository_test.py +++ b/src/borg/testsuite/repository_test.py @@ -1043,7 +1043,7 @@ def test_check_partial_rechecks_pack_sorting_before_checked_one(tmp_path): with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: repository.store_store("packs/" + bin_to_hex(intact_id), intact) - # mark the intact pack as already checked in this cycle. + # mark the intact pack as recently checked. tracker = PackTracker.new(repository.store) tracker.record(intact_id, ok=True) tracker.save() @@ -1053,11 +1053,11 @@ def test_check_partial_rechecks_pack_sorting_before_checked_one(tmp_path): assert bin_to_hex(early_id) < bin_to_hex(intact_id) repository.store_store("packs/" + bin_to_hex(early_id), b"CORRUPT-does-not-match-name") - assert repository.check(repair=False, max_duration=3600) is False + assert repository.check(repair=False, max_duration=3600, max_age=3600) is False def test_check_partial_rechecks_pack_recorded_corrupt(tmp_path): - # a pack recorded corrupt earlier in the cycle is re-verified, so the corruption keeps being reported. + # a pack recorded corrupt earlier is re-verified, so the corruption keeps being reported. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: corrupt_id = H(1) # stored content does not hash to this name repository.store_store("packs/" + bin_to_hex(corrupt_id), b"CORRUPT-does-not-match-name") @@ -1066,7 +1066,7 @@ def test_check_partial_rechecks_pack_recorded_corrupt(tmp_path): tracker.record(corrupt_id, ok=False) tracker.save() - assert repository.check(repair=False, max_duration=3600) is False + assert repository.check(repair=False, max_duration=3600, max_age=3600) is False def _spy_hash(repository, monkeypatch): @@ -1101,12 +1101,12 @@ def test_check_partial_clears_recorded_corruption_when_intact(tmp_path, monkeypa hashed_keys = _spy_hash(repository, monkeypatch) - assert repository.check(repair=False, max_duration=3600) is True + assert repository.check(repair=False, max_duration=3600, max_age=3600) is True assert pack_key in hashed_keys # re-verified def test_check_partial_skips_pack_recorded_intact(tmp_path, monkeypatch): - # a pack recorded intact in this cycle is skipped when a partial check resumes. + # a pack recorded intact recently is skipped when a partial check resumes. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: intact_id, pack_key = _store_intact_pack(repository) @@ -1116,12 +1116,12 @@ def test_check_partial_skips_pack_recorded_intact(tmp_path, monkeypatch): hashed_keys = _spy_hash(repository, monkeypatch) - assert repository.check(repair=False, max_duration=3600) is True + assert repository.check(repair=False, max_duration=3600, max_age=3600) is True assert pack_key not in hashed_keys # skipped, not re-verified def test_check_full_ignores_recorded_set(tmp_path, monkeypatch): - # a full check verifies every pack regardless of the recorded set, then drops the set. + # without max_age, a check verifies every pack regardless of the recorded set, but keeps the records. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: intact_id, pack_key = _store_intact_pack(repository) @@ -1132,10 +1132,10 @@ def test_check_full_ignores_recorded_set(tmp_path, monkeypatch): hashed_keys = _spy_hash(repository, monkeypatch) assert repository.check(repair=False) is True - assert pack_key in hashed_keys # verified + assert pack_key in hashed_keys # verified despite the fresh intact record after = PackTracker.load(repository.store) - assert len(after) == 0 # cycle complete, intact records dropped + assert after.table[intact_id].result == 1 # record kept def _store_corrupt_pack(repository, pack_id): @@ -1144,8 +1144,8 @@ def _store_corrupt_pack(repository, pack_id): return pack_id -def test_check_full_keeps_corrupt_record_after_cycle(tmp_path): - # a completed cycle keeps the records of corrupt packs and drops the records of intact ones. +def test_check_full_keeps_records_after_check(tmp_path): + # a completed check keeps the records of both corrupt and intact packs. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: intact_id, _ = _store_intact_pack(repository) corrupt_id = _store_corrupt_pack(repository, H(2)) @@ -1154,7 +1154,7 @@ def test_check_full_keeps_corrupt_record_after_cycle(tmp_path): after = PackTracker.load(repository.store) assert after.corrupt_ids() == [corrupt_id] - assert intact_id not in after.table + assert after.table[intact_id].result == 1 def test_check_full_reports_corrupt_pack_ids(tmp_path, caplog): @@ -1169,12 +1169,12 @@ def test_check_full_reports_corrupt_pack_ids(tmp_path, caplog): def test_check_full_reverifies_carried_over_corrupt_record(tmp_path, monkeypatch): - # a corrupt record from an earlier cycle is re-verified and dropped once the pack is intact again. + # a corrupt record from an earlier check is re-verified; an intact result replaces it. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: intact_id, pack_key = _store_intact_pack(repository) tracker = PackTracker.new(repository.store) - tracker.record(intact_id, ok=False) # recorded corrupt in an earlier cycle + tracker.record(intact_id, ok=False) # recorded corrupt in an earlier check tracker.save() hashed_keys = _spy_hash(repository, monkeypatch) @@ -1183,13 +1183,13 @@ def test_check_full_reverifies_carried_over_corrupt_record(tmp_path, monkeypatch assert pack_key in hashed_keys # re-verified after = PackTracker.load(repository.store) - assert len(after) == 0 # verified intact, so the corrupt record is dropped + assert after.table[intact_id].result == 1 # verified intact, corrupt record replaced def test_check_full_prunes_corrupt_record_of_vanished_pack(tmp_path): - # a corrupt record for a pack that is no longer listed is dropped at cycle end. + # a corrupt record for a pack that is no longer listed is dropped when the check finishes. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: - _store_intact_pack(repository) + intact_id, _ = _store_intact_pack(repository) tracker = PackTracker.new(repository.store) tracker.record(H(9), ok=False) # no such pack in packs/ @@ -1198,19 +1198,20 @@ def test_check_full_prunes_corrupt_record_of_vanished_pack(tmp_path): assert repository.check(repair=False) is True after = PackTracker.load(repository.store) - assert len(after) == 0 + assert H(9) not in after.table + assert intact_id in after.table def test_check_partial_keeps_corrupt_record_across_runs(tmp_path): - # corrupt records survive a partial check that completes its cycle, too. + # corrupt records survive completed partial checks, too. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: corrupt_id = _store_corrupt_pack(repository, H(1)) - assert repository.check(repair=False, max_duration=3600) is False + assert repository.check(repair=False, max_duration=3600, max_age=3600) is False assert PackTracker.load(repository.store).corrupt_ids() == [corrupt_id] # a second run re-verifies it and keeps reporting it. - assert repository.check(repair=False, max_duration=3600) is False + assert repository.check(repair=False, max_duration=3600, max_age=3600) is False assert PackTracker.load(repository.store).corrupt_ids() == [corrupt_id] @@ -1229,7 +1230,7 @@ def test_check_max_age_skips_fresh_ok(tmp_path, monkeypatch): assert pack_key not in hashed_keys # skipped, its record is fresh after = PackTracker.load(repository.store) - assert after.table[intact_id].result == 1 # kept across the cycle + assert after.table[intact_id].result == 1 # record kept def test_check_max_age_reverifies_stale_ok(tmp_path, monkeypatch): @@ -1267,7 +1268,7 @@ def test_check_max_age_reverifies_corrupt_even_when_fresh(tmp_path, monkeypatch) def test_check_max_age_prunes_vanished_ok_record(tmp_path): - # with max_age, an intact record for a pack no longer listed is pruned at cycle end. + # an intact record for a pack no longer listed is pruned when the check finishes. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: intact_id, _ = _store_intact_pack(repository) @@ -1307,6 +1308,19 @@ def test_check_max_age_partial_progress(tmp_path, monkeypatch): assert pack_a_id in after.table and pack_b_id in after.table +def test_check_max_age_reuses_records_of_plain_check(tmp_path, monkeypatch): + # a check without max_age records its results, a later check with max_age reuses them. + with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: + intact_id, pack_key = _store_intact_pack(repository) + + assert repository.check(repair=False) is True + + hashed_keys = _spy_hash(repository, monkeypatch) + + assert repository.check(repair=False, max_age=3600) is True + assert pack_key not in hashed_keys # skipped, reusing the plain check's record + + def test_check_checked_packs_ignores_foreign_entry_layout(tmp_path): # load() drops a set whose entries have a different layout than Entry, even though its sha256 matches. OtherEntry = namedtuple("OtherEntry", "timestamp result extra") From 556daa23c7a4738dcac8798a6ad4e0c7d8dc68e4 Mon Sep 17 00:00:00 2001 From: Mrityunjay Raj Date: Tue, 28 Jul 2026 23:03:15 +0530 Subject: [PATCH 006/144] check: re-verify future-dated records, reject --archives-only --max-age, log corrupt packs one per line --- docs/internals/data-structures.rst | 3 +-- src/borg/archiver/check_cmd.py | 5 ++++- src/borg/constants.py | 5 +++++ src/borg/repository.py | 12 +++++++++--- src/borg/testsuite/repository_test.py | 4 ++-- 5 files changed, 21 insertions(+), 8 deletions(-) diff --git a/docs/internals/data-structures.rst b/docs/internals/data-structures.rst index 3637448311..048a798500 100644 --- a/docs/internals/data-structures.rst +++ b/docs/internals/data-structures.rst @@ -41,8 +41,7 @@ cache/ skips packs whose intact record is younger than the given age, which also lets partial checks (``--max-duration``) continue where a previous one stopped. Records of corrupt packs are kept for repair and always re-verified. Records of - packs no longer listed in packs/ are pruned when a check finishes scanning - packs/. + packs no longer listed in packs/ are pruned when a check finishes. There is a list of pointers to archive objects in this directory: diff --git a/src/borg/archiver/check_cmd.py b/src/borg/archiver/check_cmd.py index d0ce135417..30413389c8 100644 --- a/src/borg/archiver/check_cmd.py +++ b/src/borg/archiver/check_cmd.py @@ -46,9 +46,12 @@ def do_check(self, args, repository): raise CommandError("--repair does not allow --max-duration argument.") if args.repair and args.max_age: raise CommandError("--repair does not allow the --max-age option.") + if args.archives_only and args.max_age: + # --max-age only affects the repository check, which --archives-only skips. + raise CommandError("--archives-only does not allow the --max-age option.") if args.max_duration and not args.max_age: # partial checks progress by skipping packs whose record is younger than max_age. - raise CommandError("--max-duration requires the --max-age option.") + raise CommandError("--max-duration requires the --max-age option, e.g. --max-age=4w.") if args.max_duration and not args.repo_only: # when doing a partial repo check, we can only do a low-level check of the repository files. # archives check requires that a full repo check was done before and has built/cached a ChunkIndex. diff --git a/src/borg/constants.py b/src/borg/constants.py index 42b8ea07ce..7d3874a9c6 100644 --- a/src/borg/constants.py +++ b/src/borg/constants.py @@ -71,6 +71,11 @@ # MAX_OBJECT_SIZE = MAX_DATA_SIZE + len(PUT header) MAX_OBJECT_SIZE = MAX_DATA_SIZE + 41 # see assertion at end of repository module +# Clock skew is the difference between the clocks of the machines writing to a repository (seconds). +# A check result timestamp up to this far in the future still counts as recent; further ahead than +# this, the pack is re-verified. +MAX_CLOCK_SKEW = 7200 # [s] + # How many segment files Borg puts into a single directory by default. DEFAULT_SEGMENTS_PER_DIR = 1000 diff --git a/src/borg/repository.py b/src/borg/repository.py index 8ab3035e87..c9de5c4d72 100644 --- a/src/borg/repository.py +++ b/src/borg/repository.py @@ -901,8 +901,11 @@ def store_list(namespace): pack_pi.show(increase=1) # advance for skipped packs too, so the bar tracks packs/, not work done pack_id = hex_to_bin(info.name) entry = tracker.get(pack_id) - if entry is not None and entry.result: # recorded intact; a corrupt one is verified again - if max_age and time.time() - entry.timestamp < max_age: + # skip the pack if its recorded intact result is younger than max_age; a timestamp up + # to MAX_CLOCK_SKEW in the future still counts as recent, further ahead it is re-verified. + if entry is not None and entry.result and max_age: + age = time.time() - entry.timestamp + if -MAX_CLOCK_SKEW <= age < max_age: continue pack_files += 1 ok = verify("packs", info.name) @@ -934,7 +937,10 @@ def store_list(namespace): if index_errors == 0: # the packs were checked, so the corrupt records are from this check corrupt_ids = tracker.corrupt_ids() if corrupt_ids: - logger.error("Corrupt packs: " + ", ".join(bin_to_hex(pack_id) for pack_id in corrupt_ids)) + # one id per line (the list can be long). + logger.error(f"Found {len(corrupt_ids)} corrupt pack(s):") + for pack_id in corrupt_ids: + logger.error(f"Corrupt pack: {bin_to_hex(pack_id)}") if objs_errors == 0: logger.info(f"Finished {mode} repository check, no problems found.") elif repair: diff --git a/src/borg/testsuite/repository_test.py b/src/borg/testsuite/repository_test.py index c192e12249..47a1004e76 100644 --- a/src/borg/testsuite/repository_test.py +++ b/src/borg/testsuite/repository_test.py @@ -1120,7 +1120,7 @@ def test_check_partial_skips_pack_recorded_intact(tmp_path, monkeypatch): assert pack_key not in hashed_keys # skipped, not re-verified -def test_check_full_ignores_recorded_set(tmp_path, monkeypatch): +def test_check_without_max_age_verifies_all_but_keeps_records(tmp_path, monkeypatch): # without max_age, a check verifies every pack regardless of the recorded set, but keeps the records. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: intact_id, pack_key = _store_intact_pack(repository) @@ -1165,7 +1165,7 @@ def test_check_full_reports_corrupt_pack_ids(tmp_path, caplog): with caplog.at_level(logging.ERROR, logger="borg.repository"): assert repository.check(repair=False) is False - assert f"Corrupt packs: {bin_to_hex(corrupt_id)}" in caplog.text + assert f"Corrupt pack: {bin_to_hex(corrupt_id)}" in caplog.text def test_check_full_reverifies_carried_over_corrupt_record(tmp_path, monkeypatch): From e2c03cb8ba8739f60bdb12bde9d480dbc127b8b0 Mon Sep 17 00:00:00 2001 From: Mrityunjay Raj Date: Tue, 28 Jul 2026 23:33:04 +0530 Subject: [PATCH 007/144] check: fail and report known-corrupt packs even on interrupted partial checks, show reused-result count --- src/borg/repository.py | 32 +++++++++++++++++---------- src/borg/testsuite/repository_test.py | 32 +++++++++++++++++++++++++++ 2 files changed, 52 insertions(+), 12 deletions(-) diff --git a/src/borg/repository.py b/src/borg/repository.py index c9de5c4d72..7686c7a354 100644 --- a/src/borg/repository.py +++ b/src/borg/repository.py @@ -871,7 +871,7 @@ def store_list(namespace): t_start = time.monotonic() t_last_checkpoint = t_start index_files = index_errors = 0 - pack_files = pack_errors = 0 + pack_files = pack_errors = pack_skipped = 0 # index and packs get separate progress indicators, each running from 0% to 100%. # the index is checked first and in full, on partial checks too: it is small, and index errors # stop the pack check below. @@ -906,6 +906,7 @@ def store_list(namespace): if entry is not None and entry.result and max_age: age = time.time() - entry.timestamp if -MAX_CLOCK_SKEW <= age < max_age: + pack_skipped += 1 continue pack_files += 1 ok = verify("packs", info.name) @@ -931,23 +932,30 @@ def store_list(namespace): # TODO: --repair will rebuild the index from the packs here instead of stopping (refs #8572). logger.error("Repository index is corrupted and must be repaired; skipping the pack check.") objs_errors = index_errors + pack_errors - logger.info( - f"Checked {index_files} index files ({index_errors} errors) and {pack_files} packs ({pack_errors} errors)." + summary = ( + f"Checked {index_files} index files ({index_errors} errors) " + f"and {pack_files} packs ({pack_errors} errors)." ) - if index_errors == 0: # the packs were checked, so the corrupt records are from this check - corrupt_ids = tracker.corrupt_ids() - if corrupt_ids: - # one id per line (the list can be long). - logger.error(f"Found {len(corrupt_ids)} corrupt pack(s):") - for pack_id in corrupt_ids: - logger.error(f"Corrupt pack: {bin_to_hex(pack_id)}") - if objs_errors == 0: + if pack_skipped: + summary += f" Reused {pack_skipped} recent pack check result(s)." + logger.info(summary) + # every pack recorded corrupt, not only those verified this run; empty if the index is + # corrupt, since then no packs were scanned. + corrupt_ids = tracker.corrupt_ids() if index_errors == 0 else [] + if corrupt_ids: + # one id per line (the list can be long). + logger.error(f"Found {len(corrupt_ids)} corrupt pack(s):") + for pack_id in corrupt_ids: + logger.error(f"Corrupt pack: {bin_to_hex(pack_id)}") + # fail if this run found errors, or any pack is recorded corrupt. + problems = objs_errors != 0 or bool(corrupt_ids) + if not problems: logger.info(f"Finished {mode} repository check, no problems found.") elif repair: logger.error(f"Finished {mode} repository check, errors found (repository repair not implemented).") else: logger.error(f"Finished {mode} repository check, errors found.") - return objs_errors == 0 or repair + return not problems or repair def list(self, limit=None, marker=None): """ diff --git a/src/borg/testsuite/repository_test.py b/src/borg/testsuite/repository_test.py index 47a1004e76..f074d3ade6 100644 --- a/src/borg/testsuite/repository_test.py +++ b/src/borg/testsuite/repository_test.py @@ -1215,6 +1215,38 @@ def test_check_partial_keeps_corrupt_record_across_runs(tmp_path): assert PackTracker.load(repository.store).corrupt_ids() == [corrupt_id] +def test_check_partial_break_reports_unreached_corrupt_record(tmp_path, monkeypatch, caplog): + # a partial check that stops before re-reaching a carried-over corrupt record still fails and + # reports it, so the timeout does not hide known corruption. + with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: + intact_id, _ = _store_intact_pack(repository) + # a corrupt pack whose id sorts after the intact one, so the scan reaches the intact pack first. + corrupt_id = b"\xff" * 32 + assert bin_to_hex(intact_id) < bin_to_hex(corrupt_id) + repository.store_store("packs/" + bin_to_hex(corrupt_id), b"CORRUPT-does-not-match-name") + + tracker = PackTracker.new(repository.store) + tracker.record(corrupt_id, ok=False) # recorded corrupt by an earlier check + tracker.save() + + # jump the clock past max_duration once the first pack has been hashed, so the scan breaks + # before it reaches the corrupt pack. + clock = {"t": 0} + monkeypatch.setattr(time, "monotonic", lambda: clock["t"]) + orig_hash = repository.store.hash + + def hash_then_advance(key): + result = orig_hash(key) + clock["t"] = 10**9 + return result + + monkeypatch.setattr(repository.store, "hash", hash_then_advance) + + with caplog.at_level(logging.ERROR, logger="borg.repository"): + assert repository.check(repair=False, max_duration=1, max_age=3600) is False + assert f"Corrupt pack: {bin_to_hex(corrupt_id)}" in caplog.text + + def test_check_max_age_skips_fresh_ok(tmp_path, monkeypatch): # with max_age, a pack recorded intact recently is not re-verified, and its record is kept. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: From d46523109664e815655751007230ef46cdf7770a Mon Sep 17 00:00:00 2001 From: Mrityunjay Raj Date: Wed, 29 Jul 2026 13:26:40 +0530 Subject: [PATCH 008/144] check: test clock-skew reuse window and archives-only rejection, assert partial break skips corrupt pack --- src/borg/archiver/check_cmd.py | 3 +- src/borg/testsuite/archiver/check_cmd_test.py | 5 +- src/borg/testsuite/repository_test.py | 60 +++++++++++++++---- 3 files changed, 56 insertions(+), 12 deletions(-) diff --git a/src/borg/archiver/check_cmd.py b/src/borg/archiver/check_cmd.py index 30413389c8..333ccefa0b 100644 --- a/src/borg/archiver/check_cmd.py +++ b/src/borg/archiver/check_cmd.py @@ -132,7 +132,8 @@ def build_parser_check(self, subparsers, common_parser, mid_common_parser): interval (e.g. ``--max-age=4w``) are skipped, spreading the verification cost over repeated checks. Check results are recorded in any case; ``--max-age`` only controls their reuse. Packs recorded corrupt are always - re-verified. ``--max-age`` cannot be combined with ``--repair``. + re-verified. ``--max-age`` affects only the repository check and cannot be + combined with ``--archives-only`` or ``--repair``. The ``--max-duration`` option can be used to split a long-running repository check into multiple partial checks. After the given number of seconds, the diff --git a/src/borg/testsuite/archiver/check_cmd_test.py b/src/borg/testsuite/archiver/check_cmd_test.py index fd82333764..fb225f1d37 100644 --- a/src/borg/testsuite/archiver/check_cmd_test.py +++ b/src/borg/testsuite/archiver/check_cmd_test.py @@ -78,13 +78,16 @@ def test_check_max_age(archivers, request): archiver = request.getfixturevalue(archivers) check_cmd_setup(archiver) - # --repair does not allow --max-age, and --max-duration requires --max-age. + # --repair and --archives-only do not allow --max-age, and --max-duration requires --max-age. if archiver.FORK_DEFAULT: cmd(archiver, "check", "--repair", "--max-age=1d", exit_code=CommandError().exit_code) + cmd(archiver, "check", "--archives-only", "--max-age=1d", exit_code=CommandError().exit_code) cmd(archiver, "check", "--repository-only", "--max-duration=3600", exit_code=CommandError().exit_code) else: with pytest.raises(CommandError): cmd(archiver, "check", "--repair", "--max-age=1d") + with pytest.raises(CommandError): + cmd(archiver, "check", "--archives-only", "--max-age=1d") with pytest.raises(CommandError): cmd(archiver, "check", "--repository-only", "--max-duration=3600") diff --git a/src/borg/testsuite/repository_test.py b/src/borg/testsuite/repository_test.py index f074d3ade6..059b252e4b 100644 --- a/src/borg/testsuite/repository_test.py +++ b/src/borg/testsuite/repository_test.py @@ -9,6 +9,7 @@ import pytest from borghash import HashTableNT +from ..constants import MAX_CLOCK_SKEW from ..helpers import IntegrityError, Location, bin_to_hex from ..hashindex import ChunkIndex from ..repository import Repository, MAX_DATA_SIZE, propagate_rsh, rest_serve_command, PackWriter, PackReader @@ -1216,34 +1217,39 @@ def test_check_partial_keeps_corrupt_record_across_runs(tmp_path): def test_check_partial_break_reports_unreached_corrupt_record(tmp_path, monkeypatch, caplog): - # a partial check that stops before re-reaching a carried-over corrupt record still fails and - # reports it, so the timeout does not hide known corruption. + # a partial check that breaks before re-reaching a corrupt record from an earlier check still + # fails and reports that pack. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: - intact_id, _ = _store_intact_pack(repository) + intact_id, intact_key = _store_intact_pack(repository) # a corrupt pack whose id sorts after the intact one, so the scan reaches the intact pack first. corrupt_id = b"\xff" * 32 assert bin_to_hex(intact_id) < bin_to_hex(corrupt_id) - repository.store_store("packs/" + bin_to_hex(corrupt_id), b"CORRUPT-does-not-match-name") + corrupt_key = "packs/" + bin_to_hex(corrupt_id) + repository.store_store(corrupt_key, b"CORRUPT-does-not-match-name") tracker = PackTracker.new(repository.store) tracker.record(corrupt_id, ok=False) # recorded corrupt by an earlier check tracker.save() - # jump the clock past max_duration once the first pack has been hashed, so the scan breaks - # before it reaches the corrupt pack. - clock = {"t": 0} + # freeze the clock, then jump it past max_duration right after the intact pack is hashed, so the + # pack loop breaks before it reaches the corrupt pack. + clock = {"t": 0.0} monkeypatch.setattr(time, "monotonic", lambda: clock["t"]) + hashed_keys = [] orig_hash = repository.store.hash - def hash_then_advance(key): + def hash_and_advance(key): + hashed_keys.append(key) result = orig_hash(key) - clock["t"] = 10**9 + if key == intact_key: + clock["t"] = 10**9 return result - monkeypatch.setattr(repository.store, "hash", hash_then_advance) + monkeypatch.setattr(repository.store, "hash", hash_and_advance) with caplog.at_level(logging.ERROR, logger="borg.repository"): assert repository.check(repair=False, max_duration=1, max_age=3600) is False + assert corrupt_key not in hashed_keys # the scan broke before reaching the corrupt pack assert f"Corrupt pack: {bin_to_hex(corrupt_id)}" in caplog.text @@ -1284,6 +1290,40 @@ def test_check_max_age_reverifies_stale_ok(tmp_path, monkeypatch): assert after.table[intact_id].timestamp > old_ts # record refreshed +def test_check_max_age_skips_near_future_ok(tmp_path, monkeypatch): + # a record timestamped slightly in the future (clock skew between machines) still counts as + # recent, up to MAX_CLOCK_SKEW ahead, and is not re-verified. + with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: + intact_id, pack_key = _store_intact_pack(repository) + + tracker = PackTracker.new(repository.store) + future_ts = int(time.time()) + MAX_CLOCK_SKEW // 2 + tracker.table[intact_id] = PackTracker.Entry(timestamp=future_ts, result=1) + tracker.save() + + hashed_keys = _spy_hash(repository, monkeypatch) + + assert repository.check(repair=False, max_age=3600) is True + assert pack_key not in hashed_keys # near-future timestamp still counts as recent + + +def test_check_max_age_reverifies_far_future_ok(tmp_path, monkeypatch): + # a record dated more than MAX_CLOCK_SKEW into the future is not plausible clock skew and is + # re-verified. + with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: + intact_id, pack_key = _store_intact_pack(repository) + + tracker = PackTracker.new(repository.store) + far_future_ts = int(time.time()) + MAX_CLOCK_SKEW + 3600 + tracker.table[intact_id] = PackTracker.Entry(timestamp=far_future_ts, result=1) + tracker.save() + + hashed_keys = _spy_hash(repository, monkeypatch) + + assert repository.check(repair=False, max_age=3600) is True + assert pack_key in hashed_keys # too far ahead to be clock skew, re-verified + + def test_check_max_age_reverifies_corrupt_even_when_fresh(tmp_path, monkeypatch): # the age window only applies to intact records: a fresh corrupt record is still re-verified. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: From 8ec9224bfe0d54b72fc9a4369761ab10e9d4a651 Mon Sep 17 00:00:00 2001 From: Mrityunjay Raj Date: Wed, 29 Jul 2026 13:37:40 +0530 Subject: [PATCH 009/144] check: cap max-age future-skew tolerance at the window, document persistent corrupt-pack failure --- src/borg/repository.py | 11 ++++++--- src/borg/testsuite/repository_test.py | 32 +++++++++++++++++++++------ 2 files changed, 33 insertions(+), 10 deletions(-) diff --git a/src/borg/repository.py b/src/borg/repository.py index 7686c7a354..fe3566edcf 100644 --- a/src/borg/repository.py +++ b/src/borg/repository.py @@ -836,6 +836,10 @@ def check(self, repair=False, max_duration=0, max_age=0): corrupt packs and dropping those packs is left to repair, refs #8572. The ids of the packs found corrupt are kept in cache/checked-packs for repair, refs #9696. + Any pack on record as corrupt fails the check, including on a partial run that stops before + re-reaching it. The record clears when the pack verifies intact again, when compact removes + the pack, or when repair salvages and drops it (refs #8572). + max_age (seconds, 0 = verify every pack): skip packs whose intact record is younger than max_age. Check results are always recorded and kept. """ @@ -901,11 +905,12 @@ def store_list(namespace): pack_pi.show(increase=1) # advance for skipped packs too, so the bar tracks packs/, not work done pack_id = hex_to_bin(info.name) entry = tracker.get(pack_id) - # skip the pack if its recorded intact result is younger than max_age; a timestamp up - # to MAX_CLOCK_SKEW in the future still counts as recent, further ahead it is re-verified. + # skip the pack if its recorded intact result is younger than max_age. a future + # timestamp (writer clock ahead of ours) also counts as recent, tolerated up to + # MAX_CLOCK_SKEW or max_age ahead, whichever is smaller. if entry is not None and entry.result and max_age: age = time.time() - entry.timestamp - if -MAX_CLOCK_SKEW <= age < max_age: + if -min(MAX_CLOCK_SKEW, max_age) <= age < max_age: pack_skipped += 1 continue pack_files += 1 diff --git a/src/borg/testsuite/repository_test.py b/src/borg/testsuite/repository_test.py index 059b252e4b..467722caca 100644 --- a/src/borg/testsuite/repository_test.py +++ b/src/borg/testsuite/repository_test.py @@ -1291,8 +1291,8 @@ def test_check_max_age_reverifies_stale_ok(tmp_path, monkeypatch): def test_check_max_age_skips_near_future_ok(tmp_path, monkeypatch): - # a record timestamped slightly in the future (clock skew between machines) still counts as - # recent, up to MAX_CLOCK_SKEW ahead, and is not re-verified. + # with a window wider than MAX_CLOCK_SKEW, a record timestamped up to MAX_CLOCK_SKEW into the + # future (writer clock ahead of ours) still counts as recent and is not re-verified. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: intact_id, pack_key = _store_intact_pack(repository) @@ -1303,13 +1303,13 @@ def test_check_max_age_skips_near_future_ok(tmp_path, monkeypatch): hashed_keys = _spy_hash(repository, monkeypatch) - assert repository.check(repair=False, max_age=3600) is True - assert pack_key not in hashed_keys # near-future timestamp still counts as recent + assert repository.check(repair=False, max_age=MAX_CLOCK_SKEW * 2) is True + assert pack_key not in hashed_keys # within MAX_CLOCK_SKEW ahead, still counts as recent def test_check_max_age_reverifies_far_future_ok(tmp_path, monkeypatch): - # a record dated more than MAX_CLOCK_SKEW into the future is not plausible clock skew and is - # re-verified. + # with a window wider than MAX_CLOCK_SKEW, a record more than MAX_CLOCK_SKEW into the future is + # not plausible clock skew and is re-verified. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: intact_id, pack_key = _store_intact_pack(repository) @@ -1320,10 +1320,28 @@ def test_check_max_age_reverifies_far_future_ok(tmp_path, monkeypatch): hashed_keys = _spy_hash(repository, monkeypatch) - assert repository.check(repair=False, max_age=3600) is True + assert repository.check(repair=False, max_age=MAX_CLOCK_SKEW * 2) is True assert pack_key in hashed_keys # too far ahead to be clock skew, re-verified +def test_check_max_age_reverifies_future_beyond_small_window(tmp_path, monkeypatch): + # the future tolerance is capped at max_age: with a window smaller than MAX_CLOCK_SKEW, a record + # further ahead than max_age is re-verified even though it is within MAX_CLOCK_SKEW. + with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: + intact_id, pack_key = _store_intact_pack(repository) + + small_window = MAX_CLOCK_SKEW // 4 + tracker = PackTracker.new(repository.store) + future_ts = int(time.time()) + MAX_CLOCK_SKEW // 2 # ahead of us, but < MAX_CLOCK_SKEW + tracker.table[intact_id] = PackTracker.Entry(timestamp=future_ts, result=1) + tracker.save() + + hashed_keys = _spy_hash(repository, monkeypatch) + + assert repository.check(repair=False, max_age=small_window) is True + assert pack_key in hashed_keys # further ahead than max_age, re-verified + + def test_check_max_age_reverifies_corrupt_even_when_fresh(tmp_path, monkeypatch): # the age window only applies to intact records: a fresh corrupt record is still re-verified. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: From ff6bb8b9da8d6f0ea6cf667cba975de1acf73508 Mon Sep 17 00:00:00 2001 From: Mrityunjay Raj Date: Mon, 3 Aug 2026 23:58:48 +0530 Subject: [PATCH 010/144] check: calendar-aware --max-age, symmetric clock-skew window Parse --max-age with the calendar-aware relative time marker (m, y measured against now), tolerate MAX_CLOCK_SKEW at both ends of the reuse window, and correct the stale --repository-only comment. --- src/borg/archiver/check_cmd.py | 28 ++++++++++++------- src/borg/repository.py | 15 ++++++---- src/borg/testsuite/archiver/check_cmd_test.py | 2 +- src/borg/testsuite/repository_test.py | 25 +++++++++++++++-- 4 files changed, 50 insertions(+), 20 deletions(-) diff --git a/src/borg/archiver/check_cmd.py b/src/borg/archiver/check_cmd.py index 333ccefa0b..825420e506 100644 --- a/src/borg/archiver/check_cmd.py +++ b/src/borg/archiver/check_cmd.py @@ -4,8 +4,9 @@ from ..archive import ArchiveChecker from ..constants import * # NOQA from ..helpers import set_ec, EXIT_WARNING, CancelledByUser, CommandError, IntegrityError -from ..helpers import interval, yes, ArchiveFormatter +from ..helpers import relative_time_marker_validator, yes, ArchiveFormatter from ..helpers.argparsing import ArgumentParser +from ..helpers.time import archive_ts_now, calculate_relative_offset from ..logger import create_logger @@ -45,6 +46,8 @@ def do_check(self, args, repository): if args.repair and args.max_duration: raise CommandError("--repair does not allow --max-duration argument.") if args.repair and args.max_age: + # reusing recorded results during repair depends on repository repair (refs #8572), which + # does not exist yet, so repair verifies every pack. raise CommandError("--repair does not allow the --max-age option.") if args.archives_only and args.max_age: # --max-age only affects the repository check, which --archives-only skips. @@ -53,9 +56,8 @@ def do_check(self, args, repository): # partial checks progress by skipping packs whose record is younger than max_age. raise CommandError("--max-duration requires the --max-age option, e.g. --max-age=4w.") if args.max_duration and not args.repo_only: - # when doing a partial repo check, we can only do a low-level check of the repository files. - # archives check requires that a full repo check was done before and has built/cached a ChunkIndex. - # also, there is no max_duration support in the archives check code anyway. + # --max-duration time-boxes the repository pack check only; the archives check has no + # max_duration support and builds its own chunk index, so it cannot be split this way. raise CommandError("--repository-only is required for --max-duration support.") if not args.repo_only: # if we need the key later for the archives check, ask NOW for the passphrase! #1931 @@ -72,7 +74,13 @@ def do_check(self, args, repository): # the repository check has finished, which can take hours. ArchiveFormatter.validate_format(format) if not args.archives_only: - max_age = int(args.max_age.total_seconds()) if args.max_age else 0 + if args.max_age: + # resolve the relative marker (e.g. 4w, 12m) to a concrete age in seconds; calendar + # units (m, y) are measured against now, like --older / --newer. + now = archive_ts_now() + max_age = int((now - calculate_relative_offset(args.max_age, now, earlier=True)).total_seconds()) + else: + max_age = 0 if not repository.check(repair=args.repair, max_duration=args.max_duration, max_age=max_age): set_ec(EXIT_WARNING) if not args.repo_only and not archive_checker.check( @@ -129,8 +137,8 @@ def build_parser_check(self, subparsers, common_parser, mid_common_parser): The ``--max-age`` option makes the check reuse the results of previous repository checks: packs whose intact result is younger than the given - interval (e.g. ``--max-age=4w``) are skipped, spreading the verification - cost over repeated checks. Check results are recorded in any case; + timespan (e.g. ``--max-age=4w`` or ``--max-age=12m``) are skipped, spreading + the verification cost over repeated checks. Check results are recorded in any case; ``--max-age`` only controls their reuse. Packs recorded corrupt are always re-verified. ``--max-age`` affects only the repository check and cannot be combined with ``--archives-only`` or ``--repair``. @@ -242,12 +250,12 @@ def build_parser_check(self, subparsers, common_parser, mid_common_parser): ) subparser.add_argument( "--max-age", - metavar="INTERVAL", + metavar="TIMESPAN", dest="max_age", - type=interval, + type=relative_time_marker_validator, default=None, action=Highlander, - help="reuse intact-pack check results younger than INTERVAL (e.g. 4w)", + help="reuse intact-pack check results younger than TIMESPAN, e.g. 4w or 12m", ) subparser.add_argument( "--max-duration", diff --git a/src/borg/repository.py b/src/borg/repository.py index fe3566edcf..b3f97b584e 100644 --- a/src/borg/repository.py +++ b/src/borg/repository.py @@ -841,7 +841,8 @@ def check(self, repair=False, max_duration=0, max_age=0): the pack, or when repair salvages and drops it (refs #8572). max_age (seconds, 0 = verify every pack): skip packs whose intact record is younger than - max_age. Check results are always recorded and kept. + max_age, tolerating up to MAX_CLOCK_SKEW of clock skew at both ends of the window. Check + results are always recorded and kept. """ def verify(namespace, name): @@ -869,7 +870,7 @@ def store_list(namespace): if not len(tracker): logger.info("Starting from beginning.") elif max_age: - logger.info(f"{len(tracker)} pack check results on record, reusing those younger than max_age.") + logger.info(f"{len(tracker)} pack check results on record, reusing those younger than --max-age.") else: logger.info(f"{len(tracker)} pack check results on record, verifying every pack.") t_start = time.monotonic() @@ -905,12 +906,14 @@ def store_list(namespace): pack_pi.show(increase=1) # advance for skipped packs too, so the bar tracks packs/, not work done pack_id = hex_to_bin(info.name) entry = tracker.get(pack_id) - # skip the pack if its recorded intact result is younger than max_age. a future - # timestamp (writer clock ahead of ours) also counts as recent, tolerated up to - # MAX_CLOCK_SKEW or max_age ahead, whichever is smaller. + # skip the pack if its recorded intact result is younger than max_age. the timestamp + # is written by whichever client ran the earlier check, so it may be off by up to + # MAX_CLOCK_SKEW against our clock in either direction; tolerate that much skew at both + # ends of the window, but never more than the window itself. if entry is not None and entry.result and max_age: age = time.time() - entry.timestamp - if -min(MAX_CLOCK_SKEW, max_age) <= age < max_age: + skew = min(MAX_CLOCK_SKEW, max_age) + if -skew <= age <= max_age + skew: pack_skipped += 1 continue pack_files += 1 diff --git a/src/borg/testsuite/archiver/check_cmd_test.py b/src/borg/testsuite/archiver/check_cmd_test.py index fb225f1d37..b3380202dd 100644 --- a/src/borg/testsuite/archiver/check_cmd_test.py +++ b/src/borg/testsuite/archiver/check_cmd_test.py @@ -95,7 +95,7 @@ def test_check_max_age(archivers, request): output = cmd(archiver, "check", "-v", "--repository-only", exit_code=0) assert "Starting full repository check" in output output = cmd(archiver, "check", "-v", "--repository-only", "--max-age=4w", exit_code=0) - assert "reusing those younger than max_age" in output + assert "reusing those younger than --max-age" in output assert "no problems found" in output diff --git a/src/borg/testsuite/repository_test.py b/src/borg/testsuite/repository_test.py index 467722caca..bdac69d4b9 100644 --- a/src/borg/testsuite/repository_test.py +++ b/src/borg/testsuite/repository_test.py @@ -1272,24 +1272,43 @@ def test_check_max_age_skips_fresh_ok(tmp_path, monkeypatch): def test_check_max_age_reverifies_stale_ok(tmp_path, monkeypatch): - # with max_age, a pack whose intact record is older than max_age is re-verified. + # with max_age, a pack whose intact record is older than max_age plus the skew tolerance is re-verified. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: intact_id, pack_key = _store_intact_pack(repository) + max_age = 50 tracker = PackTracker.new(repository.store) - old_ts = int(time.time()) - 100 + old_ts = int(time.time()) - (max_age + MAX_CLOCK_SKEW + 100) # clearly beyond max_age + skew tracker.table[intact_id] = PackTracker.Entry(timestamp=old_ts, result=1) tracker.save() hashed_keys = _spy_hash(repository, monkeypatch) - assert repository.check(repair=False, max_age=50) is True + assert repository.check(repair=False, max_age=max_age) is True assert pack_key in hashed_keys # stale, re-verified after = PackTracker.load(repository.store) assert after.table[intact_id].timestamp > old_ts # record refreshed +def test_check_max_age_skips_stale_within_skew(tmp_path, monkeypatch): + # a record older than max_age but within MAX_CLOCK_SKEW of it still counts as recent (writer clock + # behind ours), so it is not re-verified. + with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: + intact_id, pack_key = _store_intact_pack(repository) + + max_age = MAX_CLOCK_SKEW * 2 # wide window, so the skew tolerance is MAX_CLOCK_SKEW + tracker = PackTracker.new(repository.store) + past_ts = int(time.time()) - (max_age + MAX_CLOCK_SKEW // 2) # past the window, within skew + tracker.table[intact_id] = PackTracker.Entry(timestamp=past_ts, result=1) + tracker.save() + + hashed_keys = _spy_hash(repository, monkeypatch) + + assert repository.check(repair=False, max_age=max_age) is True + assert pack_key not in hashed_keys # within MAX_CLOCK_SKEW past the window, still recent + + def test_check_max_age_skips_near_future_ok(tmp_path, monkeypatch): # with a window wider than MAX_CLOCK_SKEW, a record timestamped up to MAX_CLOCK_SKEW into the # future (writer clock ahead of ours) still counts as recent and is not re-verified. From 0138c35e01884ddb75d79bdd1fa1ca5cc1de9b9d Mon Sep 17 00:00:00 2001 From: Mrityunjay Raj Date: Wed, 29 Jul 2026 15:26:36 +0530 Subject: [PATCH 011/144] check: report missing chunks inverted as chunk -> files -> archives, #9218 --- src/borg/archive.py | 25 +++++++++++++++++-- src/borg/testsuite/archiver/check_cmd_test.py | 24 ++++++++++++------ 2 files changed, 40 insertions(+), 9 deletions(-) diff --git a/src/borg/archive.py b/src/borg/archive.py index 3299d07399..ff76238bf7 100644 --- a/src/borg/archive.py +++ b/src/borg/archive.py @@ -2130,6 +2130,9 @@ def rebuild_archives( ): """Analyze and rebuild archives, expecting some damage and trying to make stuff consistent again.""" + missing_chunk_size: dict = {} # chunk_id -> chunk size in bytes + missing_chunk_refs: defaultdict = defaultdict(lambda: defaultdict(set)) # chunk_id -> {path: {archive_name}} + def add_callback(chunk): id_ = self.key.id_hash(chunk) cdata = self.repo_objs.format(id_, {}, chunk, ro_type=ROBJ_ARCHIVE_STREAM) @@ -2146,16 +2149,22 @@ def add_reference(id_, size, cdata): self.chunks.update_pack_info(pack_results) def verify_file_chunks(archive_name, item): - """Verifies that all file chunks are present. Missing file chunks will be logged.""" + """Verify that all file chunks are present. + + Record each missing chunk's size in missing_chunk_size and, in missing_chunk_refs, the + file path and archive it occurs in. Log each missing chunk at debug level. + """ offset = 0 for chunk in item.chunks: chunk_id, size = chunk if chunk_id not in self.chunks: - logger.error( + logger.debug( "{}: {}: Missing file chunk detected (Byte {}-{}, Chunk {}).".format( archive_name, item.path, offset, offset + size, bin_to_hex(chunk_id) ) ) + missing_chunk_size[chunk_id] = size + missing_chunk_refs[chunk_id][item.path].add(archive_name) self.error_found = True offset += size if "size" in item: @@ -2169,6 +2178,17 @@ def verify_file_chunks(archive_name, item): ) ) + def report_missing_chunks(): + """Log the missing chunks, each with its size and the files and archives referencing it.""" + if not missing_chunk_refs: + return + logger.error("The following chunks are missing in the repository:") + for chunk_id, refs in missing_chunk_refs.items(): + logger.error(f"- Chunk {bin_to_hex(chunk_id)}, {missing_chunk_size[chunk_id]:,} bytes") + for path in sorted(refs): + archive_names = ", ".join(sorted(refs[path])) + logger.error(f" - {path}: {archive_names}") + def robust_iterator(archive): """Iterates through all archive items @@ -2326,6 +2346,7 @@ def valid_item(obj): if archive_id != new_archive_id: self.manifest.archives.delete_by_id(archive_id) pi.finish() + report_missing_chunks() def finish(self): if self.repair: diff --git a/src/borg/testsuite/archiver/check_cmd_test.py b/src/borg/testsuite/archiver/check_cmd_test.py index ef42ad776b..5766093893 100644 --- a/src/borg/testsuite/archiver/check_cmd_test.py +++ b/src/borg/testsuite/archiver/check_cmd_test.py @@ -178,9 +178,13 @@ def test_missing_file_chunk(archivers, request): pytest.fail("should not happen") # convert 'fail' output = cmd(archiver, "check", exit_code=1) - assert "Missing file chunk detected" in output + assert "The following chunks are missing in the repository:" in output + assert bin_to_hex(killed_chunk.id) in output + assert src_file in output output = cmd(archiver, "check", "--repair", exit_code=0) - assert "Missing file chunk detected" in output # repair is not changing anything, just reporting. + # repair is not changing anything, just reporting. + assert "The following chunks are missing in the repository:" in output + assert bin_to_hex(killed_chunk.id) in output # check does not modify the chunks list. for archive_name in ("archive1", "archive2"): @@ -200,7 +204,7 @@ def test_missing_file_chunk(archivers, request): # check should not complain anymore about missing chunks: output = cmd(archiver, "check", "-v", "--repair", exit_code=0) - assert "Missing file chunk detected" not in output + assert "The following chunks are missing in the repository:" not in output def test_missing_archive_item_chunk(archivers, request): @@ -483,11 +487,14 @@ def test_verify_data(archivers, request, init_args): # repair will find the defect chunk and remove it output = cmd(archiver, "check", "--repair", "--verify-data", exit_code=0) assert f"{bin_to_hex(chunk.id)}, integrity error" in output - assert f"{src_file}: Missing file chunk detected" in output + assert "The following chunks are missing in the repository:" in output + assert bin_to_hex(chunk.id) in output + assert src_file in output # run with --verify-data again, it will notice the missing chunk. output = cmd(archiver, "check", "--archives-only", "--verify-data", exit_code=1) - assert f"{src_file}: Missing file chunk detected" in output + assert "The following chunks are missing in the repository:" in output + assert bin_to_hex(chunk.id) in output def test_verify_data_wrong_chunk_content(archivers, request, monkeypatch): @@ -571,12 +578,15 @@ def test_corrupted_file_chunk(archivers, request, init_args): # repair: the defect chunk will be removed. output = cmd(archiver, "check", "--repair", "--verify-data", exit_code=0) assert f"{bin_to_hex(chunk.id)}, integrity error" in output - assert f"{src_file}: Missing file chunk detected" in output + assert "The following chunks are missing in the repository:" in output + assert bin_to_hex(chunk.id) in output + assert src_file in output # run normal check again cmd(archiver, "check", "--repository-only", exit_code=0) output = cmd(archiver, "check", "--archives-only", exit_code=1) - assert f"{src_file}: Missing file chunk detected" in output + assert "The following chunks are missing in the repository:" in output + assert src_file in output @pytest.mark.skip( From 06f5889db4fd56162abe9ef65d26feeece29357a Mon Sep 17 00:00:00 2001 From: Mrityunjay Raj Date: Thu, 30 Jul 2026 18:16:19 +0530 Subject: [PATCH 012/144] check: bound the missing chunks report memory and test its grouping, #9218 Cap distinct chunks and file refs kept for the end-of-run report, combine the two collection dicts into one, and add tests for the grouping and truncation. --- src/borg/archive.py | 50 ++++++++++++++----- src/borg/testsuite/archiver/check_cmd_test.py | 37 ++++++++++++-- 2 files changed, 71 insertions(+), 16 deletions(-) diff --git a/src/borg/archive.py b/src/borg/archive.py index ff76238bf7..6e4cf01fbb 100644 --- a/src/borg/archive.py +++ b/src/borg/archive.py @@ -1849,6 +1849,11 @@ def __next__(self): class ArchiveChecker: + # Bound how many missing file chunks rebuild_archives buffers for its end-of-run report, + # so checking a badly damaged repo with very many missing chunks can not exhaust memory. + MAX_MISSING_CHUNKS = 10000 # max. distinct missing chunk ids kept for the report + MAX_REFS_PER_CHUNK = 100 # max. referencing files kept per missing chunk + def __init__(self): self.error_found = False self.key = None @@ -2130,8 +2135,28 @@ def rebuild_archives( ): """Analyze and rebuild archives, expecting some damage and trying to make stuff consistent again.""" - missing_chunk_size: dict = {} # chunk_id -> chunk size in bytes - missing_chunk_refs: defaultdict = defaultdict(lambda: defaultdict(set)) # chunk_id -> {path: {archive_name}} + # Missing file chunks, collected during the per-archive checks and reported grouped as + # chunk -> files -> archives after all archives were analyzed. Bounded by + # MAX_MISSING_CHUNKS / MAX_REFS_PER_CHUNK. + missing_chunks = {} # chunk_id -> [size, {path: {archive_name}}] + missing_chunks_truncated = False # True once the MAX_MISSING_CHUNKS cap was hit + missing_refs_truncated = set() # chunk_ids whose MAX_REFS_PER_CHUNK cap was hit + + def record_missing_chunk(archive_name, path, chunk_id, size): + nonlocal missing_chunks_truncated + entry = missing_chunks.get(chunk_id) + if entry is None: + if len(missing_chunks) >= self.MAX_MISSING_CHUNKS: + missing_chunks_truncated = True + return + entry = missing_chunks[chunk_id] = [size, {}] + refs = entry[1] + if path in refs: + refs[path].add(archive_name) + elif len(refs) < self.MAX_REFS_PER_CHUNK: + refs[path] = {archive_name} + else: + missing_refs_truncated.add(chunk_id) def add_callback(chunk): id_ = self.key.id_hash(chunk) @@ -2149,11 +2174,7 @@ def add_reference(id_, size, cdata): self.chunks.update_pack_info(pack_results) def verify_file_chunks(archive_name, item): - """Verify that all file chunks are present. - - Record each missing chunk's size in missing_chunk_size and, in missing_chunk_refs, the - file path and archive it occurs in. Log each missing chunk at debug level. - """ + """Verify that all of a file's chunks are present, collecting any missing ones for the report.""" offset = 0 for chunk in item.chunks: chunk_id, size = chunk @@ -2163,8 +2184,7 @@ def verify_file_chunks(archive_name, item): archive_name, item.path, offset, offset + size, bin_to_hex(chunk_id) ) ) - missing_chunk_size[chunk_id] = size - missing_chunk_refs[chunk_id][item.path].add(archive_name) + record_missing_chunk(archive_name, item.path, chunk_id, size) self.error_found = True offset += size if "size" in item: @@ -2179,15 +2199,19 @@ def verify_file_chunks(archive_name, item): ) def report_missing_chunks(): - """Log the missing chunks, each with its size and the files and archives referencing it.""" - if not missing_chunk_refs: + """Report the collected missing chunks, grouped as chunk -> files -> archives.""" + if not missing_chunks: return logger.error("The following chunks are missing in the repository:") - for chunk_id, refs in missing_chunk_refs.items(): - logger.error(f"- Chunk {bin_to_hex(chunk_id)}, {missing_chunk_size[chunk_id]:,} bytes") + for chunk_id, (size, refs) in missing_chunks.items(): + logger.error(f"- Chunk {bin_to_hex(chunk_id)}, {size:,} bytes") for path in sorted(refs): archive_names = ", ".join(sorted(refs[path])) logger.error(f" - {path}: {archive_names}") + if chunk_id in missing_refs_truncated: + logger.error(f" - ... (only the first {self.MAX_REFS_PER_CHUNK} files are listed)") + if missing_chunks_truncated: + logger.error(f"... (only the first {self.MAX_MISSING_CHUNKS} missing chunks are listed)") def robust_iterator(archive): """Iterates through all archive items diff --git a/src/borg/testsuite/archiver/check_cmd_test.py b/src/borg/testsuite/archiver/check_cmd_test.py index 5766093893..444df71a55 100644 --- a/src/borg/testsuite/archiver/check_cmd_test.py +++ b/src/borg/testsuite/archiver/check_cmd_test.py @@ -6,7 +6,7 @@ import pytest -from ...archive import ChunkBuffer +from ...archive import ChunkBuffer, ArchiveChecker from ...constants import * # NOQA from ...helpers import bin_to_hex, msgpack, CommandError, IntegrityError from ...manifest import Manifest @@ -179,8 +179,12 @@ def test_missing_file_chunk(archivers, request): output = cmd(archiver, "check", exit_code=1) assert "The following chunks are missing in the repository:" in output - assert bin_to_hex(killed_chunk.id) in output - assert src_file in output + # archive1 and archive2 share src_file, so the missing chunk appears once, with both archives + # listed on its single reference line. + assert output.count(bin_to_hex(killed_chunk.id)) == 1 + ref_lines = [line for line in output.splitlines() if src_file in line] + assert len(ref_lines) == 1 + assert "archive1" in ref_lines[0] and "archive2" in ref_lines[0] output = cmd(archiver, "check", "--repair", exit_code=0) # repair is not changing anything, just reporting. assert "The following chunks are missing in the repository:" in output @@ -207,6 +211,33 @@ def test_missing_file_chunk(archivers, request): assert "The following chunks are missing in the repository:" not in output +def test_missing_file_chunk_report_truncated(archivers, request): + archiver = request.getfixturevalue(archivers) + check_cmd_setup(archiver) + + # remove several distinct file chunks, so more missing chunks exist than the (patched) report limit. + archive, repository = open_archive(archiver.repository_path, "archive1") + killed_ids = [] + with repository: + for item in archive.iter_items(): + if "chunks" not in item or not item.chunks: + continue + chunk_id = item.chunks[-1].id + if chunk_id not in killed_ids: + repository.delete(chunk_id) + killed_ids.append(chunk_id) + if len(killed_ids) >= 3: + break + assert len(killed_ids) >= 2 # need several distinct missing chunks to exercise truncation + + # cap the report to a single chunk, so the remaining missing chunks are truncated. + with patch.object(ArchiveChecker, "MAX_MISSING_CHUNKS", 1): + output = cmd(archiver, "check", exit_code=1) + assert "The following chunks are missing in the repository:" in output + assert output.count("- Chunk ") == 1 # only one chunk is detailed + assert "only the first 1 missing chunks are listed" in output # the rest are noted as truncated + + def test_missing_archive_item_chunk(archivers, request): archiver = request.getfixturevalue(archivers) check_cmd_setup(archiver) From 36b29672e4ee40957f343b0a2e149cf0a32ff047 Mon Sep 17 00:00:00 2001 From: Mrityunjay Raj Date: Tue, 4 Aug 2026 00:34:35 +0530 Subject: [PATCH 013/144] check: use format_file_size for missing chunk report and test the per-chunk refs cap, #9218 --- src/borg/archive.py | 2 +- src/borg/testsuite/archiver/check_cmd_test.py | 29 +++++++++++++++++++ 2 files changed, 30 insertions(+), 1 deletion(-) diff --git a/src/borg/archive.py b/src/borg/archive.py index 6e4cf01fbb..bbab274445 100644 --- a/src/borg/archive.py +++ b/src/borg/archive.py @@ -2204,7 +2204,7 @@ def report_missing_chunks(): return logger.error("The following chunks are missing in the repository:") for chunk_id, (size, refs) in missing_chunks.items(): - logger.error(f"- Chunk {bin_to_hex(chunk_id)}, {size:,} bytes") + logger.error(f"- Chunk {bin_to_hex(chunk_id)}, {format_file_size(size)}") for path in sorted(refs): archive_names = ", ".join(sorted(refs[path])) logger.error(f" - {path}: {archive_names}") diff --git a/src/borg/testsuite/archiver/check_cmd_test.py b/src/borg/testsuite/archiver/check_cmd_test.py index 444df71a55..086df5f3f4 100644 --- a/src/borg/testsuite/archiver/check_cmd_test.py +++ b/src/borg/testsuite/archiver/check_cmd_test.py @@ -16,6 +16,7 @@ cmd, src_file, create_src_archive, + create_regular_file, open_archive, generate_archiver_tests, read_chunk, @@ -238,6 +239,34 @@ def test_missing_file_chunk_report_truncated(archivers, request): assert "only the first 1 missing chunks are listed" in output # the rest are noted as truncated +def test_missing_file_chunk_refs_truncated(archivers, request): + archiver = request.getfixturevalue(archivers) + cmd(archiver, "repo-create", RK_ENCRYPTION) + + # several distinct files with identical content dedup to the same chunk, so a single missing + # chunk ends up referenced by many files, which exercises the per-chunk reference cap. + for i in range(5): + create_regular_file(archiver.input_path, f"samefile{i}", contents=b"same content for dedup") + cmd(archiver, "create", "archive1", "input") + + archive, repository = open_archive(archiver.repository_path, "archive1") + killed_id = None + with repository: + for item in archive.iter_items(): + if item.path.endswith("samefile0"): + killed_id = item.chunks[0].id + repository.delete(killed_id) + break + assert killed_id is not None + + # cap references per chunk to 2, so the remaining referencing files are truncated. + with patch.object(ArchiveChecker, "MAX_REFS_PER_CHUNK", 2): + output = cmd(archiver, "check", exit_code=1) + assert "The following chunks are missing in the repository:" in output + assert bin_to_hex(killed_id) in output + assert "only the first 2 files are listed" in output # the remaining referencing files are truncated + + def test_missing_archive_item_chunk(archivers, request): archiver = request.getfixturevalue(archivers) check_cmd_setup(archiver) From 63ab760e2c877295a10643660f23724676b4ac3a Mon Sep 17 00:00:00 2001 From: Mrityunjay Raj Date: Wed, 5 Aug 2026 00:45:55 +0530 Subject: [PATCH 014/144] check: run truncation tests binary-safe, report missing-ref total, use a tuple for missing chunk entries, #9218 --- src/borg/archive.py | 15 ++++++++++----- src/borg/testsuite/archiver/check_cmd_test.py | 19 ++++++++++--------- 2 files changed, 20 insertions(+), 14 deletions(-) diff --git a/src/borg/archive.py b/src/borg/archive.py index bbab274445..8e9fbb3e62 100644 --- a/src/borg/archive.py +++ b/src/borg/archive.py @@ -2138,19 +2138,21 @@ def rebuild_archives( # Missing file chunks, collected during the per-archive checks and reported grouped as # chunk -> files -> archives after all archives were analyzed. Bounded by # MAX_MISSING_CHUNKS / MAX_REFS_PER_CHUNK. - missing_chunks = {} # chunk_id -> [size, {path: {archive_name}}] + missing_chunks = {} # chunk_id -> (size, {path: {archive_name}}) missing_chunks_truncated = False # True once the MAX_MISSING_CHUNKS cap was hit missing_refs_truncated = set() # chunk_ids whose MAX_REFS_PER_CHUNK cap was hit + missing_refs_total = 0 # total missing chunk references seen (every file x chunk occurrence, uncapped) def record_missing_chunk(archive_name, path, chunk_id, size): - nonlocal missing_chunks_truncated + nonlocal missing_chunks_truncated, missing_refs_total + missing_refs_total += 1 entry = missing_chunks.get(chunk_id) if entry is None: if len(missing_chunks) >= self.MAX_MISSING_CHUNKS: missing_chunks_truncated = True return - entry = missing_chunks[chunk_id] = [size, {}] - refs = entry[1] + entry = missing_chunks[chunk_id] = (size, {}) + size, refs = entry if path in refs: refs[path].add(archive_name) elif len(refs) < self.MAX_REFS_PER_CHUNK: @@ -2211,7 +2213,10 @@ def report_missing_chunks(): if chunk_id in missing_refs_truncated: logger.error(f" - ... (only the first {self.MAX_REFS_PER_CHUNK} files are listed)") if missing_chunks_truncated: - logger.error(f"... (only the first {self.MAX_MISSING_CHUNKS} missing chunks are listed)") + logger.error( + f"... (only the first {self.MAX_MISSING_CHUNKS} missing chunks are listed; " + f"{missing_refs_total} missing chunk references total)" + ) def robust_iterator(archive): """Iterates through all archive items diff --git a/src/borg/testsuite/archiver/check_cmd_test.py b/src/borg/testsuite/archiver/check_cmd_test.py index 086df5f3f4..a55b81d783 100644 --- a/src/borg/testsuite/archiver/check_cmd_test.py +++ b/src/borg/testsuite/archiver/check_cmd_test.py @@ -212,8 +212,9 @@ def test_missing_file_chunk(archivers, request): assert "The following chunks are missing in the repository:" not in output -def test_missing_file_chunk_report_truncated(archivers, request): - archiver = request.getfixturevalue(archivers) +def test_missing_file_chunk_report_truncated(archiver): + # local-only: this patches ArchiveChecker.MAX_MISSING_CHUNKS in-process, which has no effect + # when borg runs as a separate process (binary_archiver), so it must not be parametrized. check_cmd_setup(archiver) # remove several distinct file chunks, so more missing chunks exist than the (patched) report limit. @@ -243,9 +244,11 @@ def test_missing_file_chunk_refs_truncated(archivers, request): archiver = request.getfixturevalue(archivers) cmd(archiver, "repo-create", RK_ENCRYPTION) - # several distinct files with identical content dedup to the same chunk, so a single missing - # chunk ends up referenced by many files, which exercises the per-chunk reference cap. - for i in range(5): + # many distinct files with identical content dedup to the same chunk, so a single missing chunk + # ends up referenced by more files than MAX_REFS_PER_CHUNK, which exercises the per-chunk cap + # without patching (so it works in binary mode too, where borg runs as a separate process). + cap = ArchiveChecker.MAX_REFS_PER_CHUNK + for i in range(cap + 1): create_regular_file(archiver.input_path, f"samefile{i}", contents=b"same content for dedup") cmd(archiver, "create", "archive1", "input") @@ -259,12 +262,10 @@ def test_missing_file_chunk_refs_truncated(archivers, request): break assert killed_id is not None - # cap references per chunk to 2, so the remaining referencing files are truncated. - with patch.object(ArchiveChecker, "MAX_REFS_PER_CHUNK", 2): - output = cmd(archiver, "check", exit_code=1) + output = cmd(archiver, "check", exit_code=1) assert "The following chunks are missing in the repository:" in output assert bin_to_hex(killed_id) in output - assert "only the first 2 files are listed" in output # the remaining referencing files are truncated + assert f"only the first {cap} files are listed" in output # the remaining referencing files are truncated def test_missing_archive_item_chunk(archivers, request): From 112cbca449444e7da98c4032e3dec4e262537e63 Mon Sep 17 00:00:00 2001 From: Mrityunjay Raj Date: Wed, 5 Aug 2026 01:01:54 +0530 Subject: [PATCH 015/144] check: stream one line per missing chunk id, run report on abort, lower report caps, #9218 --- src/borg/archive.py | 123 +++++++++--------- src/borg/testsuite/archiver/check_cmd_test.py | 9 +- 2 files changed, 71 insertions(+), 61 deletions(-) diff --git a/src/borg/archive.py b/src/borg/archive.py index 8e9fbb3e62..271118ac86 100644 --- a/src/borg/archive.py +++ b/src/borg/archive.py @@ -1851,8 +1851,8 @@ def __next__(self): class ArchiveChecker: # Bound how many missing file chunks rebuild_archives buffers for its end-of-run report, # so checking a badly damaged repo with very many missing chunks can not exhaust memory. - MAX_MISSING_CHUNKS = 10000 # max. distinct missing chunk ids kept for the report - MAX_REFS_PER_CHUNK = 100 # max. referencing files kept per missing chunk + MAX_MISSING_CHUNKS = 1000 # max. distinct missing chunk ids kept for the grouped report + MAX_REFS_PER_CHUNK = 10 # max. referencing files kept per missing chunk def __init__(self): self.error_found = False @@ -2152,6 +2152,8 @@ def record_missing_chunk(archive_name, path, chunk_id, size): missing_chunks_truncated = True return entry = missing_chunks[chunk_id] = (size, {}) + # one line per chunk id (not per file), so an interrupted check still logs what it found. + logger.error(f"Missing chunk detected: {bin_to_hex(chunk_id)}, {format_file_size(size)}.") size, refs = entry if path in refs: refs[path].add(archive_name) @@ -2319,63 +2321,68 @@ def valid_item(obj): pi = ProgressIndicatorPercent( total=num_archives, msg="Checking archives %3.1f%%", step=0.1, msgid="check.rebuild_archives" ) - for i, info in enumerate(archive_infos): - pi.show(i) - archive_id, archive_id_hex = info.id, bin_to_hex(info.id) - try: - formatted = formatter.format_item(info, jsonline=False) - except (Archive.DoesNotExist, Repository.ObjectNotFound, IntegrityErrorBase): - # keys like {comment} need the archive metadata, which is damaged or missing here. - # use the values from the archive directory entry, they are always available. - formatted = f"{info.name} {OutputTimestamp(info.ts)} {archive_id_hex}" - logger.info(f"Analyzing archive {formatted} ({i + 1}/{num_archives})") - if archive_id not in self.chunks: - logger.error(f"Archive metadata block {archive_id_hex} is missing!") - self.error_found = True - if self.repair: - logger.error(f"Deleting broken archive {info.name} {archive_id_hex}.") - self.manifest.archives.delete_by_id(archive_id) - else: - logger.error(f"Would delete broken archive {info.name} {archive_id_hex}.") - continue - cdata = self.repository.get(archive_id) - try: - _, data = self.repo_objs.parse(archive_id, cdata, ro_type=ROBJ_ARCHIVE_META) - except IntegrityErrorBase as integrity_error: - logger.error(f"Archive metadata block {archive_id_hex} is corrupted: {integrity_error}") - self.error_found = True + # report the missing chunks collected so far even if the loop is interrupted (Ctrl-C) or aborts + # with an exception (e.g. the "Unknown archive metadata version" raise below), so a check of a + # badly damaged repo does not throw away everything it already found. + try: + for i, info in enumerate(archive_infos): + pi.show(i) + archive_id, archive_id_hex = info.id, bin_to_hex(info.id) + try: + formatted = formatter.format_item(info, jsonline=False) + except (Archive.DoesNotExist, Repository.ObjectNotFound, IntegrityErrorBase): + # keys like {comment} need the archive metadata, which is damaged or missing here. + # use the values from the archive directory entry, they are always available. + formatted = f"{info.name} {OutputTimestamp(info.ts)} {archive_id_hex}" + logger.info(f"Analyzing archive {formatted} ({i + 1}/{num_archives})") + if archive_id not in self.chunks: + logger.error(f"Archive metadata block {archive_id_hex} is missing!") + self.error_found = True + if self.repair: + logger.error(f"Deleting broken archive {info.name} {archive_id_hex}.") + self.manifest.archives.delete_by_id(archive_id) + else: + logger.error(f"Would delete broken archive {info.name} {archive_id_hex}.") + continue + cdata = self.repository.get(archive_id) + try: + _, data = self.repo_objs.parse(archive_id, cdata, ro_type=ROBJ_ARCHIVE_META) + except IntegrityErrorBase as integrity_error: + logger.error(f"Archive metadata block {archive_id_hex} is corrupted: {integrity_error}") + self.error_found = True + if self.repair: + logger.error(f"Deleting broken archive {info.name} {archive_id_hex}.") + self.manifest.archives.delete_by_id(archive_id) + else: + logger.error(f"Would delete broken archive {info.name} {archive_id_hex}.") + continue + archive = self.key.unpack_archive(data) + archive = ArchiveItem(internal_dict=archive) + if archive.version != 2: + raise Exception("Unknown archive metadata version") + items_buffer = ChunkBuffer(self.key) + items_buffer.write_chunk = add_callback + for item in robust_iterator(archive): + if "chunks" in item: + verify_file_chunks(info.name, item) + items_buffer.add(item) + items_buffer.flush(flush=True) if self.repair: - logger.error(f"Deleting broken archive {info.name} {archive_id_hex}.") - self.manifest.archives.delete_by_id(archive_id) - else: - logger.error(f"Would delete broken archive {info.name} {archive_id_hex}.") - continue - archive = self.key.unpack_archive(data) - archive = ArchiveItem(internal_dict=archive) - if archive.version != 2: - raise Exception("Unknown archive metadata version") - items_buffer = ChunkBuffer(self.key) - items_buffer.write_chunk = add_callback - for item in robust_iterator(archive): - if "chunks" in item: - verify_file_chunks(info.name, item) - items_buffer.add(item) - items_buffer.flush(flush=True) - if self.repair: - archive.item_ptrs = archive_put_items( - items_buffer.chunks, repo_objs=self.repo_objs, add_reference=add_reference - ) - data = self.key.pack_metadata(archive.as_dict()) - new_archive_id = self.key.id_hash(data) - logger.debug(f"archive id old: {bin_to_hex(archive_id)}") - logger.debug(f"archive id new: {bin_to_hex(new_archive_id)}") - cdata = self.repo_objs.format(new_archive_id, {}, data, ro_type=ROBJ_ARCHIVE_META) - add_reference(new_archive_id, len(data), cdata) - self.manifest.archives.create(info.name, new_archive_id, info.ts) - if archive_id != new_archive_id: - self.manifest.archives.delete_by_id(archive_id) - pi.finish() - report_missing_chunks() + archive.item_ptrs = archive_put_items( + items_buffer.chunks, repo_objs=self.repo_objs, add_reference=add_reference + ) + data = self.key.pack_metadata(archive.as_dict()) + new_archive_id = self.key.id_hash(data) + logger.debug(f"archive id old: {bin_to_hex(archive_id)}") + logger.debug(f"archive id new: {bin_to_hex(new_archive_id)}") + cdata = self.repo_objs.format(new_archive_id, {}, data, ro_type=ROBJ_ARCHIVE_META) + add_reference(new_archive_id, len(data), cdata) + self.manifest.archives.create(info.name, new_archive_id, info.ts) + if archive_id != new_archive_id: + self.manifest.archives.delete_by_id(archive_id) + finally: + pi.finish() + report_missing_chunks() def finish(self): if self.repair: diff --git a/src/borg/testsuite/archiver/check_cmd_test.py b/src/borg/testsuite/archiver/check_cmd_test.py index a55b81d783..64c3fc990f 100644 --- a/src/borg/testsuite/archiver/check_cmd_test.py +++ b/src/borg/testsuite/archiver/check_cmd_test.py @@ -180,9 +180,12 @@ def test_missing_file_chunk(archivers, request): output = cmd(archiver, "check", exit_code=1) assert "The following chunks are missing in the repository:" in output - # archive1 and archive2 share src_file, so the missing chunk appears once, with both archives - # listed on its single reference line. - assert output.count(bin_to_hex(killed_chunk.id)) == 1 + # archive1 and archive2 share src_file, so the missing chunk is grouped once, with both archives + # listed on its single reference line (the id also appears once in the streamed "Missing chunk + # detected" line emitted while the archives are analyzed). + killed_hex = bin_to_hex(killed_chunk.id) + chunk_header_lines = [ln for ln in output.splitlines() if ln.startswith("- Chunk ") and killed_hex in ln] + assert len(chunk_header_lines) == 1 ref_lines = [line for line in output.splitlines() if src_file in line] assert len(ref_lines) == 1 assert "archive1" in ref_lines[0] and "archive2" in ref_lines[0] From 23e2a210c3a06efffd8b1b8776b0e3d6c534bc56 Mon Sep 17 00:00:00 2001 From: Mrityunjay Raj Date: Wed, 5 Aug 2026 01:31:40 +0530 Subject: [PATCH 016/144] check: re-verify records past max-age, guard on resolved window, tighten comments --- src/borg/archiver/check_cmd.py | 34 ++++++++++--------- src/borg/repository.py | 24 ++++++------- src/borg/testsuite/archiver/check_cmd_test.py | 8 +++++ src/borg/testsuite/repository_test.py | 27 +++++++-------- 4 files changed, 49 insertions(+), 44 deletions(-) diff --git a/src/borg/archiver/check_cmd.py b/src/borg/archiver/check_cmd.py index 825420e506..6e37e61b16 100644 --- a/src/borg/archiver/check_cmd.py +++ b/src/borg/archiver/check_cmd.py @@ -43,21 +43,30 @@ def do_check(self, args, repository): ) if args.repo_only and args.find_lost_archives: raise CommandError("--repository-only contradicts the --find-lost-archives option.") + # resolve the marker (e.g. 4w, 12m) to seconds; calendar units (m, y) count from now, as for + # --older/--newer. the validator returns a string (truthy even for a zero span like 0d), so + # the --max-duration guard below tests max_age, the resolved value, not args.max_age. + if args.max_age is not None: + now = archive_ts_now() + max_age = int((now - calculate_relative_offset(args.max_age, now, earlier=True)).total_seconds()) + else: + max_age = 0 if args.repair and args.max_duration: raise CommandError("--repair does not allow --max-duration argument.") - if args.repair and args.max_age: - # reusing recorded results during repair depends on repository repair (refs #8572), which - # does not exist yet, so repair verifies every pack. + if args.repair and args.max_age is not None: + # repair verifies every pack; reusing recorded results during repair needs repository + # repair (refs #8572). raise CommandError("--repair does not allow the --max-age option.") - if args.archives_only and args.max_age: - # --max-age only affects the repository check, which --archives-only skips. + if args.archives_only and args.max_age is not None: + # --max-age only affects the repository check; --archives-only skips it. raise CommandError("--archives-only does not allow the --max-age option.") - if args.max_duration and not args.max_age: - # partial checks progress by skipping packs whose record is younger than max_age. + if args.max_duration and not max_age: + # a partial check advances by skipping packs younger than max_age, so it needs a nonzero + # window (--max-age=0d resolves to 0). raise CommandError("--max-duration requires the --max-age option, e.g. --max-age=4w.") if args.max_duration and not args.repo_only: - # --max-duration time-boxes the repository pack check only; the archives check has no - # max_duration support and builds its own chunk index, so it cannot be split this way. + # --max-duration limits only the repository check; the archives check has no max_duration + # support. raise CommandError("--repository-only is required for --max-duration support.") if not args.repo_only: # if we need the key later for the archives check, ask NOW for the passphrase! #1931 @@ -74,13 +83,6 @@ def do_check(self, args, repository): # the repository check has finished, which can take hours. ArchiveFormatter.validate_format(format) if not args.archives_only: - if args.max_age: - # resolve the relative marker (e.g. 4w, 12m) to a concrete age in seconds; calendar - # units (m, y) are measured against now, like --older / --newer. - now = archive_ts_now() - max_age = int((now - calculate_relative_offset(args.max_age, now, earlier=True)).total_seconds()) - else: - max_age = 0 if not repository.check(repair=args.repair, max_duration=args.max_duration, max_age=max_age): set_ec(EXIT_WARNING) if not args.repo_only and not archive_checker.check( diff --git a/src/borg/repository.py b/src/borg/repository.py index b3f97b584e..4d38a2c822 100644 --- a/src/borg/repository.py +++ b/src/borg/repository.py @@ -836,13 +836,13 @@ def check(self, repair=False, max_duration=0, max_age=0): corrupt packs and dropping those packs is left to repair, refs #8572. The ids of the packs found corrupt are kept in cache/checked-packs for repair, refs #9696. - Any pack on record as corrupt fails the check, including on a partial run that stops before - re-reaching it. The record clears when the pack verifies intact again, when compact removes - the pack, or when repair salvages and drops it (refs #8572). + A pack recorded corrupt fails the check, also on a partial run that stops before re-reaching + it. The record clears at the check that finds the pack intact again or gone (removed by + compact, or salvaged and dropped by repair; refs #8572); prune() does this from packs/. max_age (seconds, 0 = verify every pack): skip packs whose intact record is younger than - max_age, tolerating up to MAX_CLOCK_SKEW of clock skew at both ends of the window. Check - results are always recorded and kept. + max_age, accepting a future timestamp up to MAX_CLOCK_SKEW (clock skew). Results are recorded + regardless of max_age. """ def verify(namespace, name): @@ -906,14 +906,12 @@ def store_list(namespace): pack_pi.show(increase=1) # advance for skipped packs too, so the bar tracks packs/, not work done pack_id = hex_to_bin(info.name) entry = tracker.get(pack_id) - # skip the pack if its recorded intact result is younger than max_age. the timestamp - # is written by whichever client ran the earlier check, so it may be off by up to - # MAX_CLOCK_SKEW against our clock in either direction; tolerate that much skew at both - # ends of the window, but never more than the window itself. + # skip a pack recorded intact within the last max_age seconds. the timestamp is set + # by the client that ran the earlier check; accept a future one (negative age) up to + # MAX_CLOCK_SKEW, and re-verify anything at or past max_age. if entry is not None and entry.result and max_age: age = time.time() - entry.timestamp - skew = min(MAX_CLOCK_SKEW, max_age) - if -skew <= age <= max_age + skew: + if -min(MAX_CLOCK_SKEW, max_age) <= age < max_age: pack_skipped += 1 continue pack_files += 1 @@ -947,8 +945,8 @@ def store_list(namespace): if pack_skipped: summary += f" Reused {pack_skipped} recent pack check result(s)." logger.info(summary) - # every pack recorded corrupt, not only those verified this run; empty if the index is - # corrupt, since then no packs were scanned. + # corrupt_ids() is every pack recorded corrupt, including from earlier runs. with a corrupt + # index the packs were not scanned, so report nothing. corrupt_ids = tracker.corrupt_ids() if index_errors == 0 else [] if corrupt_ids: # one id per line (the list can be long). diff --git a/src/borg/testsuite/archiver/check_cmd_test.py b/src/borg/testsuite/archiver/check_cmd_test.py index b3380202dd..b4bad53dc4 100644 --- a/src/borg/testsuite/archiver/check_cmd_test.py +++ b/src/borg/testsuite/archiver/check_cmd_test.py @@ -79,17 +79,25 @@ def test_check_max_age(archivers, request): check_cmd_setup(archiver) # --repair and --archives-only do not allow --max-age, and --max-duration requires --max-age. + # a zero span like 0d resolves to max_age=0 (no reuse), so it must not satisfy the guards either. if archiver.FORK_DEFAULT: cmd(archiver, "check", "--repair", "--max-age=1d", exit_code=CommandError().exit_code) + cmd(archiver, "check", "--repair", "--max-age=0d", exit_code=CommandError().exit_code) cmd(archiver, "check", "--archives-only", "--max-age=1d", exit_code=CommandError().exit_code) cmd(archiver, "check", "--repository-only", "--max-duration=3600", exit_code=CommandError().exit_code) + ec = CommandError().exit_code + cmd(archiver, "check", "--repository-only", "--max-duration=3600", "--max-age=0d", exit_code=ec) else: with pytest.raises(CommandError): cmd(archiver, "check", "--repair", "--max-age=1d") + with pytest.raises(CommandError): + cmd(archiver, "check", "--repair", "--max-age=0d") with pytest.raises(CommandError): cmd(archiver, "check", "--archives-only", "--max-age=1d") with pytest.raises(CommandError): cmd(archiver, "check", "--repository-only", "--max-duration=3600") + with pytest.raises(CommandError): + cmd(archiver, "check", "--repository-only", "--max-duration=3600", "--max-age=0d") # a check records its results, a later one with --max-age reuses them. output = cmd(archiver, "check", "-v", "--repository-only", exit_code=0) diff --git a/src/borg/testsuite/repository_test.py b/src/borg/testsuite/repository_test.py index bdac69d4b9..80983e1dca 100644 --- a/src/borg/testsuite/repository_test.py +++ b/src/borg/testsuite/repository_test.py @@ -1058,7 +1058,7 @@ def test_check_partial_rechecks_pack_sorting_before_checked_one(tmp_path): def test_check_partial_rechecks_pack_recorded_corrupt(tmp_path): - # a pack recorded corrupt earlier is re-verified, so the corruption keeps being reported. + # a pack recorded corrupt earlier is re-verified and reported again. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: corrupt_id = H(1) # stored content does not hash to this name repository.store_store("packs/" + bin_to_hex(corrupt_id), b"CORRUPT-does-not-match-name") @@ -1272,13 +1272,13 @@ def test_check_max_age_skips_fresh_ok(tmp_path, monkeypatch): def test_check_max_age_reverifies_stale_ok(tmp_path, monkeypatch): - # with max_age, a pack whose intact record is older than max_age plus the skew tolerance is re-verified. + # with max_age, a pack whose intact record is older than max_age is re-verified. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: intact_id, pack_key = _store_intact_pack(repository) max_age = 50 tracker = PackTracker.new(repository.store) - old_ts = int(time.time()) - (max_age + MAX_CLOCK_SKEW + 100) # clearly beyond max_age + skew + old_ts = int(time.time()) - (max_age + 100) # clearly beyond max_age tracker.table[intact_id] = PackTracker.Entry(timestamp=old_ts, result=1) tracker.save() @@ -1291,27 +1291,25 @@ def test_check_max_age_reverifies_stale_ok(tmp_path, monkeypatch): assert after.table[intact_id].timestamp > old_ts # record refreshed -def test_check_max_age_skips_stale_within_skew(tmp_path, monkeypatch): - # a record older than max_age but within MAX_CLOCK_SKEW of it still counts as recent (writer clock - # behind ours), so it is not re-verified. +def test_check_max_age_reverifies_stale_within_skew(tmp_path, monkeypatch): + # a record past max_age gets no skew tolerance: it is re-verified even within MAX_CLOCK_SKEW of it. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: intact_id, pack_key = _store_intact_pack(repository) - max_age = MAX_CLOCK_SKEW * 2 # wide window, so the skew tolerance is MAX_CLOCK_SKEW + max_age = MAX_CLOCK_SKEW * 2 # window wider than MAX_CLOCK_SKEW tracker = PackTracker.new(repository.store) - past_ts = int(time.time()) - (max_age + MAX_CLOCK_SKEW // 2) # past the window, within skew + past_ts = int(time.time()) - (max_age + MAX_CLOCK_SKEW // 2) # just past the window tracker.table[intact_id] = PackTracker.Entry(timestamp=past_ts, result=1) tracker.save() hashed_keys = _spy_hash(repository, monkeypatch) assert repository.check(repair=False, max_age=max_age) is True - assert pack_key not in hashed_keys # within MAX_CLOCK_SKEW past the window, still recent + assert pack_key in hashed_keys # past max_age, re-verified def test_check_max_age_skips_near_future_ok(tmp_path, monkeypatch): - # with a window wider than MAX_CLOCK_SKEW, a record timestamped up to MAX_CLOCK_SKEW into the - # future (writer clock ahead of ours) still counts as recent and is not re-verified. + # a future timestamp (writer clock ahead of ours) up to MAX_CLOCK_SKEW is tolerated and skipped. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: intact_id, pack_key = _store_intact_pack(repository) @@ -1323,12 +1321,11 @@ def test_check_max_age_skips_near_future_ok(tmp_path, monkeypatch): hashed_keys = _spy_hash(repository, monkeypatch) assert repository.check(repair=False, max_age=MAX_CLOCK_SKEW * 2) is True - assert pack_key not in hashed_keys # within MAX_CLOCK_SKEW ahead, still counts as recent + assert pack_key not in hashed_keys # within MAX_CLOCK_SKEW ahead, skipped def test_check_max_age_reverifies_far_future_ok(tmp_path, monkeypatch): - # with a window wider than MAX_CLOCK_SKEW, a record more than MAX_CLOCK_SKEW into the future is - # not plausible clock skew and is re-verified. + # a future timestamp more than MAX_CLOCK_SKEW ahead exceeds the skew tolerance and is re-verified. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: intact_id, pack_key = _store_intact_pack(repository) @@ -1340,7 +1337,7 @@ def test_check_max_age_reverifies_far_future_ok(tmp_path, monkeypatch): hashed_keys = _spy_hash(repository, monkeypatch) assert repository.check(repair=False, max_age=MAX_CLOCK_SKEW * 2) is True - assert pack_key in hashed_keys # too far ahead to be clock skew, re-verified + assert pack_key in hashed_keys # more than MAX_CLOCK_SKEW ahead, re-verified def test_check_max_age_reverifies_future_beyond_small_window(tmp_path, monkeypatch): From 7171b688b8b38ee1090a26c2d2da2e7a2f86435a Mon Sep 17 00:00:00 2001 From: Mrityunjay Raj Date: Wed, 5 Aug 2026 01:46:36 +0530 Subject: [PATCH 017/144] check: verify least-recently-checked packs first on partial runs, document --max-age markers --- src/borg/archiver/check_cmd.py | 11 +++++--- src/borg/repository.py | 9 +++++++ src/borg/testsuite/repository_test.py | 39 +++++++++++++++++++++++++++ 3 files changed, 55 insertions(+), 4 deletions(-) diff --git a/src/borg/archiver/check_cmd.py b/src/borg/archiver/check_cmd.py index 6e37e61b16..4118825170 100644 --- a/src/borg/archiver/check_cmd.py +++ b/src/borg/archiver/check_cmd.py @@ -140,10 +140,13 @@ def build_parser_check(self, subparsers, common_parser, mid_common_parser): The ``--max-age`` option makes the check reuse the results of previous repository checks: packs whose intact result is younger than the given timespan (e.g. ``--max-age=4w`` or ``--max-age=12m``) are skipped, spreading - the verification cost over repeated checks. Check results are recorded in any case; - ``--max-age`` only controls their reuse. Packs recorded corrupt are always - re-verified. ``--max-age`` affects only the repository check and cannot be - combined with ``--archives-only`` or ``--repair``. + the verification cost over repeated checks. The timespan uses the same markers + as ``--older``/``--newer``: ``d``, ``w``, ``H``, ``M``, ``S`` are exact spans, + while ``m`` and ``y`` are calendar units counted from now (so ``12m`` equals + ``1y``). Check results are recorded in any case; ``--max-age`` only controls + their reuse. Packs recorded corrupt are always re-verified. ``--max-age`` + affects only the repository check and cannot be combined with + ``--archives-only`` or ``--repair``. The ``--max-duration`` option can be used to split a long-running repository check into multiple partial checks. After the given number of seconds, the diff --git a/src/borg/repository.py b/src/borg/repository.py index 4d38a2c822..6c3683b640 100644 --- a/src/borg/repository.py +++ b/src/borg/repository.py @@ -900,6 +900,15 @@ def store_list(namespace): if index_errors == 0: # packs are the bulk of the work and the part --max-duration spreads over several checks. pack_infos = store_list("packs") + if partial: + # a partial check stops after max_duration; verify the least-recently-checked packs + # first so repeated runs cover every pack. sort by recorded check time, unrecorded + # (time 0) first. + def recorded_ts(info): + entry = tracker.get(hex_to_bin(info.name)) + return entry.timestamp if entry is not None else 0 + + pack_infos.sort(key=recorded_ts) pack_pi = ProgressIndicatorPercent(total=len(pack_infos), msg="Checking packs %3.0f%%", msgid="check.packs") for info in pack_infos: self._lock_refresh() diff --git a/src/borg/testsuite/repository_test.py b/src/borg/testsuite/repository_test.py index 80983e1dca..8ca323ebab 100644 --- a/src/borg/testsuite/repository_test.py +++ b/src/borg/testsuite/repository_test.py @@ -1414,6 +1414,45 @@ def test_check_max_age_partial_progress(tmp_path, monkeypatch): assert pack_a_id in after.table and pack_b_id in after.table +def test_check_partial_verifies_least_recently_checked_first(tmp_path, monkeypatch): + # a partial check verifies the least-recently-checked packs first. with a budget for one pack, the + # pack with the oldest recorded result is verified, though its id sorts last. + max_age = 3600 + recent_id = b"\x00" * 32 # sorts first by id + older_id = b"\xff" * 32 # sorts last by id + with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: + for pid in (recent_id, older_id): + repository.store_store("packs/" + bin_to_hex(pid), b"content-" + pid[:4]) + + now = int(time.time()) + tracker = PackTracker.new(repository.store) + # both records are past max_age, so both need re-verification; the more recently recorded pack + # has the id that sorts first. + tracker.table[recent_id] = PackTracker.Entry(timestamp=now - (max_age + 100), result=1) + tracker.table[older_id] = PackTracker.Entry(timestamp=now - (max_age + 100000), result=1) + tracker.save() + + # freeze the clock and jump it past max_duration after the first hash, so exactly one pack is + # verified this run. + clock = {"t": 0.0} + monkeypatch.setattr(time, "monotonic", lambda: clock["t"]) + hashed_keys = [] + orig_hash = repository.store.hash + + def hash_and_advance(key): + hashed_keys.append(key) + result = orig_hash(key) + clock["t"] = 10**9 + return result + + monkeypatch.setattr(repository.store, "hash", hash_and_advance) + + repository.check(repair=False, max_duration=1, max_age=max_age) + + pack_hashes = [key for key in hashed_keys if key.startswith("packs/")] + assert pack_hashes == ["packs/" + bin_to_hex(older_id)] # the oldest recorded pack + + def test_check_max_age_reuses_records_of_plain_check(tmp_path, monkeypatch): # a check without max_age records its results, a later check with max_age reuses them. with Repository(str(tmp_path / "repo"), exclusive=True, create=True) as repository: From 3a8431ccd80406bd8631ef3ebfe64c60137cc972 Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Sat, 1 Aug 2026 03:05:12 +0200 Subject: [PATCH 018/144] completion: generate fish completions, remove hand-written ones, fixes #9989 fish completions are now generated by shtab, like the bash, zsh and tcsh ones, extended with a fish preamble that provides the same dynamic completions the bash/zsh scripts have (archive names, aid: archive IDs, tags, sort keys, files cache modes, compression specs, chunker params, relative time markers, timestamps, file sizes, help topics). This needs shtab >= 1.9.3, which brings the fixed fish backend (see tqdm/shtab#227, #228, #229, #230 and #231), so the pin was bumped. Arguments taking a filesystem path are now typed, so that all shells can complete them: such a path was either untyped or just a str, so the shells relied on their default completion (bash/zsh/tcsh) or offered nothing at all (fish). FilesystemPathSpec is used for arguments taking a file (--exclude-from, --patterns-from, TARFILE, the borg debug file arguments, borg key export/import PATH), the new FilesystemDirSpec for arguments taking a directory (MOUNTPOINT, borg benchmark crud PATH). Besides the completions, they now also reject empty paths and slashify them on Windows, like the other filesystem path arguments do. Note that borg key export/import PATH completed directories before (it takes a file) and borg benchmark crud PATH completed files (it takes a directory) - both are fixed by this. shtab's fish FILE completion needs a little help to also complete absolute paths, see tqdm/shtab#245 (fix: tqdm/shtab#246). The completion script is now always generated for the "borg" command, no matter how borg was invoked while generating it - completions generated by "python -m borg completion ..." were useless before, as the shells would complete a "__main__.py" command. The hand-written, unmaintained fish completions in scripts/shell_completions/ are removed, as are their tests. Co-Authored-By: Claude Fable 5 --- pyproject.toml | 2 +- scripts/shell_completions/fish/borg.fish | 545 ------------------ src/borg/archiver/_common.py | 3 + src/borg/archiver/benchmark_cmd.py | 5 +- src/borg/archiver/completion_cmd.py | 247 ++++++-- src/borg/archiver/debug_cmd.py | 44 +- src/borg/archiver/key_cmds.py | 12 +- src/borg/archiver/mount_cmds.py | 8 +- src/borg/archiver/tar_cmds.py | 12 +- src/borg/helpers/__init__.py | 1 + src/borg/helpers/parseformat.py | 9 + .../testsuite/archiver/completion_cmd_test.py | 205 +++++++ src/borg/testsuite/shell_completions_test.py | 22 - 13 files changed, 483 insertions(+), 632 deletions(-) delete mode 100644 scripts/shell_completions/fish/borg.fish delete mode 100644 src/borg/testsuite/shell_completions_test.py diff --git a/pyproject.toml b/pyproject.toml index 0b6e90d5db..74322d99e0 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -37,7 +37,7 @@ dependencies = [ "platformdirs >=3.0.0, <5.0.0; sys_platform == 'darwin'", # for macOS: breaking changes in 3.0.0. "platformdirs >=2.6.0, <5.0.0; sys_platform != 'darwin'", # for others: 2.6+ works consistently. "argon2-cffi", - "shtab>=1.8.0,!=1.8.2,!=1.9.0,!=1.9.1", # these generate broken zsh completions, see tqdm/shtab#224 + "shtab>=1.9.3", # older versions generate broken zsh (tqdm/shtab#224) or fish (tqdm/shtab#227) completions "backports-zstd; python_version < '3.14'", # for python < 3.14. "jsonargparse>=4.47.0", "PyYAML>=6.0.2", # we need to register our types with yaml, jsonargparse uses yaml for config files diff --git a/scripts/shell_completions/fish/borg.fish b/scripts/shell_completions/fish/borg.fish deleted file mode 100644 index a756e08940..0000000000 --- a/scripts/shell_completions/fish/borg.fish +++ /dev/null @@ -1,545 +0,0 @@ -# Completions for borg -# https://www.borgbackup.org/ -# Note: -# Listing archives works on password-protected repositories only if $BORG_PASSPHRASE is set. -# Install: -# Copy this file to /usr/share/fish/vendor_completions.d/ - -# Commands - -complete -c borg -f -n __fish_is_first_token -a 'analyze' -d 'Analyze archives to find "hot spots"' -complete -c borg -f -n __fish_is_first_token -a 'create' -d 'Create a new archive' -complete -c borg -f -n __fish_is_first_token -a 'extract' -d 'Extract archive contents' -complete -c borg -f -n __fish_is_first_token -a 'check' -d 'Check repository consistency' -complete -c borg -f -n __fish_is_first_token -a 'rename' -d 'Rename an existing archive' -complete -c borg -f -n __fish_is_first_token -a 'list' -d 'List archive or repository contents' -complete -c borg -f -n __fish_is_first_token -a 'diff' -d 'Find differences between archives' -complete -c borg -f -n __fish_is_first_token -a 'delete' -d 'Delete an archive' -complete -c borg -f -n __fish_is_first_token -a 'prune' -d 'Prune repository archives' -complete -c borg -f -n __fish_is_first_token -a 'compact' -d 'Free repository space' -complete -c borg -f -n __fish_is_first_token -a 'info' -d 'Show archive details' -complete -c borg -f -n __fish_is_first_token -a 'mount' -d 'Mount archive or a repository' -complete -c borg -f -n __fish_is_first_token -a 'umount' -d 'Unmount the mounted archive' -complete -c borg -f -n __fish_is_first_token -a 'repo-compress' -d 'Repository (re-)compression' -complete -c borg -f -n __fish_is_first_token -a 'repo-create' -d 'Create a new, empty repository' -complete -c borg -f -n __fish_is_first_token -a 'repo-delete' -d 'Delete a repository' -complete -c borg -f -n __fish_is_first_token -a 'repo-info' -d 'Show repository information' -complete -c borg -f -n __fish_is_first_token -a 'repo-list' -d 'List repository contents' -complete -c borg -f -n __fish_is_first_token -a 'repo-space' -d 'Manage reserved space in a repository' -complete -c borg -f -n __fish_is_first_token -a 'tag' -d 'Tag archives' -complete -c borg -f -n __fish_is_first_token -a 'transfer' -d 'Transfer of archives from another repository' -complete -c borg -f -n __fish_is_first_token -a 'undelete' -d 'Undelete archives' -complete -c borg -f -n __fish_is_first_token -a 'version' -d 'Display Borg client version / Borg server version' - -function __fish_borg_seen_key - if __fish_seen_subcommand_from key - and not __fish_seen_subcommand_from import export change-passphrase - return 0 - end - return 1 -end -complete -c borg -f -n __fish_is_first_token -a 'key' -d 'Manage a repository key' -complete -c borg -f -n __fish_borg_seen_key -a 'import' -d 'Import a repository key' -complete -c borg -f -n __fish_borg_seen_key -a 'export' -d 'Export a repository key' -complete -c borg -f -n __fish_borg_seen_key -a 'change-passphrase' -d 'Change key file passphrase' - -complete -c borg -f -n __fish_is_first_token -a 'serve' -d 'Start in server mode' -complete -c borg -f -n __fish_is_first_token -a 'recreate' -d 'Recreate contents of existing archives' -complete -c borg -f -n __fish_is_first_token -a 'export-tar' -d 'Create tarball from an archive' -complete -c borg -f -n __fish_is_first_token -a 'with-lock' -d 'Run a command while the repository lock is held' -complete -c borg -f -n __fish_is_first_token -a 'break-lock' -d 'Break the repository lock' - -function __fish_borg_seen_benchmark - if __fish_seen_subcommand_from benchmark - and not __fish_seen_subcommand_from crud - return 0 - end - return 1 -end -complete -c borg -f -n __fish_is_first_token -a 'benchmark' -d 'Benchmark borg operations' -complete -c borg -f -n __fish_borg_seen_benchmark -a 'crud' -d 'Benchmark borg CRUD operations' - -function __fish_borg_seen_help - if __fish_seen_subcommand_from help - and not __fish_seen_subcommand_from patterns placeholders compression - return 0 - end - return 1 -end -complete -c borg -f -n __fish_is_first_token -a 'help' -d 'Miscellaneous Help' -complete -c borg -f -n __fish_borg_seen_help -a 'patterns' -d 'Help for patterns' -complete -c borg -f -n __fish_borg_seen_help -a 'placeholders' -d 'Help for placeholders' -complete -c borg -f -n __fish_borg_seen_help -a 'compression' -d 'Help for compression' - -# Common options -complete -c borg -f -s h -l 'help' -d 'Show help information' -complete -c borg -f -l 'version' -d 'Show version information' -complete -c borg -f -l 'critical' -d 'Log level CRITICAL' -complete -c borg -f -l 'error' -d 'Log level ERROR' -complete -c borg -f -l 'warning' -d 'Log level WARNING (default)' -complete -c borg -f -l 'info' -d 'Log level INFO' -complete -c borg -f -s v -l 'verbose' -d 'Log level INFO' -complete -c borg -f -l 'debug' -d 'Log level DEBUG' -complete -c borg -f -l 'debug-topic' -d 'Enable TOPIC debugging' -complete -c borg -f -s p -l 'progress' -d 'Show progress information' -complete -c borg -f -l 'log-json' -d 'Output one JSON object per log line' -complete -c borg -f -l 'lock-wait' -d 'Wait for lock max N seconds [1]' -complete -c borg -f -l 'show-version' -d 'Log version information' -complete -c borg -f -l 'show-rc' -d 'Log the return code' -complete -c borg -f -l 'umask' -d 'Set umask to M [0077]' - -# borg analyze options -complete -c borg -f -s a -l 'match-archives' -d 'Only archive names matching PATTERN' -n "__fish_seen_subcommand_from analyze" -set -l sort_keys "timestamp archive name id tags host user" -complete -c borg -f -l 'sort-by' -d 'Sorting KEYS [timestamp]' -a "$sort_keys" -n "__fish_seen_subcommand_from analyze" -complete -c borg -f -l 'first' -d 'Only first N archives' -n "__fish_seen_subcommand_from analyze" -complete -c borg -f -l 'last' -d 'Only last N archives' -n "__fish_seen_subcommand_from analyze" -complete -c borg -f -l 'oldest' -d 'Consider archives within TIMESPAN from oldest' -n "__fish_seen_subcommand_from analyze" -complete -c borg -f -l 'newest' -d 'Consider archives within TIMESPAN from newest' -n "__fish_seen_subcommand_from analyze" -complete -c borg -f -l 'older' -d 'Consider archives older than TIMESPAN' -n "__fish_seen_subcommand_from analyze" -complete -c borg -f -l 'newer' -d 'Consider archives newer than TIMESPAN' -n "__fish_seen_subcommand_from analyze" - -# borg repo-compress options -# Define compression methods once at the top -set -l compression_methods "none auto lz4 zstd,1 zstd,2 zstd,3 zstd,4 zstd,5 zstd,6 zstd,7 zstd,8 zstd,9 zstd,10 zstd,11 zstd,12 zstd,13 zstd,14 zstd,15 zstd,16 zstd,17 zstd,18 zstd,19 zstd,20 zstd,21 zstd,22 zlib,1 zlib,2 zlib,3 zlib,4 zlib,5 zlib,6 zlib,7 zlib,8 zlib,9 lzma,0 lzma,1 lzma,2 lzma,3 lzma,4 lzma,5 lzma,6 lzma,7 lzma,8 lzma,9" -complete -c borg -f -s C -l 'compression' -d 'Select compression ALGORITHM,LEVEL [lz4]' -a "$compression_methods" -n "__fish_seen_subcommand_from repo-compress" -complete -c borg -f -s s -l 'stats' -d 'Print statistics' -n "__fish_seen_subcommand_from repo-compress" - -# borg create options -complete -c borg -f -s n -l 'dry-run' -d 'Do not create a backup archive' -n "__fish_seen_subcommand_from create" -complete -c borg -f -s s -l 'stats' -d 'Print verbose statistics' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'list' -d 'Print verbose list of items' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'filter' -d 'Only items with given STATUSCHARS' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'json' -d 'Print verbose stats as json' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'stdin-name' -d 'Use NAME in archive for stdin data' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'content-from-command' -d 'Interpret PATH as command and store its stdout' -n "__fish_seen_subcommand_from create" -# Exclusion options -complete -c borg -s e -l 'exclude' -d 'Exclude paths matching PATTERN' -n "__fish_seen_subcommand_from create" -complete -c borg -l 'exclude-from' -d 'Read exclude patterns from EXCLUDEFILE' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'pattern' -d 'Include/exclude paths matching PATTERN' -n "__fish_seen_subcommand_from create" -complete -c borg -l 'patterns-from' -d 'Include/exclude paths from PATTERNFILE' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'exclude-caches' -d 'Exclude directories tagged as cache' -n "__fish_seen_subcommand_from create" -complete -c borg -l 'exclude-if-present' -d 'Exclude directories that contain FILENAME' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'keep-exclude-tags' -d 'Keep tag files of excluded directories' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'exclude-nodump' -d 'Exclude files flagged NODUMP' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'exclude-dataless' -d 'Exclude files flagged DATALESS (macOS)' -n "__fish_seen_subcommand_from create" -# Filesystem options -complete -c borg -f -s x -l 'one-file-system' -d 'Stay in the same file system' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'numeric-ids' -d 'Only store numeric user:group identifiers' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'noatime' -d 'Do not store atime' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'noctime' -d 'Do not store ctime' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'nobirthtime' -d 'Do not store creation date' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'nobsdflags' -d 'Do not store bsdflags' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'noacls' -d 'Do not read and store ACLs into archive' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'noxattrs' -d 'Do not read and store xattrs into archive' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'noflags' -d 'Do not store flags' -n "__fish_seen_subcommand_from create" -set -l files_cache_mode "ctime,size,inode mtime,size,inode ctime,size mtime,size rechunk,ctime rechunk,mtime size disabled" -complete -c borg -f -l 'files-cache' -d 'Operate files cache in MODE' -a "$files_cache_mode" -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'read-special' -d 'Open device files like regular files' -n "__fish_seen_subcommand_from create" -# Archive options -complete -c borg -f -l 'comment' -d 'Add COMMENT to the archive' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'timestamp' -d 'Set creation TIME (yyyy-mm-ddThh:mm:ss)' -n "__fish_seen_subcommand_from create" -complete -c borg -l 'timestamp' -d 'Set creation time by reference FILE' -n "__fish_seen_subcommand_from create" -complete -c borg -f -l 'chunker-params' -d 'Chunker PARAMETERS [19,23,21,4095]' -n "__fish_seen_subcommand_from create" -complete -c borg -f -s C -l 'compression' -d 'Select compression ALGORITHM,LEVEL [lz4]' -a "$compression_methods" -n "__fish_seen_subcommand_from create" - -# borg extract options -complete -c borg -f -l 'list' -d 'Print verbose list of items' -n "__fish_seen_subcommand_from extract" -complete -c borg -f -s n -l 'dry-run' -d 'Do not actually extract any files' -n "__fish_seen_subcommand_from extract" -complete -c borg -f -l 'numeric-ids' -d 'Only obey numeric user:group identifiers' -n "__fish_seen_subcommand_from extract" -complete -c borg -f -l 'nobsdflags' -d 'Do not extract/set bsdflags' -n "__fish_seen_subcommand_from extract" -complete -c borg -f -l 'noflags' -d 'Do not extract/set flags' -n "__fish_seen_subcommand_from extract" -complete -c borg -f -l 'noacls' -d 'Do not extract/set ACLs' -n "__fish_seen_subcommand_from extract" -complete -c borg -f -l 'noxattrs' -d 'Do not extract/set xattrs' -n "__fish_seen_subcommand_from extract" -complete -c borg -f -l 'stdout' -d 'Write all extracted data to stdout' -n "__fish_seen_subcommand_from extract" -complete -c borg -f -l 'sparse' -d 'Create holes in output sparse file' -n "__fish_seen_subcommand_from extract" -# Exclusion options -complete -c borg -s e -l 'exclude' -d 'Exclude paths matching PATTERN' -n "__fish_seen_subcommand_from extract" -complete -c borg -l 'exclude-from' -d 'Read exclude patterns from EXCLUDEFILE' -n "__fish_seen_subcommand_from extract" -complete -c borg -l 'pattern' -d 'Include/exclude paths matching PATTERN' -n "__fish_seen_subcommand_from extract" -complete -c borg -l 'patterns-from' -d 'Include/exclude paths from PATTERNFILE' -n "__fish_seen_subcommand_from extract" -complete -c borg -f -l 'strip-components' -d 'Remove NUMBER of leading path elements' -n "__fish_seen_subcommand_from extract" - -# borg check options -complete -c borg -f -l 'repository-only' -d 'Only perform repository checks' -n "__fish_seen_subcommand_from check" -complete -c borg -f -l 'archives-only' -d 'Only perform archives checks' -n "__fish_seen_subcommand_from check" -complete -c borg -f -l 'verify-data' -d 'Cryptographic integrity verification' -n "__fish_seen_subcommand_from check" -complete -c borg -f -l 'repair' -d 'Attempt to repair found inconsistencies' -n "__fish_seen_subcommand_from check" -complete -c borg -f -l 'max-duration' -d 'Partial repo check for max. SECONDS' -n "__fish_seen_subcommand_from check" -# Archive filters -complete -c borg -f -s P -l 'prefix' -d 'Only archive names starting with PREFIX' -n "__fish_seen_subcommand_from check" -complete -c borg -f -s a -l 'match-archives' -d 'Only consider archives matching all patterns' -n "__fish_seen_subcommand_from check" -set -l sort_keys "timestamp archive name id tags host user" -complete -c borg -f -l 'sort-by' -d 'Sorting KEYS [timestamp]' -a "$sort_keys" -n "__fish_seen_subcommand_from check" -complete -c borg -f -l 'first' -d 'Only first N archives' -n "__fish_seen_subcommand_from check" -complete -c borg -f -l 'last' -d 'Only last N archives' -n "__fish_seen_subcommand_from check" -complete -c borg -f -l 'oldest' -d 'Consider archives within TIMESPAN from oldest' -n "__fish_seen_subcommand_from check" -complete -c borg -f -l 'newest' -d 'Consider archives within TIMESPAN from newest' -n "__fish_seen_subcommand_from check" -complete -c borg -f -l 'older' -d 'Consider archives older than TIMESPAN' -n "__fish_seen_subcommand_from check" -complete -c borg -f -l 'newer' -d 'Consider archives newer than TIMESPAN' -n "__fish_seen_subcommand_from check" - -# borg rename -# no specific options - -# borg list options -complete -c borg -f -l 'short' -d 'Only print file/directory names' -n "__fish_seen_subcommand_from list" -complete -c borg -f -l 'format' -d 'Specify FORMAT for file listing' -n "__fish_seen_subcommand_from list" -complete -c borg -f -l 'json' -d 'List contents in json format' -n "__fish_seen_subcommand_from list" -complete -c borg -f -l 'json-lines' -d 'List contents in json lines format' -n "__fish_seen_subcommand_from list" -# Archive filters -complete -c borg -f -s P -l 'prefix' -d 'Only archive names starting with PREFIX' -n "__fish_seen_subcommand_from list" -complete -c borg -f -s a -l 'match-archives' -d 'Only consider archives matching all patterns' -n "__fish_seen_subcommand_from list" -set -l sort_keys "timestamp archive name id tags host user" -complete -c borg -f -l 'sort-by' -d 'Sorting KEYS [timestamp]' -a "$sort_keys" -n "__fish_seen_subcommand_from list" -complete -c borg -f -l 'first' -d 'Only first N archives' -n "__fish_seen_subcommand_from list" -complete -c borg -f -l 'last' -d 'Only last N archives' -n "__fish_seen_subcommand_from list" -complete -c borg -f -l 'oldest' -d 'Consider archives within TIMESPAN from oldest' -n "__fish_seen_subcommand_from list" -complete -c borg -f -l 'newest' -d 'Consider archives within TIMESPAN from newest' -n "__fish_seen_subcommand_from list" -complete -c borg -f -l 'older' -d 'Consider archives older than TIMESPAN' -n "__fish_seen_subcommand_from list" -complete -c borg -f -l 'newer' -d 'Consider archives newer than TIMESPAN' -n "__fish_seen_subcommand_from list" -# Exclusion options -complete -c borg -s e -l 'exclude' -d 'Exclude paths matching PATTERN' -n "__fish_seen_subcommand_from list" -complete -c borg -l 'exclude-from' -d 'Read exclude patterns from EXCLUDEFILE' -n "__fish_seen_subcommand_from list" -complete -c borg -f -l 'pattern' -d 'Include/exclude paths matching PATTERN' -n "__fish_seen_subcommand_from list" -complete -c borg -l 'patterns-from' -d 'Include/exclude paths from PATTERNFILE' -n "__fish_seen_subcommand_from list" - -# borg diff options -complete -c borg -f -l 'numeric-ids' -d 'Only consider numeric user:group' -n "__fish_seen_subcommand_from diff" -complete -c borg -f -l 'same-chunker-params' -d 'Override check of chunker parameters' -n "__fish_seen_subcommand_from diff" -complete -c borg -f -l 'sort' -d 'Sort the output lines by file path' -n "__fish_seen_subcommand_from diff" -complete -c borg -f -l 'json-lines' -d 'Format output as JSON Lines' -n "__fish_seen_subcommand_from diff" -# Exclusion options -complete -c borg -s e -l 'exclude' -d 'Exclude paths matching PATTERN' -n "__fish_seen_subcommand_from diff" -complete -c borg -l 'exclude-from' -d 'Read exclude patterns from EXCLUDEFILE' -n "__fish_seen_subcommand_from diff" -complete -c borg -f -l 'pattern' -d 'Include/exclude paths matching PATTERN' -n "__fish_seen_subcommand_from diff" -complete -c borg -l 'patterns-from' -d 'Include/exclude paths from PATTERNFILE' -n "__fish_seen_subcommand_from diff" - -# borg delete options -complete -c borg -f -s n -l 'dry-run' -d 'Do not change the repository' -n "__fish_seen_subcommand_from delete" -complete -c borg -f -l 'list' -d 'Output verbose list of archives' -n "__fish_seen_subcommand_from delete" -# Archive filters -complete -c borg -f -s P -l 'prefix' -d 'Only archive names starting with PREFIX' -n "__fish_seen_subcommand_from delete" -complete -c borg -f -s a -l 'match-archives' -d 'Only consider archives matching all patterns' -n "__fish_seen_subcommand_from delete" -set -l sort_keys "timestamp archive name id tags host user" -complete -c borg -f -l 'sort-by' -d 'Sorting KEYS [timestamp]' -a "$sort_keys" -n "__fish_seen_subcommand_from delete" -complete -c borg -f -l 'first' -d 'Only first N archives' -n "__fish_seen_subcommand_from delete" -complete -c borg -f -l 'last' -d 'Only last N archives' -n "__fish_seen_subcommand_from delete" -complete -c borg -f -l 'oldest' -d 'Consider archives within TIMESPAN from oldest' -n "__fish_seen_subcommand_from delete" -complete -c borg -f -l 'newest' -d 'Consider archives within TIMESPAN from newest' -n "__fish_seen_subcommand_from delete" -complete -c borg -f -l 'older' -d 'Consider archives older than TIMESPAN' -n "__fish_seen_subcommand_from delete" -complete -c borg -f -l 'newer' -d 'Consider archives newer than TIMESPAN' -n "__fish_seen_subcommand_from delete" - -# borg prune options -complete -c borg -f -s n -l 'dry-run' -d 'Do not change the repository' -n "__fish_seen_subcommand_from prune" -complete -c borg -f -l 'force' -d 'Force pruning of corrupted archives' -n "__fish_seen_subcommand_from prune" -complete -c borg -f -s s -l 'stats' -d 'Print verbose statistics' -n "__fish_seen_subcommand_from prune" -complete -c borg -f -l 'list' -d 'Print verbose list of items' -n "__fish_seen_subcommand_from prune" -complete -c borg -f -l 'keep-within' -d 'Keep archives within time INTERVAL' -n "__fish_seen_subcommand_from prune" -complete -c borg -f -l 'keep-last' -d 'NUMBER of secondly archives to keep' -n "__fish_seen_subcommand_from prune" -complete -c borg -f -l 'keep-secondly' -d 'NUMBER of secondly archives to keep' -n "__fish_seen_subcommand_from prune" -complete -c borg -f -l 'keep-minutely' -d 'NUMBER of minutely archives to keep' -n "__fish_seen_subcommand_from prune" -complete -c borg -f -s H -l 'keep-hourly' -d 'NUMBER of hourly archives to keep' -n "__fish_seen_subcommand_from prune" -complete -c borg -f -s d -l 'keep-daily' -d 'NUMBER of daily archives to keep' -n "__fish_seen_subcommand_from prune" -complete -c borg -f -s w -l 'keep-weekly' -d 'NUMBER of weekly archives to keep' -n "__fish_seen_subcommand_from prune" -complete -c borg -f -s m -l 'keep-monthly' -d 'NUMBER of monthly archives to keep' -n "__fish_seen_subcommand_from prune" -complete -c borg -f -l 'keep-13weekly' -d 'NUMBER of quarterly archives to keep (13 week strategy)' -n "__fish_seen_subcommand_from prune" -complete -c borg -f -l 'keep-3monthly' -d 'NUMBER of quarterly archives to keep (3 month strategy)' -n "__fish_seen_subcommand_from prune" -complete -c borg -f -s y -l 'keep-yearly' -d 'NUMBER of yearly archives to keep' -n "__fish_seen_subcommand_from prune" -# Archive filters -complete -c borg -f -s P -l 'prefix' -d 'Only archive names starting with PREFIX' -n "__fish_seen_subcommand_from prune" -complete -c borg -f -s a -l 'match-archives' -d 'Only consider archives matching all patterns' -n "__fish_seen_subcommand_from prune" - -# borg compact options -complete -c borg -f -s n -l 'dry-run' -d 'Do nothing' -n "__fish_seen_subcommand_from compact" -complete -c borg -f -s s -l 'stats' -d 'Print statistics (might be much slower)' -n "__fish_seen_subcommand_from compact" - -# borg info options -complete -c borg -f -l 'json' -d 'Format output in json format' -n "__fish_seen_subcommand_from info" -# Archive filters -complete -c borg -f -s P -l 'prefix' -d 'Only archive names starting with PREFIX' -n "__fish_seen_subcommand_from info" -complete -c borg -f -s a -l 'match-archives' -d 'Only consider archives matching all patterns' -n "__fish_seen_subcommand_from info" -set -l sort_keys "timestamp archive name id tags host user" -complete -c borg -f -l 'sort-by' -d 'Sorting KEYS [timestamp]' -a "$sort_keys" -n "__fish_seen_subcommand_from info" -complete -c borg -f -l 'first' -d 'Only first N archives' -n "__fish_seen_subcommand_from info" -complete -c borg -f -l 'last' -d 'Only last N archives' -n "__fish_seen_subcommand_from info" -complete -c borg -f -l 'oldest' -d 'Consider archives within TIMESPAN from oldest' -n "__fish_seen_subcommand_from info" -complete -c borg -f -l 'newest' -d 'Consider archives within TIMESPAN from newest' -n "__fish_seen_subcommand_from info" -complete -c borg -f -l 'older' -d 'Consider archives older than TIMESPAN' -n "__fish_seen_subcommand_from info" -complete -c borg -f -l 'newer' -d 'Consider archives newer than TIMESPAN' -n "__fish_seen_subcommand_from info" - -# borg repo-list options -complete -c borg -f -l 'short' -d 'Only print the archive IDs, nothing else' -n "__fish_seen_subcommand_from repo-list" -complete -c borg -f -l 'format' -d 'Specify format for archive listing' -n "__fish_seen_subcommand_from repo-list" -complete -c borg -f -l 'json' -d 'Format output as JSON' -n "__fish_seen_subcommand_from repo-list" -# Archive filters -complete -c borg -f -s P -l 'prefix' -d 'Only archive names starting with PREFIX' -n "__fish_seen_subcommand_from repo-list" -complete -c borg -f -s a -l 'match-archives' -d 'Only consider archives matching all patterns' -n "__fish_seen_subcommand_from repo-list" -set -l sort_keys "timestamp archive name id tags host user" -complete -c borg -f -l 'sort-by' -d 'Sorting KEYS [timestamp]' -a "$sort_keys" -n "__fish_seen_subcommand_from repo-list" -complete -c borg -f -l 'first' -d 'Only first N archives' -n "__fish_seen_subcommand_from repo-list" -complete -c borg -f -l 'last' -d 'Only last N archives' -n "__fish_seen_subcommand_from repo-list" -complete -c borg -f -l 'oldest' -d 'Consider archives within TIMESPAN from oldest' -n "__fish_seen_subcommand_from repo-list" -complete -c borg -f -l 'newest' -d 'Consider archives within TIMESPAN from newest' -n "__fish_seen_subcommand_from repo-list" -complete -c borg -f -l 'older' -d 'Consider archives older than TIMESPAN' -n "__fish_seen_subcommand_from repo-list" -complete -c borg -f -l 'newer' -d 'Consider archives newer than TIMESPAN' -n "__fish_seen_subcommand_from repo-list" -complete -c borg -f -l 'deleted' -d 'Consider only soft-deleted archives' -n "__fish_seen_subcommand_from repo-list" - -# borg mount options -complete -c borg -f -s f -l 'foreground' -d 'Stay in foreground, do not daemonize' -n "__fish_seen_subcommand_from mount" -# FIXME This list is probably not full, but I tried to pick only those that are relevant to borg mount -o: -set -l fuse_options "ac_attr_timeout= allow_damaged_files allow_other allow_root attr_timeout= auto auto_cache auto_unmount default_permissions entry_timeout= gid= group_id= kernel_cache max_read= negative_timeout= noauto noforget remember= remount rootmode= uid= umask= user user_id= versions" -complete -c borg -f -s o -d 'Fuse mount OPTION' -a "$fuse_options" -n "__fish_seen_subcommand_from mount" -# Archive filters -complete -c borg -f -s P -l 'prefix' -d 'Only archive names starting with PREFIX' -n "__fish_seen_subcommand_from mount" -complete -c borg -f -s a -l 'match-archives' -d 'Only consider archives matching all patterns' -n "__fish_seen_subcommand_from mount" -set -l sort_keys "timestamp archive name id tags host user" -complete -c borg -f -l 'sort-by' -d 'Sorting KEYS [timestamp]' -a "$sort_keys" -n "__fish_seen_subcommand_from mount" -complete -c borg -f -l 'first' -d 'Only first N archives' -n "__fish_seen_subcommand_from mount" -complete -c borg -f -l 'last' -d 'Only last N archives' -n "__fish_seen_subcommand_from mount" -complete -c borg -f -l 'oldest' -d 'Consider archives within TIMESPAN from oldest' -n "__fish_seen_subcommand_from mount" -complete -c borg -f -l 'newest' -d 'Consider archives within TIMESPAN from newest' -n "__fish_seen_subcommand_from mount" -complete -c borg -f -l 'older' -d 'Consider archives older than TIMESPAN' -n "__fish_seen_subcommand_from mount" -complete -c borg -f -l 'newer' -d 'Consider archives newer than TIMESPAN' -n "__fish_seen_subcommand_from mount" -# Exclusion options -complete -c borg -s e -l 'exclude' -d 'Exclude paths matching PATTERN' -n "__fish_seen_subcommand_from mount" -complete -c borg -l 'exclude-from' -d 'Read exclude patterns from EXCLUDEFILE' -n "__fish_seen_subcommand_from mount" -complete -c borg -f -l 'pattern' -d 'Include/exclude paths matching PATTERN' -n "__fish_seen_subcommand_from mount" -complete -c borg -l 'patterns-from' -d 'Include/exclude paths from PATTERNFILE' -n "__fish_seen_subcommand_from mount" -complete -c borg -f -l 'strip-components' -d 'Remove NUMBER of leading path elements' -n "__fish_seen_subcommand_from mount" - -# borg umount -# no specific options - -# borg tag options -complete -c borg -f -l 'set' -d 'Set tags (can be given multiple times)' -n "__fish_seen_subcommand_from tag" -complete -c borg -f -l 'add' -d 'Add tags (can be given multiple times)' -n "__fish_seen_subcommand_from tag" -complete -c borg -f -l 'remove' -d 'Remove tags (can be given multiple times)' -n "__fish_seen_subcommand_from tag" -# Archive filters -complete -c borg -f -s a -l 'match-archives' -d 'Only consider archives matching all patterns' -n "__fish_seen_subcommand_from tag" -set -l sort_keys "timestamp archive name id tags host user" -complete -c borg -f -l 'sort-by' -d 'Sorting KEYS [timestamp]' -a "$sort_keys" -n "__fish_seen_subcommand_from tag" -complete -c borg -f -l 'first' -d 'Only first N archives' -n "__fish_seen_subcommand_from tag" -complete -c borg -f -l 'last' -d 'Only last N archives' -n "__fish_seen_subcommand_from tag" -complete -c borg -f -l 'oldest' -d 'Consider archives within TIMESPAN from oldest' -n "__fish_seen_subcommand_from tag" -complete -c borg -f -l 'newest' -d 'Consider archives within TIMESPAN from newest' -n "__fish_seen_subcommand_from tag" -complete -c borg -f -l 'older' -d 'Consider archives older than TIMESPAN' -n "__fish_seen_subcommand_from tag" -complete -c borg -f -l 'newer' -d 'Consider archives newer than TIMESPAN' -n "__fish_seen_subcommand_from tag" - -# borg key change-passphrase -# no specific options - -# borg key export -complete -c borg -f -l 'paper' -d 'Create an export for printing' -n "__fish_seen_subcommand_from export" -complete -c borg -f -l 'qr-html' -d 'Create an html file for printing and qr' -n "__fish_seen_subcommand_from export" - -# borg transfer options -complete -c borg -f -s n -l 'dry-run' -d 'Do not change repository, just check' -n "__fish_seen_subcommand_from transfer" -complete -c borg -f -l 'other-repo' -d 'Transfer archives from the other repository' -n "__fish_seen_subcommand_from transfer" -complete -c borg -f -l 'from-borg1' -d 'Other repository is borg 1.x' -n "__fish_seen_subcommand_from transfer" -complete -c borg -f -l 'upgrader' -d 'Use the upgrader to convert transferred data' -n "__fish_seen_subcommand_from transfer" -complete -c borg -f -s C -l 'compression' -d 'Select compression algorithm' -a "$compression_methods" -n "__fish_seen_subcommand_from transfer" -complete -c borg -f -l 'recompress' -d 'Recompress chunks CONDITION' -a "$recompress_when" -n "__fish_seen_subcommand_from transfer" -complete -c borg -f -l 'chunker-params' -d 'Chunker PARAMETERS [19,23,21,4095]' -n "__fish_seen_subcommand_from transfer" -# Archive filters -complete -c borg -f -s a -l 'match-archives' -d 'Only consider archives matching all patterns' -n "__fish_seen_subcommand_from transfer" -set -l sort_keys "timestamp archive name id tags host user" -complete -c borg -f -l 'sort-by' -d 'Sorting KEYS [timestamp]' -a "$sort_keys" -n "__fish_seen_subcommand_from transfer" -complete -c borg -f -l 'first' -d 'Only first N archives' -n "__fish_seen_subcommand_from transfer" -complete -c borg -f -l 'last' -d 'Only last N archives' -n "__fish_seen_subcommand_from transfer" -complete -c borg -f -l 'oldest' -d 'Consider archives within TIMESPAN from oldest' -n "__fish_seen_subcommand_from transfer" -complete -c borg -f -l 'newest' -d 'Consider archives within TIMESPAN from newest' -n "__fish_seen_subcommand_from transfer" -complete -c borg -f -l 'older' -d 'Consider archives older than TIMESPAN' -n "__fish_seen_subcommand_from transfer" -complete -c borg -f -l 'newer' -d 'Consider archives newer than TIMESPAN' -n "__fish_seen_subcommand_from transfer" - -# borg key import -complete -c borg -f -l 'paper' -d 'Import from a backup done with --paper' -n "__fish_seen_subcommand_from import" - -# borg undelete options -complete -c borg -f -s n -l 'dry-run' -d 'Do not change repository' -n "__fish_seen_subcommand_from undelete" -complete -c borg -f -l 'list' -d 'Output verbose list of archives' -n "__fish_seen_subcommand_from undelete" -# Archive filters -complete -c borg -f -s a -l 'match-archives' -d 'Only consider archives matching all patterns' -n "__fish_seen_subcommand_from undelete" -set -l sort_keys "timestamp archive name id tags host user" -complete -c borg -f -l 'sort-by' -d 'Sorting KEYS [timestamp]' -a "$sort_keys" -n "__fish_seen_subcommand_from undelete" -complete -c borg -f -l 'first' -d 'Only first N archives' -n "__fish_seen_subcommand_from undelete" -complete -c borg -f -l 'last' -d 'Only last N archives' -n "__fish_seen_subcommand_from undelete" -complete -c borg -f -l 'oldest' -d 'Consider archives within TIMESPAN from oldest' -n "__fish_seen_subcommand_from undelete" -complete -c borg -f -l 'newest' -d 'Consider archives within TIMESPAN from newest' -n "__fish_seen_subcommand_from undelete" -complete -c borg -f -l 'older' -d 'Consider archives older than TIMESPAN' -n "__fish_seen_subcommand_from undelete" -complete -c borg -f -l 'newer' -d 'Consider archives newer than TIMESPAN' -n "__fish_seen_subcommand_from undelete" - - -# borg recreate -complete -c borg -f -l 'list' -d 'Print verbose list of items' -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -l 'filter' -d 'Only items with given STATUSCHARS' -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -s n -l 'dry-run' -d 'Do not change the repository' -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -s s -l 'stats' -d 'Print verbose statistics' -n "__fish_seen_subcommand_from recreate" -# Exclusion options -complete -c borg -s e -l 'exclude' -d 'Exclude paths matching PATTERN' -n "__fish_seen_subcommand_from recreate" -complete -c borg -l 'exclude-from' -d 'Read exclude patterns from EXCLUDEFILE' -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -l 'pattern' -d 'Include/exclude paths matching PATTERN' -n "__fish_seen_subcommand_from recreate" -complete -c borg -l 'patterns-from' -d 'Include/exclude paths from PATTERNFILE' -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -l 'exclude-caches' -d 'Exclude directories tagged as cache' -n "__fish_seen_subcommand_from recreate" -complete -c borg -l 'exclude-if-present' -d 'Exclude directories that contain FILENAME' -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -l 'keep-exclude-tags' -d 'Keep tag files of excluded directories' -n "__fish_seen_subcommand_from recreate" -# Archive filters -complete -c borg -f -s a -l 'match-archives' -d 'Only consider archives matching all patterns' -n "__fish_seen_subcommand_from recreate" -set -l sort_keys "timestamp archive name id tags host user" -complete -c borg -f -l 'sort-by' -d 'Sorting KEYS [timestamp]' -a "$sort_keys" -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -l 'first' -d 'Only first N archives' -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -l 'last' -d 'Only last N archives' -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -l 'oldest' -d 'Consider archives within TIMESPAN from oldest' -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -l 'newest' -d 'Consider archives within TIMESPAN from newest' -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -l 'older' -d 'Consider archives older than TIMESPAN' -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -l 'newer' -d 'Consider archives newer than TIMESPAN' -n "__fish_seen_subcommand_from recreate" -# Archive options -complete -c borg -f -l 'target' -d "Create a new ARCHIVE" -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -l 'comment' -d 'Add COMMENT to the archive' -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -l 'timestamp' -d 'Set creation TIME (yyyy-mm-ddThh:mm:ss)' -n "__fish_seen_subcommand_from recreate" -complete -c borg -l 'timestamp' -d 'Set creation time using reference FILE' -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -s C -l 'compression' -d 'Select compression ALGORITHM,LEVEL [lz4]' -a "$compression_methods" -n "__fish_seen_subcommand_from recreate" -set -l recompress_when "if-different always never" -complete -c borg -f -l 'recompress' -d 'Recompress chunks CONDITION' -a "$recompress_when" -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -l 'chunker-params' -d 'Chunker PARAMETERS [19,23,21,4095]' -n "__fish_seen_subcommand_from recreate" - -# borg export-tar options -complete -c borg -l 'tar-filter' -d 'Filter program to pipe data through' -n "__fish_seen_subcommand_from export-tar" -complete -c borg -f -l 'list' -d 'Print verbose list of items' -n "__fish_seen_subcommand_from export-tar" -complete -c borg -f -l 'tar-format' -d 'Select tar format: BORG, PAX or GNU' -n "__fish_seen_subcommand_from export-tar" -# Exclusion options -complete -c borg -s e -l 'exclude' -d 'Exclude paths matching PATTERN' -n "__fish_seen_subcommand_from export-tar" -complete -c borg -l 'exclude-from' -d 'Read exclude patterns from EXCLUDEFILE' -n "__fish_seen_subcommand_from export-tar" -complete -c borg -f -l 'pattern' -d 'Include/exclude paths matching PATTERN' -n "__fish_seen_subcommand_from export-tar" -complete -c borg -l 'patterns-from' -d 'Include/exclude paths from PATTERNFILE' -n "__fish_seen_subcommand_from export-tar" -complete -c borg -f -l 'strip-components' -d 'Remove NUMBER of leading path elements' -n "__fish_seen_subcommand_from export-tar" - -# borg import-tar options -complete -c borg -l 'tar-filter' -d 'Filter program to pipe data through' -n "__fish_seen_subcommand_from import-tar" -complete -c borg -f -s s -l 'stats' -d 'Print statistics for the created archive' -n "__fish_seen_subcommand_from import-tar" -complete -c borg -f -l 'list' -d 'Print verbose list of items' -n "__fish_seen_subcommand_from import-tar" -complete -c borg -f -l 'filter' -d 'Only display items with given STATUSCHARS' -n "__fish_seen_subcommand_from import-tar" -complete -c borg -f -l 'json' -d 'Output stats as JSON' -n "__fish_seen_subcommand_from import-tar" -complete -c borg -f -l 'ignore-zeros' -d 'Ignore zero-filled blocks in the input' -n "__fish_seen_subcommand_from import-tar" -complete -c borg -f -l 'comment' -d 'Add COMMENT to the archive' -n "__fish_seen_subcommand_from import-tar" -complete -c borg -f -l 'timestamp' -d 'Set creation TIME (yyyy-mm-ddThh:mm:ss)' -n "__fish_seen_subcommand_from import-tar" -complete -c borg -f -l 'chunker-params' -d 'Chunker PARAMETERS [19,23,21,4095]' -n "__fish_seen_subcommand_from import-tar" -complete -c borg -f -s C -l 'compression' -d 'Select compression ALGORITHM,LEVEL [lz4]' -a "$compression_methods" -n "__fish_seen_subcommand_from import-tar" -# Exclusion options -complete -c borg -s e -l 'exclude' -d 'Exclude paths matching PATTERN' -n "__fish_seen_subcommand_from recreate" -complete -c borg -l 'exclude-from' -d 'Read exclude patterns from EXCLUDEFILE' -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -l 'pattern' -d 'Include/exclude paths matching PATTERN' -n "__fish_seen_subcommand_from recreate" -complete -c borg -l 'patterns-from' -d 'Include/exclude paths from PATTERNFILE' -n "__fish_seen_subcommand_from recreate" -complete -c borg -f -l 'strip-components' -d 'Remove NUMBER of leading path elements' -n "__fish_seen_subcommand_from recreate" - -# borg serve -complete -c borg -l 'restrict-to-path' -d 'Restrict repository access to PATH' -n "__fish_seen_subcommand_from serve" -complete -c borg -l 'restrict-to-repository' -d 'Restrict repository access at PATH' -n "__fish_seen_subcommand_from serve" - - -# borg with-lock -# no specific options - -# borg break-lock -# no specific options - -# borg benchmark -# no specific options - -# borg help -# no specific options - - -function __fish_borg_archives - # additionally to aid:XXXXXXXX we show archive (series) name and timestamp - borg repo-list --format="aid:{id:.8}{TAB}{archive} {start}{NEWLINE}" 2>/dev/null -end - -function __fish_borg_archive_arg --description 'Test if current command is a specific borg command with token count' --argument command token_count - # Check if we're in the context of a specific borg command - set -l tokens (commandline --tokenize) - set -l cmdline (commandline --current-process) - - # Make sure we're in a borg command context - if not test $tokens[1] = "borg" - return 1 - end - - # Make sure we're in the specific command context - if not test $tokens[2] = "$command" - return 1 - end - - # Check if we're at the right token position - if not test (count $tokens) "-eq" "$token_count" - return 1 - end - - # Additional check to ensure we're not in the middle of typing an option - if string match --quiet --regex -- "^-" (commandline --current-token) - return 1 - end - - return 0 -end - -# The following completions use the -F flag to force disable filename completion -# for various borg commands, ensuring only archive names are suggested. -# We also use the -e flag to explicitly erase all default completions before adding our custom ones. - -# Global rules to disable filename completions for specific borg commands -# This ensures that no filename completions are shown for these commands -complete -c borg -e -n '__fish_seen_subcommand_from diff delete list info extract mount export-tar rename tag undelete recreate transfer check analyze' - -# First, explicitly erase all default completions for each command -# This is the most specific rule and should take precedence -complete -c borg -e -n '__fish_borg_archive_arg "diff" 2' -complete -c borg -e -n '__fish_borg_archive_arg "diff" 3' -complete -c borg -e -n '__fish_borg_archive_arg "delete" 2' -complete -c borg -e -n '__fish_borg_archive_arg "list" 2' -complete -c borg -e -n '__fish_borg_archive_arg "info" 2' -complete -c borg -e -n '__fish_borg_archive_arg "extract" 2' -complete -c borg -e -n '__fish_borg_archive_arg "mount" 2' -complete -c borg -e -n '__fish_borg_archive_arg "export-tar" 2' -complete -c borg -e -n '__fish_borg_archive_arg "rename" 2' -complete -c borg -e -n '__fish_borg_archive_arg "tag" 2' -complete -c borg -e -n '__fish_borg_archive_arg "undelete" 2' -complete -c borg -e -n '__fish_borg_archive_arg "recreate" 2' -complete -c borg -e -n '__fish_borg_archive_arg "transfer" 2' -complete -c borg -e -n '__fish_borg_archive_arg "check" 2' -complete -c borg -e -n '__fish_borg_archive_arg "analyze" 2' - -# Also add specific rules to disable filename completions at the exact position -# This ensures that no filename completions are shown at the position where we expect archive names -complete -c borg --no-files -n '__fish_borg_archive_arg "diff" 2' -complete -c borg --no-files -n '__fish_borg_archive_arg "diff" 3' -complete -c borg --no-files -n '__fish_borg_archive_arg "delete" 2' -complete -c borg --no-files -n '__fish_borg_archive_arg "list" 2' -complete -c borg --no-files -n '__fish_borg_archive_arg "info" 2' -complete -c borg --no-files -n '__fish_borg_archive_arg "extract" 2' -complete -c borg --no-files -n '__fish_borg_archive_arg "mount" 2' -complete -c borg --no-files -n '__fish_borg_archive_arg "export-tar" 2' -complete -c borg --no-files -n '__fish_borg_archive_arg "rename" 2' -complete -c borg --no-files -n '__fish_borg_archive_arg "tag" 2' -complete -c borg --no-files -n '__fish_borg_archive_arg "undelete" 2' -complete -c borg --no-files -n '__fish_borg_archive_arg "recreate" 2' -complete -c borg --no-files -n '__fish_borg_archive_arg "transfer" 2' -complete -c borg --no-files -n '__fish_borg_archive_arg "check" 2' -complete -c borg --no-files -n '__fish_borg_archive_arg "analyze" 2' - -# Then add our custom completions with high priority and no-files -# This ensures no filename completions are shown and our custom completions take precedence -complete -c borg -p 100 -n '__fish_borg_archive_arg "diff" 2' -a '(__fish_borg_archives)' --no-files -complete -c borg -p 100 -n '__fish_borg_archive_arg "diff" 3' -a '(__fish_borg_archives)' --no-files -complete -c borg -p 100 -n '__fish_borg_archive_arg "delete" 2' -a '(__fish_borg_archives)' --no-files -complete -c borg -p 100 -n '__fish_borg_archive_arg "list" 2' -a '(__fish_borg_archives)' --no-files -complete -c borg -p 100 -n '__fish_borg_archive_arg "info" 2' -a '(__fish_borg_archives)' --no-files -complete -c borg -p 100 -n '__fish_borg_archive_arg "extract" 2' -a '(__fish_borg_archives)' --no-files -complete -c borg -p 100 -n '__fish_borg_archive_arg "mount" 2' -a '(__fish_borg_archives)' --no-files -complete -c borg -p 100 -n '__fish_borg_archive_arg "export-tar" 2' -a '(__fish_borg_archives)' --no-files -complete -c borg -p 100 -n '__fish_borg_archive_arg "rename" 2' -a '(__fish_borg_archives)' --no-files -complete -c borg -p 100 -n '__fish_borg_archive_arg "tag" 2' -a '(__fish_borg_archives)' --no-files -complete -c borg -p 100 -n '__fish_borg_archive_arg "undelete" 2' -a '(__fish_borg_archives)' --no-files -complete -c borg -p 100 -n '__fish_borg_archive_arg "recreate" 2' -a '(__fish_borg_archives)' --no-files -complete -c borg -p 100 -n '__fish_borg_archive_arg "transfer" 2' -a '(__fish_borg_archives)' --no-files -complete -c borg -p 100 -n '__fish_borg_archive_arg "check" 2' -a '(__fish_borg_archives)' --no-files -complete -c borg -p 100 -n '__fish_borg_archive_arg "analyze" 2' -a '(__fish_borg_archives)' --no-files diff --git a/src/borg/archiver/_common.py b/src/borg/archiver/_common.py index 77787c26a5..a07cb30bfb 100644 --- a/src/borg/archiver/_common.py +++ b/src/borg/archiver/_common.py @@ -8,6 +8,7 @@ from ..cache import Cache, assert_secure from ..helpers import Error from ..helpers import SortBySpec, location_validator, Location, relative_time_marker_validator +from ..helpers import FilesystemPathSpec from ..helpers import Highlander, octal_int from ..helpers.argparsing import SUPPRESS, PositiveInt from ..helpers.nanorst import rst_to_terminal @@ -313,6 +314,7 @@ def define_exclude_and_patterns(add_option, *, tag_files=False, strip_components add_option( "--exclude-from", metavar="EXCLUDEFILE", + type=FilesystemPathSpec, action=ArgparseExcludeFileAction, help="read exclude patterns from EXCLUDEFILE, one per line", ) @@ -322,6 +324,7 @@ def define_exclude_and_patterns(add_option, *, tag_files=False, strip_components add_option( "--patterns-from", metavar="PATTERNFILE", + type=FilesystemPathSpec, action=ArgparsePatternFileAction, help="read include/exclude patterns from PATTERNFILE, one per line", ) diff --git a/src/borg/archiver/benchmark_cmd.py b/src/borg/archiver/benchmark_cmd.py index fade20ede2..b1a81abadf 100644 --- a/src/borg/archiver/benchmark_cmd.py +++ b/src/borg/archiver/benchmark_cmd.py @@ -7,6 +7,7 @@ from ..constants import * # NOQA from ..helpers import format_file_size, CompressionSpec +from ..helpers import FilesystemDirSpec from ..helpers import json_print from ..helpers import msgpack from ..helpers import get_reset_ec @@ -391,7 +392,9 @@ def build_parser_benchmarks(self, subparsers, common_parser, mid_common_parser): "crud", subparser, help="benchmarks Borg CRUD (create, extract, update, delete)." ) - subparser.add_argument("path", metavar="PATH", help="path where to create benchmark input data") + subparser.add_argument( + "path", metavar="PATH", type=FilesystemDirSpec, help="path where to create benchmark input data" + ) subparser.add_argument("--json-lines", action="store_true", help="Format output as JSON Lines.") bench_cpu_epilog = process_epilog( diff --git a/src/borg/archiver/completion_cmd.py b/src/borg/archiver/completion_cmd.py index e238cfa8bc..6a39c8f79d 100644 --- a/src/borg/archiver/completion_cmd.py +++ b/src/borg/archiver/completion_cmd.py @@ -2,8 +2,10 @@ Shell completion support for Borg commands. This module implements the `borg completion` command, which generates shell completion -scripts for bash and zsh. It uses the shtab library for basic completion generation -and extends it with custom dynamic completions for Borg-specific argument types. +scripts for bash, zsh, tcsh and fish. It uses the shtab library for basic completion +generation and extends it with custom dynamic completions for Borg-specific argument +types, via a per-shell preamble of completion functions that the generated script +refers to. Dynamic Completions ------------------- @@ -13,7 +15,7 @@ 1. Archive names/IDs (archivename_validator): - Completes archive names by default (e.g., "my-backup-2024") - Completes archive IDs when prefixed with "aid:" (e.g., "aid:12345678") - - In zsh, shows archive metadata (name, timestamp, user@host) as descriptions + - In zsh and fish, shows archive metadata (name, timestamp, user@host) as descriptions - Respects --repo/-r flags to query the correct repository 2. Sort keys (SortBySpec): @@ -30,8 +32,10 @@ 5. Chunker parameters (ChunkerParams): - Suggests chunker param examples (default, fixed,4194304, buzhash,19,23,21,4095, etc.) -6. Paths (PathSpec): - - Completes directories using standard shell directory completion +6. Paths (PathSpec, FilesystemPathSpec, FilesystemDirSpec): + - Completes files and directories for arguments taking a filesystem path + (FilesystemPathSpec, e.g. PATH, EXCLUDEFILE, TARFILE), and directories only + for arguments taking a directory (FilesystemDirSpec, e.g. MOUNTPOINT) 7. Help topics: - Completes help command topics and subcommand names @@ -59,6 +63,8 @@ SortBySpec, FilesCacheMode, PathSpec, + FilesystemPathSpec, + FilesystemDirSpec, ChunkerParams, CompressionSpec, tag_validator, @@ -626,6 +632,163 @@ } """ +# Global fish preamble providing the dynamic completion functions the generated +# script refers to (the fish counterparts of the bash/zsh ones above). +FISH_PREAMBLE_TMPL = r""" +# derive repo context from the command line: --repo=V, --repo V, -r=V, -rV, or -r V +function _borg_repo_args + set -l tokens (commandline -opc) + set -l n (count $tokens) + for i in (seq 2 $n) + set -l w $tokens[$i] + switch "$w" + case '--repo=*' + printf '%s\n' --repo (string replace -- '--repo=' '' $w) + return + case '-r=*' + printf '%s\n' -r (string replace -- '-r=' '' $w) + return + case '--repo' '-r' + if test $i -lt $n + printf '%s\n' $w $tokens[(math $i + 1)] + end + return + case '-r*' + printf '%s\n' -r (string replace -r -- '^-r' '' $w) + return + end + end +end + +# Complete archive names, or archive IDs when the current token starts with "aid:" +function _borg_complete_archive + set -l cur (commandline -ct) + set -l repo_args (_borg_repo_args) + if string match -q -- 'aid:*' $cur + set -l prefix (string replace -- 'aid:' '' $cur) + if test -n "$prefix"; and not string match -qr -- '^[0-9a-fA-F]+$' $prefix + return + end + # ask borg for IDs with metadata; avoid prompts and suppress stderr + borg repo-list $repo_args --format '{id}{TAB}{archive}{TAB}{time}{TAB}{username}@{hostname}{NL}' \ + 2>/dev/null /dev/null /dev/null = 4 +# (macOS still ships bash 3.2) +needs_bash4 = pytest.mark.skipif(bash_version() < 4, reason="Bash >= 4 not available") needs_zsh = pytest.mark.skipif(not cmd_available("zsh --version"), reason="Zsh not available") +needs_fish = pytest.mark.skipif(not cmd_available("fish --version"), reason="Fish not available") def _run_bash_completion_fn(completion_script, setup_code): @@ -38,6 +56,42 @@ def _run_bash_completion_fn(completion_script, setup_code): return result +def _fish_quote(text): + """Escape text for inclusion in a single-quoted fish string.""" + return text.replace("\\", "\\\\").replace("'", "\\'") + + +def _fish_complete_candidates(completion_script, cmdline, path_prepend=None): + """Source the completion script in fish, return the completion candidates for cmdline.""" + code = completion_script + f"\ncomplete -C'{_fish_quote(cmdline)}'\n" + with tempfile.NamedTemporaryFile(mode="w", suffix=".fish", delete=False) as f: + f.write(code) + script_path = f.name + env = dict(os.environ) + if path_prepend: + env["PATH"] = path_prepend + os.pathsep + env.get("PATH", "") + try: + result = subprocess.run(["fish", script_path], capture_output=True, text=True, timeout=120, env=env) + finally: + os.unlink(script_path) + assert result.returncode == 0, f"fish failed: {result.stderr}" + # each output line is "candidatedescription" (or just "candidate") + return [line.split("\t")[0] for line in result.stdout.splitlines() if line.strip()] + + +def _borg_shim_dir(tmp_path): + """Create a dir with a borg script that runs the borg under test, for use inside fish.""" + src_dir = os.path.dirname(os.path.dirname(os.path.abspath(borg.__file__))) + shim_dir = tmp_path / "borg-shim" + shim_dir.mkdir() + shim = shim_dir / "borg" + shim.write_text( + f'#!/bin/sh\nexport PYTHONPATH="{src_dir}${{PYTHONPATH:+:$PYTHONPATH}}"\nexec "{sys.executable}" -m borg "$@"\n' + ) + shim.chmod(0o755) + return str(shim_dir) + + # -- output sanity checks ----------------------------------------------------- @@ -57,6 +111,14 @@ def test_zsh_completion_nontrivial(archivers, request): assert output.count("\n") > 100, f"Zsh completion suspiciously few lines: {output.count(chr(10))}" +def test_fish_completion_nontrivial(archivers, request): + """Verify the generated Fish completion is non-trivially sized.""" + archiver = request.getfixturevalue(archivers) + output = cmd(archiver, "completion", "fish") + assert len(output) > 5000, f"Fish completion suspiciously small: {len(output)} chars" + assert output.count("\n") > 100, f"Fish completion suspiciously few lines: {output.count(chr(10))}" + + # -- syntax validation -------------------------------------------------------- @@ -90,6 +152,15 @@ def test_zsh_completion_syntax(archivers, request): assert result.returncode == 0, f"Generated Zsh completion has syntax errors: {result.stderr.decode()}" +@needs_fish +def test_fish_completion_syntax(archivers, request): + """Verify the generated Fish completion script has valid syntax.""" + archiver = request.getfixturevalue(archivers) + output = cmd(archiver, "completion", "fish") + result = _check_shell_syntax(output, "fish", ".fish") + assert result.returncode == 0, f"Generated Fish completion has syntax errors: {result.stderr.decode()}" + + # -- borg-specific preamble function behavior (bash) -------------------------- @@ -182,3 +253,137 @@ def test_bash_archive_aid_completion(archivers, request): assert len(lines) >= 1, "Expected at least one archive ID completion" for line in lines: assert line.startswith("aid:"), f"Expected aid: prefix, got: {line}" + + +# -- borg-specific completion behavior (fish) --------------------------------- + + +@needs_fish +def test_fish_subcommand_completion(archivers, request): + """Command names should be completed at the top level and inside command groups.""" + archiver = request.getfixturevalue(archivers) + script = cmd(archiver, "completion", "fish") + + candidates = _fish_complete_candidates(script, "borg ") + for name in ("create", "extract", "repo-list", "key", "debug"): + assert name in candidates, f"expected command {name} in: {candidates}" + + candidates = _fish_complete_candidates(script, "borg key ") + for name in ("export", "import", "change-passphrase"): + assert name in candidates, f"expected key subcommand {name} in: {candidates}" + + +@needs_fish +def test_fish_option_completion(archivers, request): + """Option names and static option values should be completed.""" + archiver = request.getfixturevalue(archivers) + script = cmd(archiver, "completion", "fish") + + candidates = _fish_complete_candidates(script, "borg create --compres") + assert "--compression" in candidates, f"expected --compression in: {candidates}" + + candidates = _fish_complete_candidates(script, "borg create --compression ") + assert "lz4" in candidates, f"expected lz4 in: {candidates}" + assert "zstd,3" in candidates, f"expected zstd,3 in: {candidates}" + + +@needs_fish +def test_fish_sortby_dedup(archivers, request): + """_borg_complete_sortby should not re-offer already-selected sort keys.""" + archiver = request.getfixturevalue(archivers) + script = cmd(archiver, "completion", "fish") + + candidates = _fish_complete_candidates(script, "borg repo-list --sort-by timestamp,") + assert "timestamp,archive" in candidates, f"expected timestamp,archive in: {candidates}" + assert "timestamp,timestamp" not in candidates, f"timestamp was re-offered: {candidates}" + + +@needs_fish +def test_fish_filescachemode_exclusivity(archivers, request): + """_borg_complete_filescachemode should enforce ctime/mtime and disabled mutual exclusion.""" + archiver = request.getfixturevalue(archivers) + script = cmd(archiver, "completion", "fish") + + candidates = _fish_complete_candidates(script, "borg create --files-cache ctime,") + assert "ctime,size" in candidates, f"expected ctime,size in: {candidates}" + assert "ctime,mtime" not in candidates, f"mtime offered after ctime: {candidates}" + assert "ctime,disabled" not in candidates, f"disabled offered after ctime: {candidates}" + + candidates = _fish_complete_candidates(script, "borg create --files-cache disabled,") + assert candidates == [], f"completions offered after disabled: {candidates}" + + +@needs_fish +def test_fish_path_completion(archivers, request, tmp_path): + """Arguments taking a filesystem path complete files/directories.""" + archiver = request.getfixturevalue(archivers) + (tmp_path / "somefile.txt").touch() + (tmp_path / "somedir").mkdir() + script = cmd(archiver, "completion", "fish") + prefix = str(tmp_path / "some") + + # positional PATH of "borg create" ... + candidates = _fish_complete_candidates(script, f"borg create archivename {prefix}") + assert any(c.endswith("somefile.txt") for c in candidates), f"file missing: {candidates}" + # ... and the value of an option taking a file + candidates = _fish_complete_candidates(script, f"borg create --exclude-from {prefix}") + assert any(c.endswith("somefile.txt") for c in candidates), f"file missing: {candidates}" + # arguments taking a directory only offer directories + candidates = _fish_complete_candidates(script, f"borg mount --repo /repo archivename {prefix}") + assert any(c.rstrip("/").endswith("somedir") for c in candidates), f"dir missing: {candidates}" + assert not any(c.endswith("somefile.txt") for c in candidates), f"file offered for MOUNTPOINT: {candidates}" + + +@needs_bash4 +def test_bash_path_completion(archivers, request, tmp_path): + """Arguments taking a filesystem path complete files in bash, too.""" + archiver = request.getfixturevalue(archivers) + (tmp_path / "somefile.txt").touch() + script = cmd(archiver, "completion", "bash") + prefix = str(tmp_path / "some") + + result = _run_bash_completion_fn( + script, + f'COMP_WORDS=(borg create --exclude-from "{prefix}")\n' + f"COMP_CWORD=3\n" + f"_shtab_borg\n" + f'printf "%s\\n" "${{COMPREPLY[@]}}"\n', + ) + assert result.returncode == 0, f"stderr: {result.stderr}" + assert "somefile.txt" in result.stdout, f"file missing: {result.stdout}" + + +@needs_fish +def test_fish_archive_name_completion(archivers, request, tmp_path): + """Archive names should be completed from a real repo.""" + archiver = request.getfixturevalue(archivers) + cmd(archiver, "repo-create", RK_ENCRYPTION) + cmd(archiver, "create", "mybackup-2024", archiver.input_path) + cmd(archiver, "create", "mybackup-2025", archiver.input_path) + + script = cmd(archiver, "completion", "fish") + repo = archiver.repository_path + + candidates = _fish_complete_candidates( + script, f"borg delete --repo {repo} mybackup", path_prepend=_borg_shim_dir(tmp_path) + ) + assert "mybackup-2024" in candidates, f"archive name missing: {candidates}" + assert "mybackup-2025" in candidates, f"archive name missing: {candidates}" + + +@needs_fish +def test_fish_archive_aid_completion(archivers, request, tmp_path): + """aid: prefixed archive IDs should be completed from a real repo.""" + archiver = request.getfixturevalue(archivers) + cmd(archiver, "repo-create", RK_ENCRYPTION) + cmd(archiver, "create", "testarchive", archiver.input_path) + + script = cmd(archiver, "completion", "fish") + repo = archiver.repository_path + + candidates = _fish_complete_candidates( + script, f"borg info --repo {repo} aid:", path_prepend=_borg_shim_dir(tmp_path) + ) + assert len(candidates) >= 1, "Expected at least one archive ID completion" + for candidate in candidates: + assert candidate.startswith("aid:"), f"Expected aid: prefix, got: {candidate}" diff --git a/src/borg/testsuite/shell_completions_test.py b/src/borg/testsuite/shell_completions_test.py deleted file mode 100644 index 4fa26faa5a..0000000000 --- a/src/borg/testsuite/shell_completions_test.py +++ /dev/null @@ -1,22 +0,0 @@ -import subprocess -from pathlib import Path - -import pytest - -SHELL_COMPLETIONS_DIR = Path(__file__).parent / ".." / ".." / ".." / "scripts" / "shell_completions" - - -def test_fish_completion_is_valid(): - """Test that the Fish completion file is valid Fish syntax.""" - fish_completion_file = SHELL_COMPLETIONS_DIR / "fish" / "borg.fish" - assert fish_completion_file.is_file() - - # Check if Fish is available - try: - subprocess.run(["fish", "--version"], capture_output=True, check=True) - except (subprocess.SubprocessError, FileNotFoundError): - pytest.skip("Fish not available") - - # Test whether the Fish completion file can be sourced without errors - result = subprocess.run(["fish", "-c", f"source {str(fish_completion_file)}"], capture_output=True) - assert result.returncode == 0, f"Fish completion file has syntax errors: {result.stderr.decode()}" From bfb0f5cc5f9b2e95694057338b38110830740d92 Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Wed, 18 Mar 2026 12:56:49 +0100 Subject: [PATCH 019/144] add tcsh completion support `borg completion tcsh` now generates a usable completion script: - add a tcsh preamble with the dynamic completion helpers (as aliases, since tcsh has no functions) and wire the tcsh patterns up for sort keys, files-cache mode, compression specs, chunker params, relative times, timestamps, file sizes and help topics. - complete archive names, archive IDs (when the token starts with "aid:") and tags. tcsh can neither define functions nor use backquotes there (the completion rule calling the helper is backquoted already), so both helpers run one POSIX sh script that parses $COMMAND_LINE for --repo/-r and queries `borg repo-list`. tcsh has no completion descriptions, so unlike zsh and fish these are plain candidate lists. The generator fixes this needed are all upstream in shtab now (tqdm/shtab#213, released in 1.9.3, which we already require): positional completion under subcommands at any depth, no out-of-range `$cmd` indexing, custom `.complete` patterns in multi-requirement rules, `--opt=` completion, and rule deduplication. One upstream fix is merged but not yet released (tqdm/shtab#241): completion patterns (`f`, `d`, ...) for a positional of a subcommand end up inside a `p@N@` rule, where tcsh runs the clauses as commands and only uses their output, so they do nothing - e.g. `borg umount ` would not complete a mountpoint. `_tcsh_anchor_positional_patterns` rewrites those into `n/` rules keyed off the preceding (sub)command word, producing exactly what a shtab with #241 generates. It is a no-op with such a shtab and can then be removed. Note that tcsh matches completions for positional arguments by word position, so options before an archive name shift it out of place and it is not completed - the `borg completion` epilog points at BORG_REPO for this. --- src/borg/archiver/completion_cmd.py | 120 +++++++++++++++++- .../testsuite/archiver/completion_cmd_test.py | 56 ++++++++ 2 files changed, 172 insertions(+), 4 deletions(-) diff --git a/src/borg/archiver/completion_cmd.py b/src/borg/archiver/completion_cmd.py index 6a39c8f79d..c50287735d 100644 --- a/src/borg/archiver/completion_cmd.py +++ b/src/borg/archiver/completion_cmd.py @@ -16,6 +16,7 @@ - Completes archive names by default (e.g., "my-backup-2024") - Completes archive IDs when prefixed with "aid:" (e.g., "aid:12345678") - In zsh and fish, shows archive metadata (name, timestamp, user@host) as descriptions + (tcsh has no completion descriptions) - Respects --repo/-r flags to query the correct repository 2. Sort keys (SortBySpec): @@ -54,6 +55,8 @@ - Suggests common file size values (500M, 1G, 10G, 100G, 1T, etc.) """ +import re + import shtab from ._common import process_epilog @@ -790,6 +793,103 @@ """ +TCSH_PREAMBLE_TMPL = r""" +# Dynamic completion helpers for tcsh + +alias _borg_complete_timestamp 'date +"%Y-%m-%dT%H:%M:%S"' + + +alias _borg_complete_sortby "echo {SORT_KEYS}" +alias _borg_complete_filescachemode "echo {FCM_KEYS}" +alias _borg_help_topics "echo {HELP_CHOICES}" +alias _borg_complete_compression_spec "echo {COMP_SPEC_CHOICES}" +alias _borg_complete_chunker_params "echo {CHUNKER_PARAMS_CHOICES}" +alias _borg_complete_relative_time "echo {RELATIVE_TIME_CHOICES}" +alias _borg_complete_file_size "echo {FILE_SIZE_CHOICES}" + +# Complete archive names (archive IDs when the current token starts with "aid:") and tags. +# These need the command line (for --repo/-r) and some logic, which tcsh cannot do itself: +# it has no functions, and an alias cannot use backquotes here because the completion rule +# calling the alias is backquoted already. So the work is done by a POSIX sh script, kept in +# a variable (a single-quoted csh string, hence no single quotes in it) and run via "sh -c". +set _borg_sh_complete = '{SH_COMPLETE}' + +alias _borg_complete_archive 'sh -c "$_borg_sh_complete" borg-completion archive "$COMMAND_LINE"' +alias _borg_complete_tags 'sh -c "$_borg_sh_complete" borg-completion tags "$COMMAND_LINE"' +""" + +# the sh script the tcsh preamble runs, as one line (`sh -c + +.. only:: latex + + PATH + paths to find; patterns are supported + + + options + --format FORMAT specify format for file listing (default: "{archiveid:.8} {archivename} {mode} {user:6} {group:6} {size:8} {mtime} {path}{extra}{NL}") + --json-lines Format output as JSON Lines. The form of ``--format`` is ignored, but keys used in it are added to the JSON output. Some keys are always present. Note: JSON can only represent text. + + + :ref:`common_options` + | + + Archive filters + -a PATTERN, --match-archives PATTERN only consider archives matching all patterns. See "borg help match-archives". + --sort-by KEYS Comma-separated list of sorting keys; valid keys are: timestamp, archive, name, id, tags, host, user; default is: timestamp + --first N consider the first N archives after other filters are applied + --last N consider the last N archives after other filters are applied + --oldest TIMESPAN consider archives between the oldest archive's timestamp and (oldest + TIMESPAN), e.g., 7d or 12m. + --newest TIMESPAN consider archives between the newest archive's timestamp and (newest - TIMESPAN), e.g., 7d or 12m. + --older TIMESPAN consider archives older than (now - TIMESPAN), e.g., 7d or 12m. + --newer TIMESPAN consider archives newer than (now - TIMESPAN), e.g., 7d or 12m. + + + Include/Exclude options + -e PATTERN, --exclude PATTERN exclude paths matching PATTERN + --exclude-from EXCLUDEFILE read exclude patterns from EXCLUDEFILE, one per line + --pattern PATTERN include/exclude paths matching PATTERN + --patterns-from PATTERNFILE read include/exclude patterns from PATTERNFILE, one per line + + +Description +~~~~~~~~~~~ + +This command finds files matching the given paths or patterns in the archives +selected by the usual archive filter options (all archives, if no filters are +given). It iterates over the matching archives from newest to oldest, over all +items of each archive, and outputs one line per match, like ``borg list``, but +prefixed with a short archive ID and the archive name. As the archives of a +series all share the same name, only the archive ID uniquely identifies the +archive. + +This makes it easy to answer questions like "which archives contain this file?" +or "where did that file end up?":: + + $ borg find home/user/file.txt # in which archives is this file? + $ borg find 'sh:**/*.jpg' --last 3 # all jpg files in the last 3 archives + +The given PATHs match like in ``borg list`` or ``borg extract``: a plain path +matches the item with that path as well as everything below it, and the pattern +styles (``fm:``, ``sh:``, ``re:``, ``pp:``, ``pf:``) are supported as well. +For more help on include/exclude patterns, see the output of :ref:`borg_patterns`. + +Note: there is no extra index for the file paths, so this command reads the +metadata of all selected archives, which may take a while for many/big archives. + +.. man NOTES + +The FORMAT specifier syntax ++++++++++++++++++++++++++++ + +The ``--format`` option uses Python's `format string syntax +`_. + +Examples: +:: + + # only print the archive and the path, nothing else + $ borg find --format '{archiveid:.8} {archivename} {path}{NL}' 'sh:**/*.jpg' + 20e70e3a photos photos/paris/eiffel.jpg + ... + +{archiveid:.8} prints a short archive ID (use {archiveid} for the full ID), +{archivename} the archive name (the archives of a series all share the name). + +The following keys are always available: +- NEWLINE: OS dependent line separator +- NL: alias of NEWLINE +- NUL: NUL character for creating print0 / xargs -0 like output +- SPACE: space character +- TAB: tab character +- CR: carriage return character +- LF: line feed character + + +Keys available only when finding files in an archive: + +- type: file type (file, dir, symlink, ...) +- mode: file mode (as in stat) +- uid: user id of file owner +- gid: group id of file owner +- user: user name of file owner +- group: group name of file owner +- path: file path +- target: link target for symlinks +- hlid: hard link identity (same if hardlinking same fs object) +- inode: inode number +- flags: file flags + +- size: file size +- num_chunks: number of chunks in this file + +- mtime: file modification time +- ctime: file change time +- atime: file access time +- isomtime: file modification time (ISO 8601 format) +- isoctime: file change time (ISO 8601 format) +- isoatime: file access time (ISO 8601 format) + +- fingerprint: Fingerprint of the file content (may have false negatives), format: H(conditions)-H(chunk_ids) +- blake2b +- blake2s +- blake3 +- md5 +- sha1 +- sha224 +- sha256 +- sha384 +- sha3_224 +- sha3_256 +- sha3_384 +- sha3_512 +- sha512 + +- archiveid: internal ID of the archive +- archivename: name of the archive +- extra: prepends {target} with " -> " for soft links and " link to " for hard links diff --git a/docs/usage/general/environment.rst.inc b/docs/usage/general/environment.rst.inc index d9ac6db6db..1f4fcdcc97 100644 --- a/docs/usage/general/environment.rst.inc +++ b/docs/usage/general/environment.rst.inc @@ -288,6 +288,8 @@ General: Output formatting: BORG_CHECK_FORMAT Giving the default value for ``borg check --format=X``. + BORG_FIND_FORMAT + Giving the default value for ``borg find --format=X``. BORG_LIST_FORMAT Giving the default value for ``borg list --format=X``. BORG_REPO_LIST_FORMAT diff --git a/src/borg/archiver/__init__.py b/src/borg/archiver/__init__.py index e3a35c99be..c8730977eb 100644 --- a/src/borg/archiver/__init__.py +++ b/src/borg/archiver/__init__.py @@ -74,6 +74,7 @@ from .delete_cmd import DeleteMixIn from .diff_cmd import DiffMixIn from .extract_cmd import ExtractMixIn +from .find_cmd import FindMixIn from .help_cmd import HelpMixIn from .info_cmd import InfoMixIn from .key_cmds import KeysMixIn @@ -109,6 +110,7 @@ class Archiver( DeleteMixIn, DiffMixIn, ExtractMixIn, + FindMixIn, HelpMixIn, InfoMixIn, KeysMixIn, @@ -301,6 +303,7 @@ def build_parser(self): self.build_parser_delete(subparsers, common_parser, mid_common_parser) self.build_parser_diff(subparsers, common_parser, mid_common_parser) self.build_parser_extract(subparsers, common_parser, mid_common_parser) + self.build_parser_find(subparsers, common_parser, mid_common_parser) self.build_parser_help(subparsers, common_parser, mid_common_parser, parser) self.build_parser_info(subparsers, common_parser, mid_common_parser) self.build_parser_keys(subparsers, common_parser, mid_common_parser) diff --git a/src/borg/archiver/find_cmd.py b/src/borg/archiver/find_cmd.py new file mode 100644 index 0000000000..7432c7b284 --- /dev/null +++ b/src/borg/archiver/find_cmd.py @@ -0,0 +1,134 @@ +import os +import textwrap +import sys + +from ._common import with_repository, build_matcher, define_archive_filters_group, Highlander +from ..archive import Archive +from ..cache import Cache +from ..constants import * # NOQA +from ..helpers import ItemFormatter, BaseFormatter, PathSpec +from ..helpers.argparsing import ArgumentParser +from ..manifest import Manifest + +from ..logger import create_logger + +logger = create_logger() + + +class FindMixIn: + @with_repository(compatibility=(Manifest.Operation.READ,)) + def do_find(self, args, repository, manifest): + """Find files across archives.""" + matcher = build_matcher(args.patterns, args.paths) + if args.format is not None: + format = args.format + else: + format = os.environ.get( + "BORG_FIND_FORMAT", + "{archiveid:.8} {archivename} {mode} {user:6} {group:6} {size:8} {mtime} {path}{extra}{NL}", + ) + # check the format before doing any work with it (also: the ItemFormatter is only built later) + ItemFormatter.validate_format(format) + + archive_infos = manifest.archives.list_considering(args, reverse=True) + num_archives = len(archive_infos) + + def _find_inner(cache): + for i, info in enumerate(archive_infos): + logger.info(f"Searching archive {info.name} {info.ts.astimezone()} ({i + 1}/{num_archives})") + archive = Archive(manifest, info.id, cache=cache) + formatter = ItemFormatter(archive, format) + for item in archive.iter_items(lambda item: matcher.match(item.path)): + sys.stdout.write(formatter.format_item(item, args.json_lines, sort=True)) + + # Only load the cache if it will be used + if ItemFormatter.format_needs_cache(format): + with Cache(repository, manifest) as cache: + _find_inner(cache) + else: + _find_inner(cache=None) + + def build_parser_find(self, subparsers, common_parser, mid_common_parser): + from ._common import process_epilog, define_exclusion_group + + find_epilog = ( + process_epilog( + """ + This command finds files matching the given paths or patterns in the archives + selected by the usual archive filter options (all archives, if no filters are + given). It iterates over the matching archives from newest to oldest, over all + items of each archive, and outputs one line per match, like ``borg list``, but + prefixed with a short archive ID and the archive name. As the archives of a + series all share the same name, only the archive ID uniquely identifies the + archive. + + This makes it easy to answer questions like "which archives contain this file?" + or "where did that file end up?":: + + $ borg find home/user/file.txt # in which archives is this file? + $ borg find 'sh:**/*.jpg' --last 3 # all jpg files in the last 3 archives + + The given PATHs match like in ``borg list`` or ``borg extract``: a plain path + matches the item with that path as well as everything below it, and the pattern + styles (``fm:``, ``sh:``, ``re:``, ``pp:``, ``pf:``) are supported as well. + For more help on include/exclude patterns, see the output of :ref:`borg_patterns`. + + Note: there is no extra index for the file paths, so this command reads the + metadata of all selected archives, which may take a while for many/big archives. + + .. man NOTES + + The FORMAT specifier syntax + +++++++++++++++++++++++++++ + + The ``--format`` option uses Python's `format string syntax + `_. + + Examples: + :: + + # only print the archive and the path, nothing else + $ borg find --format '{archiveid:.8} {archivename} {path}{NL}' 'sh:**/*.jpg' + 20e70e3a photos photos/paris/eiffel.jpg + ... + + {archiveid:.8} prints a short archive ID (use {archiveid} for the full ID), + {archivename} the archive name (the archives of a series all share the name). + + The following keys are always available: + + """ + ) + + BaseFormatter.keys_help() + + textwrap.dedent( + """ + + Keys available only when finding files in an archive: + + """ + ) + + ItemFormatter.keys_help() + ) + subparser = ArgumentParser(parents=[common_parser], description=self.do_find.__doc__, epilog=find_epilog) + subparsers.add_subcommand("find", subparser, help="find files across archives") + subparser.add_argument( + "--format", + metavar="FORMAT", + dest="format", + action=Highlander, + help="specify format for file listing (default: " + '"{archiveid:.8} {archivename} {mode} {user:6} {group:6} {size:8} {mtime} {path}{extra}{NL}")', + ) + subparser.add_argument( + "--json-lines", + action="store_true", + help="Format output as JSON Lines. " + "The form of ``--format`` is ignored, " + "but keys used in it are added to the JSON output. " + "Some keys are always present. Note: JSON can only represent text.", + ) + subparser.add_argument( + "paths", metavar="PATH", nargs="*", type=PathSpec, help="paths to find; patterns are supported" + ) + define_archive_filters_group(subparser) + define_exclusion_group(subparser) diff --git a/src/borg/archiver/help_cmd.py b/src/borg/archiver/help_cmd.py index 1b1ac4d564..ce34b6c06a 100644 --- a/src/borg/archiver/help_cmd.py +++ b/src/borg/archiver/help_cmd.py @@ -870,6 +870,8 @@ class HelpMixIn: Output formatting: BORG_CHECK_FORMAT Giving the default value for ``borg check --format=X``. + BORG_FIND_FORMAT + Giving the default value for ``borg find --format=X``. BORG_LIST_FORMAT Giving the default value for ``borg list --format=X``. BORG_REPO_LIST_FORMAT diff --git a/src/borg/manifest.py b/src/borg/manifest.py index 4881cf4780..d25070d8f4 100644 --- a/src/borg/manifest.py +++ b/src/borg/manifest.py @@ -120,7 +120,7 @@ def list( newest=None, deleted=False, ): ... - def list_considering(self, args): ... + def list_considering(self, args, *, reverse=False): ... def get_one(self, match, *, match_end=r"\Z", deleted=False): ... @@ -413,7 +413,7 @@ def list( archive_infos.reverse() return archive_infos - def list_considering(self, args): + def list_considering(self, args, *, reverse=False): """ get a list of archives, considering --first/last/prefix/match-archives/sort cmdline args """ @@ -424,6 +424,7 @@ def list_considering(self, args): ) return self.list( sort_by=args.sort_by.split(","), + reverse=reverse, match=args.match_archives, first=getattr(args, "first", None), last=getattr(args, "last", None), diff --git a/src/borg/testsuite/archiver/find_cmd_test.py b/src/borg/testsuite/archiver/find_cmd_test.py new file mode 100644 index 0000000000..b6ff983706 --- /dev/null +++ b/src/borg/testsuite/archiver/find_cmd_test.py @@ -0,0 +1,132 @@ +import json + +from ...constants import * # NOQA +from . import cmd, create_regular_file, generate_archiver_tests, RK_ENCRYPTION + +pytest_generate_tests = lambda metafunc: generate_archiver_tests(metafunc, kinds="local,binary") # NOQA + +# terse format for tests that only care about which archives/items matched +FMT = "{archiveid:.8} {archivename} {path}{NL}" + + +def short_ids(archiver): + """map archive name -> short (8 hex digits) archive id, as printed by the default find output""" + output = cmd(archiver, "repo-list", "--json") + return {archive["name"]: archive["id"][:8] for archive in json.loads(output)["archives"]} + + +def test_find_basic(archivers, request): + archiver = request.getfixturevalue(archivers) + cmd(archiver, "repo-create", RK_ENCRYPTION) + create_regular_file(archiver.input_path, "file1", size=1024) + cmd(archiver, "create", "archive1", "input") + create_regular_file(archiver.input_path, "file2", size=1024) + cmd(archiver, "create", "archive2", "input") + ids = short_ids(archiver) + + # file1 is in both archives, output ordered from newest to oldest archive. + output = cmd(archiver, "find", "input/file1", "--format", FMT) + assert output.splitlines() == [f"{ids['archive2']} archive2 input/file1", f"{ids['archive1']} archive1 input/file1"] + + # file2 is only in the second archive. + output = cmd(archiver, "find", "input/file2", "--format", FMT) + assert output.splitlines() == [f"{ids['archive2']} archive2 input/file2"] + + # a path matches itself and everything below it. + output = cmd(archiver, "find", "input", "--format", FMT) + lines = output.splitlines() + # ordered by archive; within one archive, the items come in the order stored in the archive. + assert lines[0] == f"{ids['archive2']} archive2 input" + assert set(lines[1:3]) == {f"{ids['archive2']} archive2 input/file1", f"{ids['archive2']} archive2 input/file2"} + assert lines[3:] == [f"{ids['archive1']} archive1 input", f"{ids['archive1']} archive1 input/file1"] + + # without any PATH, all items of all archives match. + assert cmd(archiver, "find", "--format", FMT) == output + + # no match: no output. + output = cmd(archiver, "find", "input/does-not-exist", "--format", FMT) + assert output == "" + + +def test_find_patterns(archivers, request): + archiver = request.getfixturevalue(archivers) + cmd(archiver, "repo-create", RK_ENCRYPTION) + create_regular_file(archiver.input_path, "dir1/file.jpg", size=1) + create_regular_file(archiver.input_path, "dir2/file.txt", size=1) + cmd(archiver, "create", "archive1", "input") + ids = short_ids(archiver) + + output = cmd(archiver, "find", "sh:**/*.jpg", "--format", FMT) + assert output.splitlines() == [f"{ids['archive1']} archive1 input/dir1/file.jpg"] + + output = cmd(archiver, "find", "fm:*.txt", "--format", FMT) + assert output.splitlines() == [f"{ids['archive1']} archive1 input/dir2/file.txt"] + + output = cmd(archiver, "find", "input", "--exclude", "sh:**/*.jpg", "--format", FMT) + assert "file.jpg" not in output + assert f"{ids['archive1']} archive1 input/dir2/file.txt" in output + + output = cmd(archiver, "find", "--pattern", "+ re:\\.jpg$", "--pattern", "- re:^.*$", "--format", FMT) + assert output.splitlines() == [f"{ids['archive1']} archive1 input/dir1/file.jpg"] + + +def test_find_archive_filters(archivers, request): + archiver = request.getfixturevalue(archivers) + cmd(archiver, "repo-create", RK_ENCRYPTION) + create_regular_file(archiver.input_path, "file1", size=1024) + cmd(archiver, "create", "archive1", "input") + cmd(archiver, "create", "archive2", "input") + ids = short_ids(archiver) + + output = cmd(archiver, "find", "input/file1", "--last", "1", "--format", FMT) + assert output.splitlines() == [f"{ids['archive2']} archive2 input/file1"] + + output = cmd(archiver, "find", "input/file1", "--first", "1", "--format", FMT) + assert output.splitlines() == [f"{ids['archive1']} archive1 input/file1"] + + output = cmd(archiver, "find", "input/file1", "-a", "archive1", "--format", FMT) + assert output.splitlines() == [f"{ids['archive1']} archive1 input/file1"] + + +def test_find_default_format(archivers, request): + archiver = request.getfixturevalue(archivers) + cmd(archiver, "repo-create", RK_ENCRYPTION) + create_regular_file(archiver.input_path, "file1", size=1024) + cmd(archiver, "create", "archive1", "input") + cmd(archiver, "create", "archive2", "input") + ids = short_ids(archiver) + + # the default format is the borg list default format with "{archiveid:.8} {archivename} " prepended. + find_output = cmd(archiver, "find", "input/file1") + expected = [] + for name in ("archive2", "archive1"): # newest to oldest + for line in cmd(archiver, "list", name, "input/file1").splitlines(): + expected.append(f"{ids[name]} {name} {line}") + assert find_output.splitlines() == expected + + +def test_find_format(archivers, request): + archiver = request.getfixturevalue(archivers) + cmd(archiver, "repo-create", RK_ENCRYPTION) + create_regular_file(archiver.input_path, "file1", size=1024) + cmd(archiver, "create", "archive1", "input") + + output = cmd(archiver, "find", "input/file1", "--format", "{size} {path}{NL}") + assert output.splitlines() == ["1024 input/file1"] + + +def test_find_json_lines(archivers, request): + archiver = request.getfixturevalue(archivers) + cmd(archiver, "repo-create", RK_ENCRYPTION) + create_regular_file(archiver.input_path, "file1", size=1024) + cmd(archiver, "create", "archive1", "input") + cmd(archiver, "create", "archive2", "input") + ids = short_ids(archiver) + + output = cmd(archiver, "find", "input/file1", "--json-lines") + rows = [json.loads(line) for line in output.splitlines()] + # the default format contains {archiveid} and {archivename}, so they are present in the JSON output. + assert [(row["archiveid"][:8], row["archivename"], row["path"]) for row in rows] == [ + (ids["archive2"], "archive2", "input/file1"), + (ids["archive1"], "archive1", "input/file1"), + ] From 448f1c6c9db600772c653db66847f7775dd30acf Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Fri, 14 Aug 2026 15:55:56 +0200 Subject: [PATCH 121/144] create: support --stats with --dry-run, fixes #1648 A dry run already walks the files and stats them, so it can cheaply count the files and sum up their sizes from file system metadata, without reading any file contents. --stats (and --json) with --dry-run now report the number of files and the original size. The deduplicated size is unknown in a dry run, as data is not read, chunked, and deduplicated. The sizes of data read from stdin, from a command's output, and from special files are also unknown in a dry run and counted as zero. Hardlinks count once per link, as in a real create. Co-Authored-By: Claude Fable 5 --- docs/internals/frontends.rst | 4 ++ docs/usage/create.rst.inc | 10 +++- src/borg/archiver/__init__.py | 3 +- src/borg/archiver/create_cmd.py | 53 ++++++++++++++++--- .../testsuite/archiver/create_cmd_test.py | 33 ++++++++++++ 5 files changed, 94 insertions(+), 9 deletions(-) diff --git a/docs/internals/frontends.rst b/docs/internals/frontends.rst index 6af8114b2d..5cec57a0f6 100644 --- a/docs/internals/frontends.rst +++ b/docs/internals/frontends.rst @@ -356,6 +356,10 @@ Archive formats array under the *archives* key, while :ref:`borg_create` returns a single archive object under the *archive* key. +:ref:`borg_create` with ``--dry-run`` does not create an archive, so there is no *archive* key. +Instead, it returns *dry_run* (true) and a reduced *stats* object with *nfiles* and *original_size*, +both computed from file system metadata without reading the file contents. + Both formats contain a *name* key with the archive name, the *id* key with the hexadecimal archive ID, and the *start* key with the start timestamp. diff --git a/docs/usage/create.rst.inc b/docs/usage/create.rst.inc index b914e9f204..f3a5f7b4a5 100644 --- a/docs/usage/create.rst.inc +++ b/docs/usage/create.rst.inc @@ -274,8 +274,14 @@ does not go to a terminal (e.g. into a logfile), the precise format is always us When using ``--stats``, you will get some statistics about how much data was added - the "This Archive" deduplicated size there is most interesting as that is how much your repository will grow. Please note that the "All archives" stats refer to -the state after creation. Also, the ``--stats`` and ``--dry-run`` options are mutually -exclusive because the data is not actually compressed and deduplicated during a dry run. +the state after creation. + +When ``--stats`` is used together with ``--dry-run``, only the number of files and the +original size are reported. They are computed from file system metadata, without reading +the file contents, so a dry run stays fast. As data is not actually read, chunked, and +deduplicated during a dry run, the deduplicated size is unknown. The sizes of data read +from standard input, from a command's output, or from special files (``--read-special``) +are also unknown in a dry run and counted as zero. The ``--stats`` output also reports the store statistics (lines prefixed with "Store"), taken from the storage layer after this run. These cover the backend and diff --git a/src/borg/archiver/__init__.py b/src/borg/archiver/__init__.py index e3a35c99be..92d96d7fbd 100644 --- a/src/borg/archiver/__init__.py +++ b/src/borg/archiver/__init__.py @@ -473,7 +473,8 @@ def run(self, args): self._setup_implied_logging(vars(args)) self._setup_topic_debugging(args) # extract --dry-run reads, decrypts and decompresses every object, so its store stats are accurate. - stats_supported_with_dry_run = func_name == "do_extract" + # create --dry-run counts the files and sums up their sizes from file system metadata. + stats_supported_with_dry_run = func_name in ("do_extract", "do_create") if getattr(args, "stats", False) and getattr(args, "dry_run", False) and not stats_supported_with_dry_run: logger.warning("Ignoring --stats. It is not supported when using --dry-run.") args.stats = False diff --git a/src/borg/archiver/create_cmd.py b/src/borg/archiver/create_cmd.py index 1727ca14d2..c53c6e149d 100644 --- a/src/borg/archiver/create_cmd.py +++ b/src/borg/archiver/create_cmd.py @@ -10,7 +10,7 @@ from ._common import with_repository, Highlander from .. import helpers -from ..archive import Archive, is_special, SF_DATALESS +from ..archive import Archive, Statistics, is_special, SF_DATALESS from ..archive import BackupError, BackupOSError, BackupItemExcluded, backup_io, OsOpen, stat_update_check from ..archive import FilesystemObjectProcessors, MetadataCollector, ChunksProcessor from ..cache import Cache @@ -22,7 +22,7 @@ from ..helpers import get_cache_dir, os_stat, get_strip_prefix, slashify from ..helpers import dir_is_tagged from ..helpers import log_multi -from ..helpers import basic_json_data, json_print +from ..helpers import basic_json_data, json_print, FileSize from ..helpers import flags_dir, flags_special_follow, flags_special from ..helpers import prepare_subprocess_env from ..helpers import sig_int, ignore_sigint @@ -97,6 +97,7 @@ def create_inner(archive, cache, fso): raise Error(f"{path!r}: {e}") else: status = "+" # included + self.dry_run_stats.nfiles += 1 # size unknown without running the command self.print_file_status(status, path) elif args.paths_from_command or args.paths_from_shell_command or args.paths_from_stdin: paths_sep = eval_escapes(args.paths_delimiter) if args.paths_delimiter is not None else "\n" @@ -174,6 +175,7 @@ def create_inner(archive, cache, fso): status = "E" else: status = "+" # included + self.dry_run_stats.nfiles += 1 # size unknown without reading stdin self.print_file_status(status, path) if not dry_run and status is not None: fso.stats.files_stats[status] += 1 @@ -226,6 +228,7 @@ def create_inner(archive, cache, fso): self.noxattrs = args.noxattrs self.exclude_dataless = args.exclude_dataless dry_run = args.dry_run + self.dry_run_stats = Statistics() if dry_run else None self.start_backup = time.time_ns() t0 = archive_ts_now() logger.info('Creating archive "%s" in repository %s' % (args.name, args.location.processed)) @@ -288,6 +291,24 @@ def create_inner(archive, cache, fso): log_multi(str(archive), str(archive.stats), logger=logging.getLogger("borg.output.stats")) else: create_inner(None, None, None) + args.stats |= args.json + if args.stats: + stats = self.dry_run_stats + if args.json: + json_data = basic_json_data( + manifest, + extra={ + "dry_run": True, + "stats": {"nfiles": stats.nfiles, "original_size": FileSize(stats.osize)}, + }, + ) + json_print(json_data) + else: + log_multi( + f"Number of files: {stats.nfiles}", + f"Original size: {stats.osize_fmt}", + logger=logging.getLogger("borg.output.stats"), + ) def _process_any(self, *, path, parent_fd, name, st, fso, cache, read_special, dry_run, strip_prefix): """ @@ -295,6 +316,22 @@ def _process_any(self, *, path, parent_fd, name, st, fso, cache, read_special, d """ if dry_run: + stats = self.dry_run_stats + if stat.S_ISREG(st.st_mode): + stats.nfiles += 1 + stats.osize += st.st_size + elif read_special: + if stat.S_ISLNK(st.st_mode): + try: + st_target = os_stat(path=path, parent_fd=parent_fd, name=name, follow_symlinks=True) + except OSError: + special = False + else: + special = is_special(st_target.st_mode) + else: + special = is_special(st.st_mode) + if special: + stats.nfiles += 1 # size unknown without reading the special file return "+" # included MAX_RETRIES = 10 # count includes the initial try (initial try == "retry 0") for retry in range(MAX_RETRIES): @@ -673,8 +710,14 @@ def build_parser_create(self, subparsers, common_parser, mid_common_parser): When using ``--stats``, you will get some statistics about how much data was added - the "This Archive" deduplicated size there is most interesting as that is how much your repository will grow. Please note that the "All archives" stats refer to - the state after creation. Also, the ``--stats`` and ``--dry-run`` options are mutually - exclusive because the data is not actually compressed and deduplicated during a dry run. + the state after creation. + + When ``--stats`` is used together with ``--dry-run``, only the number of files and the + original size are reported. They are computed from file system metadata, without reading + the file contents, so a dry run stays fast. As data is not actually read, chunked, and + deduplicated during a dry run, the deduplicated size is unknown. The sizes of data read + from standard input, from a command's output, or from special files (``--read-special``) + are also unknown in a dry run and counted as zero. The ``--stats`` output also reports the store statistics (lines prefixed with "Store"), taken from the storage layer after this run. These cover the backend and @@ -834,8 +877,6 @@ def build_parser_create(self, subparsers, common_parser, mid_common_parser): subparser = ArgumentParser(parents=[common_parser], description=self.do_create.__doc__, epilog=create_epilog) subparsers.add_subcommand("create", subparser, help="create a backup") - # note: --dry-run and --stats are mutually exclusive, but we do not want to abort when - # parsing, but rather proceed with the dry-run, but without stats (see run() method). subparser.add_argument( "-n", "--dry-run", dest="dry_run", action="store_true", help="do not create a backup archive" ) diff --git a/src/borg/testsuite/archiver/create_cmd_test.py b/src/borg/testsuite/archiver/create_cmd_test.py index 15c2f064e8..b3a51ec785 100644 --- a/src/borg/testsuite/archiver/create_cmd_test.py +++ b/src/borg/testsuite/archiver/create_cmd_test.py @@ -687,6 +687,39 @@ def test_create_dry_run(archivers, request): assert manifest.archives.count() == 0 +def test_create_dry_run_stats(archivers, request): + archiver = request.getfixturevalue(archivers) + create_regular_file(archiver.input_path, "file1", size=1024 * 80) + create_regular_file(archiver.input_path, "file2", size=1024 * 20) + expected_nfiles = 2 + if are_hardlinks_supported(): + os.link(os.path.join(archiver.input_path, "file1"), os.path.join(archiver.input_path, "hardlink1")) + expected_nfiles = 3 # as in a real create, each hardlink counts as a file + cmd(archiver, "repo-create", RK_ENCRYPTION) + output = cmd(archiver, "create", "--dry-run", "--stats", "test", "input") + assert f"Number of files: {expected_nfiles}" in output + assert "Original size:" in output + assert "Deduplicated size:" not in output + # Make sure no archive has been created + with Repository(archiver.repository_path) as repository: + manifest = Manifest.load(repository, Manifest.NO_OPERATION_CHECK) + assert manifest.archives.count() == 0 + + +def test_create_dry_run_json(archivers, request): + archiver = request.getfixturevalue(archivers) + create_regular_file(archiver.input_path, "file1", size=1024 * 80) + create_regular_file(archiver.input_path, "file2", size=1024 * 20) + cmd(archiver, "repo-create", RK_ENCRYPTION) + output = cmd(archiver, "create", "--dry-run", "--json", "test", "input") + result = json.loads(output) + assert result["dry_run"] is True + assert result["stats"]["nfiles"] == 2 + assert result["stats"]["original_size"] == 1024 * 100 + assert "archive" not in result + assert "repository" in result + + def test_progress_on(archivers, request): archiver = request.getfixturevalue(archivers) create_regular_file(archiver.input_path, "file1", size=1024 * 80) From b893d49154b3b1b349cec2f920677ada8a5a88b9 Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Fri, 14 Aug 2026 16:04:07 +0200 Subject: [PATCH 122/144] chunkers: release the GIL in the AES chunkers' scan kernels ra_scan/gl_scan/tp_scan are pure C (OpenSSL EVP or AES hw intrinsics, no Python API calls), but were called with the GIL held, so a scan span - up to the 8 MiB max chunk size - blocked all other threads, e.g. the PackWriter store-thread between its GIL-releasing parts. Wrap the calls in `with nogil` (and declare the externs nogil), the same pattern fastcdc/buzhash/buzhash64 already use (#10014). Chunker output and single-thread performance are unchanged. Two threads chunking concurrently (rabin-aes, evp kernel, arm64): the wall-clock ratio vs one thread drops from 2.02x (fully serialized) to 1.13x (parallel). Co-Authored-By: Claude Fable 5 --- src/borg/chunkers/goldilocks_aes.pyx | 7 +++++-- src/borg/chunkers/rabin_aes.pyx | 7 +++++-- src/borg/chunkers/toeplitz_aes.pyx | 7 +++++-- 3 files changed, 15 insertions(+), 6 deletions(-) diff --git a/src/borg/chunkers/goldilocks_aes.pyx b/src/borg/chunkers/goldilocks_aes.pyx index 4c86a118c0..2e8d1add53 100644 --- a/src/borg/chunkers/goldilocks_aes.pyx +++ b/src/borg/chunkers/goldilocks_aes.pyx @@ -53,7 +53,7 @@ cdef extern from "goldilocks_aes_impl.h": void gl_free(GL_CTX *ctx) const char *gl_kind(const GL_CTX *ctx) uint64_t gl_digest64(const GL_CTX *ctx, const uint8_t *q) - int64_t gl_scan(GL_CTX *ctx, const uint8_t *p, size_t n, uint64_t *digest, uint64_t mask) + int64_t gl_scan(GL_CTX *ctx, const uint8_t *p, size_t n, uint64_t *digest, uint64_t mask) nogil # --- Goldilocks field helpers (pure Python, used at chunker setup only) --- @@ -165,7 +165,10 @@ cdef class ChunkerGoldilocksAES(ChunkerPHTE): self.ctx = NULL cdef int64_t _scan(self, const uint8_t *p, size_t n, uint64_t *digest, uint64_t mask) noexcept: - return gl_scan(self.ctx, p, n, digest, mask) + cdef int64_t r + with nogil: + r = gl_scan(self.ctx, p, n, digest, mask) + return r cdef uint64_t _digest64(self, const uint8_t *q) noexcept: return gl_digest64(self.ctx, q) diff --git a/src/borg/chunkers/rabin_aes.pyx b/src/borg/chunkers/rabin_aes.pyx index 953d73c444..7a310c6c67 100644 --- a/src/borg/chunkers/rabin_aes.pyx +++ b/src/borg/chunkers/rabin_aes.pyx @@ -64,7 +64,7 @@ cdef extern from "rabin_aes_impl.h": void ra_free(RA_CTX *ctx) const char *ra_kind(const RA_CTX *ctx) uint64_t ra_digest64(const RA_CTX *ctx, const uint8_t *q) - int64_t ra_scan(RA_CTX *ctx, const uint8_t *p, size_t n, uint64_t *digest, uint64_t mask) + int64_t ra_scan(RA_CTX *ctx, const uint8_t *p, size_t n, uint64_t *digest, uint64_t mask) nogil # --- GF(2)[x] polynomial helpers (pure Python, used at chunker setup only) --- @@ -234,7 +234,10 @@ cdef class ChunkerRabinAES(ChunkerPHTE): self.ctx = NULL cdef int64_t _scan(self, const uint8_t *p, size_t n, uint64_t *digest, uint64_t mask) noexcept: - return ra_scan(self.ctx, p, n, digest, mask) + cdef int64_t r + with nogil: + r = ra_scan(self.ctx, p, n, digest, mask) + return r cdef uint64_t _digest64(self, const uint8_t *q) noexcept: return ra_digest64(self.ctx, q) diff --git a/src/borg/chunkers/toeplitz_aes.pyx b/src/borg/chunkers/toeplitz_aes.pyx index f9fc7cbb44..084663b4b8 100644 --- a/src/borg/chunkers/toeplitz_aes.pyx +++ b/src/borg/chunkers/toeplitz_aes.pyx @@ -57,7 +57,7 @@ cdef extern from "toeplitz_aes_impl.h": void tp_free(TP_CTX *ctx) const char *tp_kind(const TP_CTX *ctx) uint64_t tp_digest64(const TP_CTX *ctx, const uint8_t *q) - int64_t tp_scan(TP_CTX *ctx, const uint8_t *p, size_t n, uint64_t *digest, uint64_t mask) + int64_t tp_scan(TP_CTX *ctx, const uint8_t *p, size_t n, uint64_t *digest, uint64_t mask) nogil # --- GF(2)[x] helpers (pure Python, used at chunker setup only) ---------- @@ -166,7 +166,10 @@ cdef class ChunkerToeplitzAES(ChunkerPHTE): self.ctx = NULL cdef int64_t _scan(self, const uint8_t *p, size_t n, uint64_t *digest, uint64_t mask) noexcept: - return tp_scan(self.ctx, p, n, digest, mask) + cdef int64_t r + with nogil: + r = tp_scan(self.ctx, p, n, digest, mask) + return r cdef uint64_t _digest64(self, const uint8_t *q) noexcept: return tp_digest64(self.ctx, q) From 9cc37123fd4892c02059ad301749174ed2537fe4 Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Fri, 14 Aug 2026 16:20:26 +0200 Subject: [PATCH 123/144] docs: NetBSD xattr support is implemented, update platform feature table, #1332 Co-Authored-By: Claude Fable 5 --- docs/usage/general/file-metadata.rst.inc | 11 +++++------ 1 file changed, 5 insertions(+), 6 deletions(-) diff --git a/docs/usage/general/file-metadata.rst.inc b/docs/usage/general/file-metadata.rst.inc index a01d72d3b6..5bf4fba63c 100644 --- a/docs/usage/general/file-metadata.rst.inc +++ b/docs/usage/general/file-metadata.rst.inc @@ -35,11 +35,11 @@ On some platforms additional features are supported: +-------------------------+----------+-----------+------------+ | OpenBSD | n/a | n/a | Yes (all) | +-------------------------+----------+-----------+------------+ -| NetBSD | n/a | No [2]_ | Yes (all) | +| NetBSD | n/a | Yes | Yes (all) | +-------------------------+----------+-----------+------------+ -| Solaris and derivatives | No [3]_ | No [3]_ | n/a | +| Solaris and derivatives | No [2]_ | No [2]_ | n/a | +-------------------------+----------+-----------+------------+ -| Windows (cygwin) | No [4]_ | No | No | +| Windows (cygwin) | No [3]_ | No | No | +-------------------------+----------+-----------+------------+ Other Unix-like operating systems may work as well, but have not been tested yet. @@ -49,9 +49,8 @@ For example, ntfs-3g on Linux is not able to convey NTFS ACLs. .. [1] Only "nodump", "immutable", "compressed" and "append" are supported. Feature request :issue:`618` for more flags. -.. [2] Feature request :issue:`1332` -.. [3] Feature request :issue:`1337` -.. [4] Cygwin tries to map NTFS ACLs to permissions with varying degrees of success. +.. [2] Feature request :issue:`1337` +.. [3] Cygwin tries to map NTFS ACLs to permissions with varying degrees of success. .. [#acls] The native access control list mechanism of the OS. This normally limits access to non-native ACLs. For example, NTFS ACLs are not completely accessible on Linux with ntfs-3g. From 64d622ca5d9e84ae775d6ea513c5d611b7f1c3c5 Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Fri, 14 Aug 2026 16:47:22 +0200 Subject: [PATCH 124/144] illumos/Solaris: add xattr support, see #1337 xattrs on these platforms are regular files inside a hidden attribute directory attached to each file (opened via openat(2) with O_XATTR, see fsattr(7)), so this can be implemented in pure Python: xattr names/values map to the names/contents of the files in the attribute directory. Co-Authored-By: Claude Fable 5 --- .github/workflows/ci.yml | 6 ++ docs/usage/general/file-metadata.rst.inc | 2 +- src/borg/platform/__init__.py | 11 ++- src/borg/platform/solaris.py | 101 +++++++++++++++++++++++ src/borg/platformflags.py | 1 + src/borg/testsuite/xattr_test.py | 16 ++-- src/borg/xattr.py | 2 +- 7 files changed, 131 insertions(+), 8 deletions(-) create mode 100644 src/borg/platform/solaris.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 8de5fb3694..aaa5728f5a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -598,6 +598,12 @@ jobs: export TMPDIR=/var/tmp/borg-ci mkdir -p "$TMPDIR" + # show whether xattrs work on the ZFS-backed TMPDIR (they are files in a + # hidden per-file attribute directory there, listed via runat(1)) + touch "$TMPDIR/testfile" + /usr/bin/runat "$TMPDIR/testfile" ls -a && echo "*** xattrs supported on $TMPDIR ***" + rm "$TMPDIR/testfile" + python3 -m venv .venv . .venv/bin/activate python -V diff --git a/docs/usage/general/file-metadata.rst.inc b/docs/usage/general/file-metadata.rst.inc index 5bf4fba63c..068ee1152b 100644 --- a/docs/usage/general/file-metadata.rst.inc +++ b/docs/usage/general/file-metadata.rst.inc @@ -37,7 +37,7 @@ On some platforms additional features are supported: +-------------------------+----------+-----------+------------+ | NetBSD | n/a | Yes | Yes (all) | +-------------------------+----------+-----------+------------+ -| Solaris and derivatives | No [2]_ | No [2]_ | n/a | +| Solaris and derivatives | No [2]_ | Yes | n/a | +-------------------------+----------+-----------+------------+ | Windows (cygwin) | No [3]_ | No | No | +-------------------------+----------+-----------+------------+ diff --git a/src/borg/platform/__init__.py b/src/borg/platform/__init__.py index 168d4ce705..4ff1b77a07 100644 --- a/src/borg/platform/__init__.py +++ b/src/borg/platform/__init__.py @@ -6,7 +6,7 @@ from types import ModuleType -from ..platformflags import is_win32, is_linux, is_freebsd, is_netbsd, is_darwin, is_cygwin, is_haiku +from ..platformflags import is_win32, is_linux, is_freebsd, is_netbsd, is_darwin, is_cygwin, is_haiku, is_sunos from .base import ENOATTR from .base import SaveFile, sync_dir, fdatasync, safe_fadvise @@ -59,6 +59,15 @@ from .posix import get_errno from .posix import getosusername from . import posix_ug as platform_ug +elif is_sunos: # pragma: sunos only + from .solaris import listxattr, getxattr, setxattr + from .base import acl_get, acl_set + from .base import set_flags, get_flags + from .base import SyncFile + from .posix import process_alive, local_pid_alive + from .posix import get_errno + from .posix import getosusername + from . import posix_ug as platform_ug elif not is_win32: # pragma: posix only # Generic code for all other POSIX OSes from .base import listxattr, getxattr, setxattr diff --git a/src/borg/platform/solaris.py b/src/borg/platform/solaris.py new file mode 100644 index 0000000000..56d8f7ab28 --- /dev/null +++ b/src/borg/platform/solaris.py @@ -0,0 +1,101 @@ +""" +xattr support for illumos / Solaris and derivatives. + +On these platforms, the extended attributes of a file are regular files inside a hidden +attribute directory attached to that file. The attribute directory is opened by giving +O_XATTR to open(2)/openat(2), see fsattr(7) — xattr names/values map to the names/contents +of the files in there. There are no xattr namespaces, so names are used verbatim. +""" + +import errno +import os + +from .base import ENOATTR + +# CPython exposes os.O_XATTR on Solaris-derived platforms; 0x4000 is its value on illumos +# and Oracle Solaris (belt and braces in case the os module does not have it). +O_XATTR = getattr(os, "O_XATTR", 0x4000) + +# "Extended system attributes" maintained by the OS inside every attribute directory +# (e.g. on ZFS) — these are not user-set xattrs, so hide them from listing and refuse +# to write them. +SYSATTR_PREFIX = "SUNWattr_" + +# Attribute files can be arbitrarily large — refuse to read values bigger than this +# (the other platforms' xattr support is limited alike, via their Buffer(limit=2**24)). +XATTR_SIZE_LIMIT = 2**24 + + +def _open_attrdir(path, follow_symlinks): + # Open the hidden attribute directory of *path* (a bytes path or an open file descriptor). + # O_XATTR is only interpreted by openat() when looking up relative to a file descriptor + # referring to the file, so for a path, the file itself must be opened first. + if isinstance(path, int): + return os.open(".", os.O_RDONLY | O_XATTR, dir_fd=path) + flags = os.O_RDONLY | os.O_NONBLOCK # O_NONBLOCK: do not hang on FIFOs + if not follow_symlinks: + flags |= os.O_NOFOLLOW + fd = os.open(path, flags) + try: + return os.open(".", os.O_RDONLY | O_XATTR, dir_fd=fd) + finally: + os.close(fd) + + +def listxattr(path, *, follow_symlinks=False): + try: + dirfd = _open_attrdir(path, follow_symlinks) + except OSError as e: + if e.errno == errno.ELOOP: + # symlinks cannot have extended attributes on this platform + return [] + if e.errno == errno.EINVAL: + # open(2): the filesystem does not support extended attributes + raise OSError(errno.ENOTSUP, os.strerror(errno.ENOTSUP), path) from None + raise + try: + names = os.listdir(dirfd) + finally: + os.close(dirfd) + return [os.fsencode(name) for name in names if not name.startswith(SYSATTR_PREFIX)] + + +def getxattr(path, name, *, follow_symlinks=False): + dirfd = _open_attrdir(path, follow_symlinks) + try: + try: + fd = os.open(name, os.O_RDONLY | os.O_NOFOLLOW, dir_fd=dirfd) + except FileNotFoundError: + # no attribute file with that name -> no such xattr + raise OSError(ENOATTR, os.strerror(ENOATTR), path) from None + try: + if os.fstat(fd).st_size > XATTR_SIZE_LIMIT: + raise OSError(errno.EFBIG, os.strerror(errno.EFBIG), path) + chunks = [] + while chunk := os.read(fd, 2**20): + chunks.append(chunk) + return b"".join(chunks) + finally: + os.close(fd) + finally: + os.close(dirfd) + + +def setxattr(path, name, value, *, follow_symlinks=False): + name_str = os.fsdecode(name) + if not name_str or "/" in name_str or "\0" in name_str or name_str in (".", ".."): + # such a name cannot be the name of an attribute file + raise OSError(errno.EINVAL, os.strerror(errno.EINVAL), path) + if name_str.startswith(SYSATTR_PREFIX): + raise OSError(errno.EPERM, os.strerror(errno.EPERM), path) + dirfd = _open_attrdir(path, follow_symlinks) + try: + fd = os.open(name, os.O_WRONLY | os.O_CREAT | os.O_TRUNC | os.O_NOFOLLOW, mode=0o644, dir_fd=dirfd) + try: + mv = memoryview(value) + while mv: + mv = mv[os.write(fd, mv) :] + finally: + os.close(fd) + finally: + os.close(dirfd) diff --git a/src/borg/platformflags.py b/src/borg/platformflags.py index cc84f7c7ac..c6f98270b7 100644 --- a/src/borg/platformflags.py +++ b/src/borg/platformflags.py @@ -16,6 +16,7 @@ is_openbsd = sys.platform.startswith("openbsd") is_darwin = sys.platform.startswith("darwin") is_haiku = sys.platform.startswith("haiku") +is_sunos = sys.platform.startswith("sunos") # illumos and Solaris # MSYS2 (on Windows) is_msystem = is_win32 and "MSYSTEM" in os.environ diff --git a/src/borg/testsuite/xattr_test.py b/src/borg/testsuite/xattr_test.py index f99aa961c2..4f41588fbc 100644 --- a/src/borg/testsuite/xattr_test.py +++ b/src/borg/testsuite/xattr_test.py @@ -4,7 +4,12 @@ from ..platform.xattr import buffer, split_lstring from ..xattr import is_enabled, getxattr, setxattr, listxattr, XATTR_FAKEROOT -from ..platformflags import is_linux +from ..platformflags import is_linux, is_sunos + +# Whether xattrs can be set on a symlink itself: +# Linux does not allow setting user.* xattrs on symlinks; on illumos/Solaris, +# symlinks cannot have extended attributes at all. +symlink_xattrs = not is_linux and not is_sunos @pytest.fixture() @@ -35,22 +40,22 @@ def test(tempfile_symlink): setxattr(tmp_fn, b"user.foo", b"bar") setxattr(tmp_fd, b"user.bar", b"foo") setxattr(tmp_fn, b"user.empty", b"") - if not is_linux: - # Linux does not allow setting user.* xattrs on symlinks. + if symlink_xattrs: setxattr(tmp_lfn, b"user.linkxattr", b"baz") assert_equal_se(listxattr(tmp_fn), [b"user.foo", b"user.bar", b"user.empty"]) assert_equal_se(listxattr(tmp_fd), [b"user.foo", b"user.bar", b"user.empty"]) assert_equal_se(listxattr(tmp_lfn, follow_symlinks=True), [b"user.foo", b"user.bar", b"user.empty"]) - if not is_linux: + if symlink_xattrs: assert_equal_se(listxattr(tmp_lfn), [b"user.linkxattr"]) assert getxattr(tmp_fn, b"user.foo") == b"bar" assert getxattr(tmp_fd, b"user.foo") == b"bar" assert getxattr(tmp_lfn, b"user.foo", follow_symlinks=True) == b"bar" - if not is_linux: + if symlink_xattrs: assert getxattr(tmp_lfn, b"user.linkxattr") == b"baz" assert getxattr(tmp_fn, b"user.empty") == b"" +@pytest.mark.skipif(is_sunos, reason="the illumos/Solaris xattr backend does not use the shared buffer") def test_listxattr_buffer_growth(tempfile_symlink): temp_file, symlink = tempfile_symlink tmp_fn = os.fsencode(temp_file.name) @@ -65,6 +70,7 @@ def test_listxattr_buffer_growth(tempfile_symlink): assert len(buffer) > 64 +@pytest.mark.skipif(is_sunos, reason="the illumos/Solaris xattr backend does not use the shared buffer") def test_getxattr_buffer_growth(tempfile_symlink): temp_file, symlink = tempfile_symlink tmp_fn = os.fsencode(temp_file.name) diff --git a/src/borg/xattr.py b/src/borg/xattr.py index 776a946774..eb04149f22 100644 --- a/src/borg/xattr.py +++ b/src/borg/xattr.py @@ -1,4 +1,4 @@ -"""A basic extended attributes (xattr) implementation for Linux, FreeBSD and macOS.""" +"""A basic extended attributes (xattr) implementation for Linux, FreeBSD, macOS, NetBSD and illumos/Solaris.""" import errno import os From 0ada4c73b7e4de661467ac76151fca614c7ce22e Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Fri, 14 Aug 2026 17:18:32 +0200 Subject: [PATCH 125/144] tests: relax wall-clock bound in read-special-timeout test test_create_read_special_timeout_expired asserted that borg create with --read-special-timeout=1 returns within 30s. On heavily loaded CI runners the whole command (incl. remote archiver setup) occasionally takes longer than that (observed: 37s on a macOS runner), failing the test although the timeout mechanism worked correctly. Relax the bound to 300s: it still proves borg gave up soon after the requested 1s timeout instead of waiting for the 30 minutes default timeout, but no longer trips over slow runners. --- src/borg/testsuite/archiver/create_cmd_test.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/src/borg/testsuite/archiver/create_cmd_test.py b/src/borg/testsuite/archiver/create_cmd_test.py index 15c2f064e8..d0e4b072f7 100644 --- a/src/borg/testsuite/archiver/create_cmd_test.py +++ b/src/borg/testsuite/archiver/create_cmd_test.py @@ -1037,7 +1037,9 @@ def test_create_read_special_timeout_expired(archivers, request): "input", exit_code=expected_ec, # WARNING status: could not back up the fifo. ) - assert time.monotonic() - started < 30 # timed out, no eternal hang + # Generous bound for heavily loaded CI runners: proves borg gave up soon after the requested + # 1s timeout (and did not e.g. wait for the 30 minutes default timeout). + assert time.monotonic() - started < 300 assert "retry: 1 of " not in out # timeouts are NOT retried listing = cmd(archiver, "list", "test", "--format={path}{NL}") assert "input/file1" in listing From 4dfcf29ad2906df47fc0b4d681a587bdd0ab4ada Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Fri, 14 Aug 2026 19:40:41 +0200 Subject: [PATCH 126/144] CI: pin all GitHub Actions to commit SHAs Tags are mutable. An attacker who compromises an action repository can move an existing tag to a malicious commit, which then runs in our workflows with the workflow token and can tamper with build artifacts before they get attested. Pin every action to the full commit SHA of the release it currently resolves to and keep the human-readable version in a trailing comment, matching the convention already used for psf/black and peter-evans/create-pull-request. Dependabot keeps updating SHA-pinned actions, bumping both the SHA and the version comment. --- .github/workflows/backport.yml | 4 +- .github/workflows/black.yaml | 2 +- .github/workflows/canary-py312-ubuntu2604.yml | 2 +- .github/workflows/canary.yml | 10 ++-- .github/workflows/ci.yml | 54 +++++++++---------- .github/workflows/codeql-analysis.yml | 10 ++-- .github/workflows/fame.yml | 4 +- 7 files changed, 43 insertions(+), 43 deletions(-) diff --git a/.github/workflows/backport.yml b/.github/workflows/backport.yml index deab27e150..7f06c99e9c 100644 --- a/.github/workflows/backport.yml +++ b/.github/workflows/backport.yml @@ -31,8 +31,8 @@ jobs: startsWith(github.event.comment.body, '/backport') ) steps: - - uses: actions/checkout@v7 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - name: Create backport pull requests - uses: korthout/backport-action@v4 + uses: korthout/backport-action@2e830a1d0b8269505846ddd407a70876913ad1f8 # v4.6.0 with: label_pattern: '^port/(.+)$' diff --git a/.github/workflows/black.yaml b/.github/workflows/black.yaml index 33ef1f60f1..94c22ea57d 100644 --- a/.github/workflows/black.yaml +++ b/.github/workflows/black.yaml @@ -24,7 +24,7 @@ jobs: runs-on: ubuntu-24.04 timeout-minutes: 5 steps: - - uses: actions/checkout@v7 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - uses: psf/black@87928e6d6761a4a6d22250e1fee5601b3998086e # 26.5.1 with: version: "~= 24.0" diff --git a/.github/workflows/canary-py312-ubuntu2604.yml b/.github/workflows/canary-py312-ubuntu2604.yml index a770915ad8..c4658898da 100644 --- a/.github/workflows/canary-py312-ubuntu2604.yml +++ b/.github/workflows/canary-py312-ubuntu2604.yml @@ -30,7 +30,7 @@ jobs: - name: Try to set up Python 3.12 id: setup continue-on-error: true - uses: actions/setup-python@v7 + uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 with: python-version: '3.12' diff --git a/.github/workflows/canary.yml b/.github/workflows/canary.yml index b812d61542..818cee44b7 100644 --- a/.github/workflows/canary.yml +++ b/.github/workflows/canary.yml @@ -44,13 +44,13 @@ jobs: toxenv: py314-none steps: - - uses: actions/checkout@v7 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: fetch-depth: 0 fetch-tags: true - name: Set up Python ${{ matrix.python-version }} - uses: actions/setup-python@v7 + uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 with: python-version: ${{ matrix.python-version }} @@ -125,7 +125,7 @@ jobs: shell: msys2 {0} steps: - - uses: actions/checkout@v7 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: fetch-depth: 0 @@ -134,14 +134,14 @@ jobs: # unlocked requirements may resolve to different versions; falls back # to (and gets picked up by) the ci.yml cache via the shared prefix. - name: Cache pip-built wheels - uses: actions/cache@v6 + uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 with: path: .pip-cache key: windows-msys2-pip-canary-${{ hashFiles('pyproject.toml', 'requirements.d/pyinstaller.txt') }} restore-keys: | windows-msys2-pip- - - uses: msys2/setup-msys2@v2 + - uses: msys2/setup-msys2@66cd2cce69caa17b53920067426061ca1de3a884 # v2.32.0 with: msystem: UCRT64 update: true diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index aaa5728f5a..334ef1ca13 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -42,8 +42,8 @@ jobs: timeout-minutes: 5 steps: - - uses: actions/checkout@v7 - - uses: astral-sh/ruff-action@v4.1.0 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + - uses: astral-sh/ruff-action@278981a28ce3188b1e39527901f38254bf3aac89 # v4.1.0 security: @@ -51,9 +51,9 @@ jobs: timeout-minutes: 5 steps: - - uses: actions/checkout@v7 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - name: Set up Python - uses: actions/setup-python@v7 + uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 with: python-version: '3.11' - name: Install dependencies @@ -71,14 +71,14 @@ jobs: needs: [lint] steps: - - uses: actions/checkout@v7 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: # Just fetching one commit is not enough for setuptools-scm, so we fetch all. fetch-depth: 0 fetch-tags: true - name: Set up Python - uses: actions/setup-python@v7 + uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 with: python-version: '3.12' @@ -182,19 +182,19 @@ jobs: timeout-minutes: 360 steps: - - uses: actions/checkout@v7 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: # Just fetching one commit is not enough for setuptools-scm, so we fetch all. fetch-depth: 0 fetch-tags: true - name: Set up Python ${{ matrix.python-version }} - uses: actions/setup-python@v7 + uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 with: python-version: ${{ matrix.python-version }} - name: Cache pip - uses: actions/cache@v6 + uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 with: path: ~/.cache/pip key: ${{ runner.os }}-${{ runner.arch }}-pip-${{ hashFiles('requirements.d/development.lock.txt') }} @@ -203,7 +203,7 @@ jobs: ${{ runner.os }}-${{ runner.arch }}- - name: Cache tox environments - uses: actions/cache@v6 + uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 with: path: .tox key: ${{ runner.os }}-${{ runner.arch }}-tox-${{ matrix.toxenv }}-${{ hashFiles('requirements.d/development.lock.txt', 'pyproject.toml') }} @@ -352,13 +352,13 @@ jobs: - name: Attest binaries provenance (${{ matrix.binary }}) if: ${{ matrix.binary && startsWith(github.ref, 'refs/tags/') }} - uses: actions/attest-build-provenance@v4 + uses: actions/attest-build-provenance@4d101475d8b20a2381f78447822ac1eab6504dd8 # v4.2.2 with: subject-path: 'artifacts/*' - name: Upload binaries (${{ matrix.binary }}) if: ${{ matrix.binary && startsWith(github.ref, 'refs/tags/') }} - uses: actions/upload-artifact@v7 + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: ${{ matrix.binary }} path: artifacts/* @@ -374,7 +374,7 @@ jobs: - name: Upload test results to Codecov if: ${{ !cancelled() && !contains(matrix.toxenv, 'mypy') && !contains(matrix.toxenv, 'docs') }} - uses: codecov/codecov-action@v7 + uses: codecov/codecov-action@fb8b3582c8e4def4969c97caa2f19720cb33a72f # v7.0.0 env: OS: ${{ runner.os }} python: ${{ matrix.python-version }} @@ -386,7 +386,7 @@ jobs: - name: Upload coverage to Codecov if: ${{ !cancelled() && !contains(matrix.toxenv, 'mypy') && !contains(matrix.toxenv, 'docs') }} - uses: codecov/codecov-action@v7 + uses: codecov/codecov-action@fb8b3582c8e4def4969c97caa2f19720cb33a72f # v7.0.0 env: OS: ${{ runner.os }} python: ${{ matrix.python-version }} @@ -436,7 +436,7 @@ jobs: steps: - name: Check out repository - uses: actions/checkout@v7 + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: fetch-depth: 0 fetch-tags: true @@ -445,7 +445,7 @@ jobs: # rsyncs into the VM and back, so wheels built from sdists in one run # (most of the VM setup time) are reused by the next one. - name: Cache pip-built wheels - uses: actions/cache@v6 + uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 with: path: .pip-cache key: ${{ matrix.os }}-${{ matrix.version }}-pip-${{ hashFiles('requirements.d/development.lock.txt') }} @@ -457,7 +457,7 @@ jobs: # a normal boot takes < 1 minute; since 1.4.0 the action bounds # waiting for boot itself, this is just an additional safety net. timeout-minutes: 15 - uses: cross-platform-actions/action@v1.4.0 + uses: cross-platform-actions/action@24ef01df165c76df1ed2b9f9e9212e78dc2fc963 # v1.4.0 with: operating_system: ${{ matrix.os }} version: ${{ matrix.version }} @@ -644,7 +644,7 @@ jobs: - name: Upload artifacts if: startsWith(github.ref, 'refs/tags/') && matrix.do_binaries - uses: actions/upload-artifact@v7 + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: ${{ matrix.artifact_prefix }} path: artifacts/* @@ -652,13 +652,13 @@ jobs: - name: Attest provenance if: startsWith(github.ref, 'refs/tags/') && matrix.do_binaries - uses: actions/attest-build-provenance@v4 + uses: actions/attest-build-provenance@4d101475d8b20a2381f78447822ac1eab6504dd8 # v4.2.2 with: subject-path: 'artifacts/*' - name: Upload test results to Codecov if: ${{ !cancelled() }} - uses: codecov/codecov-action@v7 + uses: codecov/codecov-action@fb8b3582c8e4def4969c97caa2f19720cb33a72f # v7.0.0 env: OS: ${{ matrix.os }} with: @@ -669,7 +669,7 @@ jobs: - name: Upload coverage to Codecov if: ${{ !cancelled() }} - uses: codecov/codecov-action@v7 + uses: codecov/codecov-action@fb8b3582c8e4def4969c97caa2f19720cb33a72f # v7.0.0 env: OS: ${{ matrix.os }} with: @@ -697,7 +697,7 @@ jobs: shell: msys2 {0} steps: - - uses: actions/checkout@v7 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: fetch-depth: 0 @@ -705,14 +705,14 @@ jobs: # all compiled deps from source - incl. building maturin via cargo, just # to build the blake3 wheel. Persist the wheels pip builds. - name: Cache pip-built wheels - uses: actions/cache@v6 + uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 with: path: .pip-cache key: windows-msys2-pip-${{ hashFiles('pyproject.toml', 'requirements.d/pyinstaller.txt') }} restore-keys: | windows-msys2-pip- - - uses: msys2/setup-msys2@v2 + - uses: msys2/setup-msys2@66cd2cce69caa17b53920067426061ca1de3a884 # v2.32.0 with: msystem: UCRT64 update: true @@ -738,7 +738,7 @@ jobs: # build sdist and wheel in dist/... python -m build - - uses: actions/upload-artifact@v7 + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: borg-windows path: dist/binary/borg.exe @@ -753,7 +753,7 @@ jobs: - name: Upload test results to Codecov if: ${{ !cancelled() }} - uses: codecov/codecov-action@v7 + uses: codecov/codecov-action@fb8b3582c8e4def4969c97caa2f19720cb33a72f # v7.0.0 env: OS: ${{ runner.os }} python: '3.11' @@ -765,7 +765,7 @@ jobs: - name: Upload coverage to Codecov if: ${{ !cancelled() }} - uses: codecov/codecov-action@v7 + uses: codecov/codecov-action@fb8b3582c8e4def4969c97caa2f19720cb33a72f # v7.0.0 env: OS: ${{ runner.os }} python: '3.11' diff --git a/.github/workflows/codeql-analysis.yml b/.github/workflows/codeql-analysis.yml index b4be4306cb..834bffadcc 100644 --- a/.github/workflows/codeql-analysis.yml +++ b/.github/workflows/codeql-analysis.yml @@ -46,16 +46,16 @@ jobs: steps: - name: Checkout repository - uses: actions/checkout@v7 + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: # Just fetching one commit is not enough for setuptools-scm, so we fetch all. fetch-depth: 0 - name: Set up Python - uses: actions/setup-python@v7 + uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 with: python-version: 3.11 - name: Cache pip - uses: actions/cache@v6 + uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 with: path: ~/.cache/pip key: ${{ runner.os }}-pip-${{ hashFiles('requirements.d/development.txt') }} @@ -69,7 +69,7 @@ jobs: sudo apt-get install -y libssl-dev libacl1-dev liblz4-dev # Initializes the CodeQL tools for scanning. - name: Initialize CodeQL - uses: github/codeql-action/init@v4.37.6 + uses: github/codeql-action/init@5595ccaf912efad79be6eef63a5619ff05969be3 # v4.37.6 with: languages: ${{ matrix.language }} # If you wish to specify custom queries, you can do so here or in a config file. @@ -83,4 +83,4 @@ jobs: pip3 install -r requirements.d/development.txt pip3 install -ve . - name: Perform CodeQL Analysis - uses: github/codeql-action/analyze@v4.37.6 + uses: github/codeql-action/analyze@5595ccaf912efad79be6eef63a5619ff05969be3 # v4.37.6 diff --git a/.github/workflows/fame.yml b/.github/workflows/fame.yml index cb96c605b4..1baf34394c 100644 --- a/.github/workflows/fame.yml +++ b/.github/workflows/fame.yml @@ -42,13 +42,13 @@ jobs: pull-requests: write steps: - - uses: actions/checkout@v7 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: ref: master fetch-depth: 0 # git blame needs the whole history - name: Set up Python - uses: actions/setup-python@v7 + uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 with: python-version: '3.14' From daa1944e74fecb5aa53ff00458b3332d60ce8edd Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Fri, 14 Aug 2026 19:48:37 +0200 Subject: [PATCH 127/144] CI: add a Dependabot cooldown for GitHub Actions Malicious releases are usually caught and pulled within days, so waiting before adopting a new version buys herd immunity for little cost. A compromised action is the more severe case of the two we consume: it runs with the workflow token and can tamper with build artifacts before they get attested. Mirror the major/minor values already used for pip. default-days additionally covers patch releases and actions that do not use semver, which the pip block does not set. Security updates are not delayed by cooldown. --- .github/dependabot.yml | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/.github/dependabot.yml b/.github/dependabot.yml index e18ae10ccc..b65cdf26b3 100644 --- a/.github/dependabot.yml +++ b/.github/dependabot.yml @@ -4,6 +4,11 @@ updates: directory: "/" schedule: interval: "weekly" + cooldown: + # default-days also covers patch releases and actions not using semver. + default-days: 14 + semver-major-days: 90 + semver-minor-days: 30 groups: actions: patterns: From 18def9bdebb2471693db88e283479f7e48e736d0 Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Fri, 14 Aug 2026 20:18:57 +0200 Subject: [PATCH 128/144] docs: FAQ, 2 ways to limit bandwidth, fixes #8838 BORGSTORE_BANDWIDTH works for all backends and in both directions, but does not shape the traffic. pv on both sides of a ssh ProxyCommand is smoother and has a separate limit per direction, but only works for rest:// repositories. Give pros and cons of each. Also mention BORGSTORE_LATENCY and rclone's own --bwlimit. --- docs/faq.rst | 76 +++++++++++++++++++++++++++++++++++++++++----------- 1 file changed, 61 insertions(+), 15 deletions(-) diff --git a/docs/faq.rst b/docs/faq.rst index e4fd87e59a..2c731ebc8e 100644 --- a/docs/faq.rst +++ b/docs/faq.rst @@ -1032,33 +1032,79 @@ code). Is there a way to limit bandwidth with Borg? -------------------------------------------- -Borg has no built-in bandwidth limiting - the ``--remote-ratelimit`` and -``--upload-ratelimit`` options were removed. +Borg has no bandwidth limiting option - ``--remote-ratelimit`` and +``--upload-ratelimit`` were removed. There are 2 ways to do it anyway: -For repositories accessed via ssh, bandwidth can be limited with pipeviewer_: +Using borgstore's bandwidth limit +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ -Create a wrapper script: /usr/local/bin/pv-wrapper +borgstore, which Borg uses for repository access, can limit the transfer rate:: -:: + # 16 Mbit/s == 2 MB/s, 0 (the default) means unlimited: + export BORGSTORE_BANDWIDTH=16000000 + +Pros: + +- works for all backends (``sftp://``, ``rest://``, ``rclone:``, ``s3://``, ...). +- limits both directions. +- needs no additional software. + +Cons: + +- the value is given in **bits** per second (easy to confuse with ``pv``, which + uses bytes per second). +- only the object payload is accounted for, protocol overhead (ssh, TLS, http, + ...) comes on top, so the real usage is a bit higher than the given rate. +- it does not shape the traffic: an object is transferred at full speed and + Borg then waits until the given rate is reached on average. As Borg transfers + rather big objects (pack files are up to 50MB), the connection will be busy + for a while and idle afterwards. + +There is also ``BORGSTORE_LATENCY`` (in microseconds), which adds a delay per +backend call. + +Using pv on the ssh connection +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ - #!/bin/sh +For ``rest://`` repositories, Borg connects via ssh, so the transfer can be +limited with pipeviewer_. Put a ``pv`` on each side of the connection using an +ssh ``ProxyCommand`` (this needs ``nc``), e.g. in ``~/.ssh/config``:: + + Host borghost ## -q, --quiet do not output any transfer information at all ## -L, --rate-limit RATE limit transfer to RATE bytes per second - RATE=307200 - pv -q -L $RATE | "$@" + ProxyCommand pv -q -L 307200 | nc %h %p | pv -q -L 307200 -Add BORG_RSH environment variable to use pipeviewer wrapper script with ssh. +The first ``pv`` limits the upload, the second one the download, each to RATE +bytes per second. -:: +Pros: - export BORG_RSH='/usr/local/bin/pv-wrapper ssh' +- smoother, as ``pv`` limits the rate of the data stream itself. +- each direction has its own limit. +- the ssh protocol overhead is included in the limit. +- the rate can be changed on the fly:: -Now Borg will be bandwidth limited. The nice thing about ``pv`` is that you can -change rate-limit on the fly: + pv -R $(pidof pv) -L 102400 -:: + As the ``ProxyCommand`` above runs 2 ``pv`` processes, ``pidof`` will print 2 + pids - give ``pv -R`` the pid of the direction you want to change. + +Cons: + +- only works for ``rest://`` repositories. ``sftp://`` does not run an external + ssh command (borgstore uses paramiko and creates the connection itself) and + a ``ProxyCommand`` in ``~/.ssh/config`` is not used there. +- needs ``pv`` and ``nc``. + +Using rclone's bandwidth limit +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +For ``rclone:`` repositories, you can also use rclone's own bandwidth limiting, +see its ``--bwlimit`` option. rclone picks up options from the environment, so +you can use:: - pv -R $(pidof pv) -L 102400 + export RCLONE_BWLIMIT=300k .. _pipeviewer: https://www.ivarch.com/programs/pv.shtml From 9dd91fcb8bdec950bd4deb98e01917a95e24a124 Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Fri, 14 Aug 2026 23:49:51 +0200 Subject: [PATCH 129/144] create: do not use ctime for the files cache on Windows, fixes #7193 On Windows, os.stat_result.st_ctime[_ns] is the file *creation* time (CPython literally assigns st_ctime = st_birthtime there), not the "metadata change time" it is on POSIX systems. It therefore never changes when a file's contents change. With the default files cache mode "ctime,size,inode", a file that was modified in place while keeping its size and inode number was thus considered unchanged and its new contents were silently not backed up - only the very first archive was complete. So, on Windows: - the default files cache mode is "mtime,size,inode" now - an explicitly given ctime based mode warns and uses the corresponding mtime based mode instead (ctime,size,inode -> mtime,size,inode, ctime,size -> mtime,size, rechunk,ctime -> rechunk,mtime) This is the same approach as already used for --files-changed. Note: the files cache always memorizes both ctime and mtime, so this does not invalidate existing files cache entries. --- docs/changes.rst | 6 ++++ docs/usage/create.rst.inc | 15 ++++++--- src/borg/archiver/create_cmd.py | 32 ++++++++++++++++--- src/borg/constants.py | 8 +++-- src/borg/helpers/__init__.py | 1 + src/borg/helpers/parseformat.py | 17 ++++++++++ .../testsuite/archiver/create_cmd_test.py | 21 +++++++++++- .../testsuite/helpers/parseformat_test.py | 30 +++++++++++++++++ 8 files changed, 118 insertions(+), 12 deletions(-) diff --git a/docs/changes.rst b/docs/changes.rst index b74f07c795..5ccd15cc17 100644 --- a/docs/changes.rst +++ b/docs/changes.rst @@ -253,6 +253,12 @@ New features: Fixes: +- create: do not use ctime for the files cache on Windows, #7193. + + ctime is the file *creation* time on Windows, so a ctime based files cache mode + did not notice content changes of files that kept their size and inode number. + The default is ``mtime,size,inode`` there now and an explicitly given ctime based + mode warns and uses the corresponding mtime based mode. - re-add XXH64 to read borg 1.x integrity data, #9935 - list: add {blake3} format key, #9984 - support date: archive patterns for --from-borg1, #9949 diff --git a/docs/usage/create.rst.inc b/docs/usage/create.rst.inc index b914e9f204..3f87001085 100644 --- a/docs/usage/create.rst.inc +++ b/docs/usage/create.rst.inc @@ -91,7 +91,7 @@ borg create +-------------------------------------------------------+---------------------------------------------------+------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ | | ``--sparse`` | detect sparse holes in input (supported only by fixed chunker) | +-------------------------------------------------------+---------------------------------------------------+------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ - | | ``--files-cache MODE`` | operate files cache in MODE. default: ctime,size,inode | + | | ``--files-cache MODE`` | operate files cache in MODE. default: ctime,size,inode (on Windows: mtime,size,inode, because ctime is file creation time there). | +-------------------------------------------------------+---------------------------------------------------+------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ | | ``--files-changed MODE`` | specify how to detect if a file has changed during backup (ctime, mtime, disabled). default: ctime (on Windows: mtime, because ctime is file creation time there). | +-------------------------------------------------------+---------------------------------------------------+------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ @@ -169,7 +169,7 @@ borg create --noacls do not read and store ACLs into archive --noxattrs do not read and store xattrs into archive --sparse detect sparse holes in input (supported only by fixed chunker) - --files-cache MODE operate files cache in MODE. default: ctime,size,inode + --files-cache MODE operate files cache in MODE. default: ctime,size,inode (on Windows: mtime,size,inode, because ctime is file creation time there). --files-changed MODE specify how to detect if a file has changed during backup (ctime, mtime, disabled). default: ctime (on Windows: mtime, because ctime is file creation time there). --read-special open and read block and char device files as well as FIFOs as if they were regular files. Also follows symlinks pointing to these kinds of files. --read-special-timeout SECONDS when reading from FIFOs or character devices (see --read-special): skip the file with an error if no data arrives for more than SECONDS (this includes waiting for a FIFO's writer to connect). Give 0 to wait forever. default: 1800 seconds. @@ -220,8 +220,8 @@ the files cache. This comparison can operate in different modes as given by ``--files-cache``: -- ctime,size,inode (default) -- mtime,size,inode (default behaviour of borg versions older than 1.1.0rc4) +- ctime,size,inode (default on POSIX systems) +- mtime,size,inode (default on Windows) - ctime,size (ignore the inode number) - mtime,size (ignore the inode number) - rechunk,ctime (all files are considered modified - rechunk, cache ctime) @@ -249,6 +249,13 @@ ctime vs. mtime: safety vs. speed it had before a content change happened. This can be used maliciously as well as well-meant, but in both cases mtime-based cache modes can be problematic. +On Windows, ctime is the file *creation* time, not the "metadata change time" it is +on POSIX systems. A ctime based mode would therefore not notice content changes of a +file that keeps its size and inode number, so borg defaults to ``mtime,size,inode`` +there. If a ctime based mode is given explicitly on Windows, borg warns and uses the +corresponding mtime based mode instead: ctime,size,inode -> mtime,size,inode, +ctime,size -> mtime,size, rechunk,ctime -> rechunk,mtime. + The ``--files-changed`` option controls how Borg detects if a file has changed during backup: - ctime (default on POSIX): Use ctime to detect changes. This is the safest option. Not supported on Windows (ctime is file creation time there). diff --git a/src/borg/archiver/create_cmd.py b/src/borg/archiver/create_cmd.py index 1727ca14d2..748dbe3be9 100644 --- a/src/borg/archiver/create_cmd.py +++ b/src/borg/archiver/create_cmd.py @@ -16,7 +16,8 @@ from ..cache import Cache from ..constants import * # NOQA from ..helpers import comment_validator, ChunkerParams, FilesystemPathSpec, CompressionSpec -from ..helpers import archivename_validator, FilesCacheMode, octal_int, nonnegative_seconds +from ..helpers import archivename_validator, FilesCacheMode, files_cache_mode_no_ctime +from ..helpers import octal_int, nonnegative_seconds from ..helpers import eval_escapes from ..helpers import timestamp, archive_ts_now from ..helpers import get_cache_dir, os_stat, get_strip_prefix, slashify @@ -50,6 +51,19 @@ def do_create(self, args, repository, manifest): read_special_timeout = READ_SPECIAL_TIMEOUT_DEFAULT if read_special_timeout == 0: read_special_timeout = None # wait forever + if is_win32: + # st_ctime is the file *creation* time on Windows, not the "metadata change time", + # so a ctime based files cache mode would not detect content changes of a file that + # keeps its size and inode number. Use the mtime based equivalent instead, see #7193. + # note: this must happen before the Cache is created (it gets the files cache mode). + files_cache_mode, changed = files_cache_mode_no_ctime(args.files_cache_mode) + if changed: + self.print_warning( + "--files-cache=ctime,... is not supported on Windows " + "(ctime is file creation time, not change time). Using mtime instead.", + wc=None, + ) + args.files_cache_mode = files_cache_mode key = manifest.key matcher = PatternMatcher(fallback=True) matcher.add_inclexcl(args.patterns) @@ -619,8 +633,8 @@ def build_parser_create(self, subparsers, common_parser, mid_common_parser): This comparison can operate in different modes as given by ``--files-cache``: - - ctime,size,inode (default) - - mtime,size,inode (default behaviour of borg versions older than 1.1.0rc4) + - ctime,size,inode (default on POSIX systems) + - mtime,size,inode (default on Windows) - ctime,size (ignore the inode number) - mtime,size (ignore the inode number) - rechunk,ctime (all files are considered modified - rechunk, cache ctime) @@ -648,6 +662,13 @@ def build_parser_create(self, subparsers, common_parser, mid_common_parser): it had before a content change happened. This can be used maliciously as well as well-meant, but in both cases mtime-based cache modes can be problematic. + On Windows, ctime is the file *creation* time, not the "metadata change time" it is + on POSIX systems. A ctime based mode would therefore not notice content changes of a + file that keeps its size and inode number, so borg defaults to ``mtime,size,inode`` + there. If a ctime based mode is given explicitly on Windows, borg warns and uses the + corresponding mtime based mode instead: ctime,size,inode -> mtime,size,inode, + ctime,size -> mtime,size, rechunk,ctime -> rechunk,mtime. + The ``--files-changed`` option controls how Borg detects if a file has changed during backup: - ctime (default on POSIX): Use ctime to detect changes. This is the safest option. Not supported on Windows (ctime is file creation time there). @@ -971,8 +992,9 @@ def build_parser_create(self, subparsers, common_parser, mid_common_parser): dest="files_cache_mode", action=Highlander, type=FilesCacheMode, - default=FILES_CACHE_MODE_UI_DEFAULT, - help="operate files cache in MODE. default: %s" % FILES_CACHE_MODE_UI_DEFAULT, + default=FILES_CACHE_MODE_UI_DEFAULT_WIN32 if is_win32 else FILES_CACHE_MODE_UI_DEFAULT_POSIX, + help="operate files cache in MODE. default: %s (on Windows: %s, because ctime is " + "file creation time there)." % (FILES_CACHE_MODE_UI_DEFAULT_POSIX, FILES_CACHE_MODE_UI_DEFAULT_WIN32), ) fs_group.add_argument( "--files-changed", diff --git a/src/borg/constants.py b/src/borg/constants.py index 6103eb630e..fb8483fb8b 100644 --- a/src/borg/constants.py +++ b/src/borg/constants.py @@ -177,8 +177,12 @@ # normal on-disk data, allocated (but not written, all zeros), not allocated hole (all zeros) CH_DATA, CH_ALLOC, CH_HOLE = 0, 1, 2 -# operating mode of the files cache (for fast skipping of unchanged files) -FILES_CACHE_MODE_UI_DEFAULT = "ctime,size,inode" # default for "borg create" command (CLI UI) +# operating mode of the files cache (for fast skipping of unchanged files). +# note: on Windows, os.stat_result.st_ctime[_ns] is the file *creation* time, not the +# "metadata change time" it is on POSIX systems. A ctime based mode would not detect +# content changes of a file that keeps its size and inode number there. See #7193. +FILES_CACHE_MODE_UI_DEFAULT_POSIX = "ctime,size,inode" # default for "borg create" command (CLI UI) +FILES_CACHE_MODE_UI_DEFAULT_WIN32 = "mtime,size,inode" # same, but on Windows FILES_CACHE_MODE_DISABLED = "d" # most borg commands do not use the files cache at all (disable) # account for clocks being slightly out-of-sync, timestamps granularity. diff --git a/src/borg/helpers/__init__.py b/src/borg/helpers/__init__.py index 7a0fdbccf5..898e92a59a 100644 --- a/src/borg/helpers/__init__.py +++ b/src/borg/helpers/__init__.py @@ -38,6 +38,7 @@ CompressionSpec, ChunkerParams, FilesCacheMode, + files_cache_mode_no_ctime, partial_format, DatetimeWrapper, ) diff --git a/src/borg/helpers/parseformat.py b/src/borg/helpers/parseformat.py index fe91fc7cd8..d0c423b80c 100644 --- a/src/borg/helpers/parseformat.py +++ b/src/borg/helpers/parseformat.py @@ -429,6 +429,23 @@ def FilesCacheMode(s): return mode +# ctime based files cache modes and their mtime based equivalent, see #7193. +FILES_CACHE_MODE_CTIME_TO_MTIME = {"cis": "ims", "cs": "ms", "cr": "mr"} + + +def files_cache_mode_no_ctime(mode): + """ + Return (new_mode, changed): with ctime replaced by mtime. + + may be given in long form ("ctime,size,inode") or in short form ("cis"), + the returned mode is always the short form, as FilesCacheMode() gives it. + "changed" tells whether a ctime -> mtime replacement actually happened. + """ + mode = FilesCacheMode(mode) # long form -> short form (idempotent for the short form) + new_mode = FILES_CACHE_MODE_CTIME_TO_MTIME.get(mode, mode) + return new_mode, new_mode != mode + + def partial_format(format, mapping): """ Apply format.format_map(mapping) while preserving unknown keys diff --git a/src/borg/testsuite/archiver/create_cmd_test.py b/src/borg/testsuite/archiver/create_cmd_test.py index d0e4b072f7..22a712761b 100644 --- a/src/borg/testsuite/archiver/create_cmd_test.py +++ b/src/borg/testsuite/archiver/create_cmd_test.py @@ -741,7 +741,7 @@ def test_create_invalid_tags(archivers, request): @pytest.mark.skipif( - is_win32, reason="ctime attribute is file creation time on Windows" + is_win32, reason="ctime is file creation time on Windows, ctime cache modes fall back to mtime there, see #7193" ) # see https://docs.python.org/3/library/os.html#os.stat_result.st_ctime def test_file_status_cs_cache_mode(archivers, request): archiver = request.getfixturevalue(archivers) @@ -760,6 +760,25 @@ def test_file_status_cs_cache_mode(archivers, request): assert "M input/file1" in output +@pytest.mark.skipif(not is_win32, reason="Windows-only: ctime is file creation time there, see #7193") +def test_files_cache_ctime_fallback_win32(archivers, request): + """test that a ctime based --files-cache mode warns and falls back to the mtime based mode on Windows""" + archiver = request.getfixturevalue(archivers) + create_regular_file(archiver.input_path, "file1", contents=b"123") + granularity_sleep() # file2 must have newer timestamps than file1 + create_regular_file(archiver.input_path, "file2", size=10) + cmd(archiver, "repo-create", RK_ENCRYPTION) + # note: cmd() asserts rc == 0, so this also covers that the warning does not change the exit code. + output = cmd(archiver, "create", "test", "input", "--list", "--files-cache=ctime,size") + assert "ctime is file creation time" in output + granularity_sleep() # the rewrite below must get a newer mtime + # rewrite in place: same size, same inode, same creation time (== st_ctime on Windows), + # only the mtime changes. Without the fallback, this file would be considered unchanged. + create_regular_file(archiver.input_path, "file1", contents=b"321") + output = cmd(archiver, "create", "test", "input", "--list", "--files-cache=ctime,size") + assert "M input/file1" in output + + def test_files_changed_modes(archivers, request): """test that all --files-changed modes are accepted and work""" archiver = request.getfixturevalue(archivers) diff --git a/src/borg/testsuite/helpers/parseformat_test.py b/src/borg/testsuite/helpers/parseformat_test.py index fa95f8e8d4..0e1d7cd713 100644 --- a/src/borg/testsuite/helpers/parseformat_test.py +++ b/src/borg/testsuite/helpers/parseformat_test.py @@ -30,6 +30,7 @@ swidth_slice, eval_escapes, ChunkerParams, + files_cache_mode_no_ctime, get_size_units, normalize_local_path, ArchiveFormatter, @@ -915,6 +916,35 @@ def test_invalid_chunkerparams(invalid_chunker_params): ChunkerParams(invalid_chunker_params) +@pytest.mark.parametrize( + "mode, expected_mode, expected_changed", + [ + # ctime based modes get replaced by their mtime based equivalent: + ("ctime,size,inode", "ims", True), + ("cis", "ims", True), + ("ctime,size", "ms", True), + ("cs", "ms", True), + ("rechunk,ctime", "mr", True), + ("cr", "mr", True), + # everything else stays as it is (but is normalized to the short form): + ("mtime,size,inode", "ims", False), + ("ims", "ims", False), + ("mtime,size", "ms", False), + ("rechunk,mtime", "mr", False), + ("disabled", "d", False), + ("d", "d", False), + ], +) +def test_files_cache_mode_no_ctime(mode, expected_mode, expected_changed): + assert files_cache_mode_no_ctime(mode) == (expected_mode, expected_changed) + + +def test_files_cache_mode_ui_default(): + # borg create must not default to a ctime based files cache mode on Windows, see #7193. + assert FILES_CACHE_MODE_UI_DEFAULT_POSIX == "ctime,size,inode" + assert FILES_CACHE_MODE_UI_DEFAULT_WIN32 == "mtime,size,inode" + + @pytest.mark.parametrize( "env_value, expect_newlines", [ From 527fc57567732b467597438ff9ec046ac6b18e6b Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Fri, 14 Aug 2026 23:54:11 +0200 Subject: [PATCH 130/144] compression: cap the default zstd MT workers at 4 for chunks The BORG_ZSTD_MT_WORKERS default was the cpu count, but a chunk only yields ceil(size / 512KiB) compression jobs - 4 for the 2MiB chunks the default chunker aims at - so most of a big thread pool gets no work, while the pool is still created and torn down again for every chunk. Measured at the chunk level on a 12-core machine (fastcdc 512KiB..8MiB chunks, real data), 4 workers beat 12 on every corpus at the default zstd,-4: +13% (source code) .. +37% (VM image) big-chunk throughput. A sweep over 1/2/4/6/8/12 workers, bucketed by jobs-per-chunk, shows the throughput peak tracking the job count (2-job chunks peak at 2 workers, 4-job chunks at 4) and 12 workers winning nowhere: beyond the job count, extra threads only add startup overhead. export-tar compresses one long stream through a single pool using libzstd's default (large) job size, so it keeps the cpu count as its default via the new stream=True parameter. An explicitly set BORG_ZSTD_MT_WORKERS still overrides both defaults, e.g. for chunker configurations producing much bigger chunks. Co-Authored-By: Claude Fable 5 --- src/borg/archiver/help_cmd.py | 11 +++++++- src/borg/archiver/tar_cmds.py | 2 +- src/borg/compress.pyi | 2 +- src/borg/compress.pyx | 41 ++++++++++++++++++++--------- src/borg/testsuite/compress_test.py | 15 ++++++----- 5 files changed, 49 insertions(+), 22 deletions(-) diff --git a/src/borg/archiver/help_cmd.py b/src/borg/archiver/help_cmd.py index ce34b6c06a..28980f7db6 100644 --- a/src/borg/archiver/help_cmd.py +++ b/src/borg/archiver/help_cmd.py @@ -764,15 +764,24 @@ class HelpMixIn: 0 means "always multi-threaded", a very large value effectively disables multi-threading. BORG_ZSTD_MT_WORKERS When set to a numeric value, use that many threads to zstd-compress a single chunk - (default: the cpu count). 0 or 1 means single-threaded compression. + (default: the cpu count, but at most 4). 0 or 1 means single-threaded compression. Only relevant when compressing with ``zstd``. Chunks below 768KiB are always compressed single-threaded: libzstd will not use a compression job smaller than 512KiB, so a small chunk gets split very unevenly and multi-threading it would be slower than not doing it at all. + The default is capped at 4 because a chunk of the size the default chunker aims at + (2MiB) splits into just 4 such jobs: threads beyond that get (nearly) no work, but + the whole thread pool is created again for every chunk. Measured on a 12-core + machine, 4 threads beat 12 on every test corpus at the default ``zstd,-4`` + (+13% .. +37%). Raising the value only pays off if you configured the chunker + for much bigger chunks. ``borg export-tar`` compresses one long stream instead of + separate chunks and always defaults to the cpu count. Multi-threading trades a little compression ratio for speed (measured at ``zstd,3``: +0.05% archive size for 1MiB chunks, +0.64% for 8MiB ones, more at higher levels), and it uses more cpu time in total to reduce the wallclock time. Set it to 1 if you would rather have the smaller archive, or if borg has to share the cpu with other work. + Single-threaded can even be faster on data zstd races through anyway, e.g. + already-compressed/incompressible data or long-repeat data like VM images. BORG_FASTCDC_KERNEL / BORG_BUZHASH64_KERNEL Select the scan kernel the ``fastcdc`` / ``buzhash64`` chunker uses (default: ``scalar``, the plain sequential loop). Accepted values are ``avx512``, ``avx2``, diff --git a/src/borg/archiver/tar_cmds.py b/src/borg/archiver/tar_cmds.py index f3fd0a8dd5..f4a49ed627 100644 --- a/src/borg/archiver/tar_cmds.py +++ b/src/borg/archiver/tar_cmds.py @@ -188,7 +188,7 @@ def create_zstd_filter(stream, stream_close, decompress): if decompress: zstream = zstd.ZstdFile(stream, "rb") else: - workers = get_zstd_mt_workers() + workers = get_zstd_mt_workers(stream=True) if workers > 1: params = zstd.CompressionParameter options = {params.compression_level: ZSTD_TAR_LEVEL, params.nb_workers: workers} diff --git a/src/borg/compress.pyi b/src/borg/compress.pyi index d11b7bff0f..e3575840ea 100644 --- a/src/borg/compress.pyi +++ b/src/borg/compress.pyi @@ -4,7 +4,7 @@ ZSTD_JOB_SIZE_MIN: int ZSTD_MT_MIN_SIZE: int def get_compressor(name: str, **kwargs) -> Any: ... -def get_zstd_mt_workers() -> int: ... +def get_zstd_mt_workers(stream: bool = ...) -> int: ... class Compressor: def __init__(self, name: Any = ..., **kwargs) -> None: ... diff --git a/src/borg/compress.pyx b/src/borg/compress.pyx index c31014dcb4..766b7ab756 100644 --- a/src/borg/compress.pyx +++ b/src/borg/compress.pyx @@ -53,17 +53,30 @@ ZSTD_JOB_SIZE_MIN = 512 * 1024 # 768KiB (= 1.5 jobs) keeps some margin over break-even, as the overhead is machine dependent. ZSTD_MT_MIN_SIZE = 768 * 1024 -_zstd_mt_workers = None +_zstd_mt_workers = None # cached (chunk workers, stream workers) -def get_zstd_mt_workers(): - """How many threads libzstd may use to compress a single chunk. +def get_zstd_mt_workers(stream=False): + """How many threads libzstd may use to compress a single chunk (or stream). - Defaults to the cpu count, BORG_ZSTD_MT_WORKERS overrides it. 0 or 1 means - single-threaded compression, which also avoids the small loss of compression ratio - that splitting a chunk into jobs causes (measured at zstd,3: +0.05% for a 1MiB chunk, - +0.64% for an 8MiB one; higher levels lose a bit more as they rely on longer match - history). + BORG_ZSTD_MT_WORKERS overrides the defaults below (for chunks and streams alike). + 0 or 1 means single-threaded compression, which also avoids the small loss of + compression ratio that splitting a chunk into jobs causes (measured at zstd,3: + +0.05% for a 1MiB chunk, +0.64% for an 8MiB one; higher levels lose a bit more as + they rely on longer match history). + + For chunks (stream=False), the default is the cpu count, but at most 4: a chunk + only yields ceil(size / ZSTD_JOB_SIZE_MIN) jobs - 4 for the 2 MiB chunks the + default chunker aims at - so threads beyond that get (nearly) no work, while the + whole thread pool is still created and torn down again for every single chunk. + Measured on a 12-core machine, 4 workers beat 12 on every corpus tested at the + default zstd,-4, by +13% (source code) .. +37% (VM image) big-chunk throughput; + higher levels showed smaller differences, but no clear win for 12 anywhere. + Raise the value if you configured the chunker for much bigger chunks. + + For long streams (stream=True, used by export-tar), the default is the cpu count: + one pool with libzstd's default (large) job size compresses the whole stream, so + there are enough jobs and the pool overhead is paid only once. This is called for every chunk, so the result is cached. The env var is evaluated on first use rather than at import time, so that an invalid value is reported via borg's @@ -72,17 +85,19 @@ def get_zstd_mt_workers(): global _zstd_mt_workers if _zstd_mt_workers is None: value = os.environ.get("BORG_ZSTD_MT_WORKERS") + cpus = os.cpu_count() or 1 if value is None: - workers = os.cpu_count() or 1 + workers = (min(cpus, 4), cpus) else: try: - workers = int(value) + configured = int(value) except ValueError: raise Error(f"BORG_ZSTD_MT_WORKERS must be an integer, but is: {value!r}") from None - if workers < 0: - raise Error(f"BORG_ZSTD_MT_WORKERS must not be negative, but is: {workers}") + if configured < 0: + raise Error(f"BORG_ZSTD_MT_WORKERS must not be negative, but is: {configured}") + workers = (configured, configured) _zstd_mt_workers = workers - return _zstd_mt_workers + return _zstd_mt_workers[1 if stream else 0] cdef extern from "lz4.h": int LZ4_compress_default(const char* source, char* dest, int inputSize, int maxOutputSize) nogil diff --git a/src/borg/testsuite/compress_test.py b/src/borg/testsuite/compress_test.py index 77981d94f3..73318fe0a9 100644 --- a/src/borg/testsuite/compress_test.py +++ b/src/borg/testsuite/compress_test.py @@ -339,17 +339,20 @@ def test_robj_specific_obfuscation(data_length, expected_padding, robj_type): def test_zstd_mt_workers_from_env(monkeypatch): - def workers_for(env_value): + def workers_for(env_value, stream=False): monkeypatch.setattr("borg.compress._zstd_mt_workers", None) # drop the cache if env_value is None: monkeypatch.delenv("BORG_ZSTD_MT_WORKERS", raising=False) else: monkeypatch.setenv("BORG_ZSTD_MT_WORKERS", env_value) - return get_zstd_mt_workers() - - assert workers_for(None) == (os.cpu_count() or 1) - for value, expected in [("0", 0), ("1", 1), ("4", 4)]: - assert workers_for(value) == expected + return get_zstd_mt_workers(stream=stream) + + cpus = os.cpu_count() or 1 + assert workers_for(None) == min(cpus, 4) # per-chunk default is capped + assert workers_for(None, stream=True) == cpus # stream default is not + for value, expected in [("0", 0), ("1", 1), ("4", 4), ("12", 12)]: + assert workers_for(value) == expected # the env var is not capped + assert workers_for(value, stream=True) == expected for invalid in ["", "yes", "4x", "1.5", "-1"]: with pytest.raises(Error): workers_for(invalid) From 9ef369534e934a8ed5517c7d4cbb0f4559d1a57c Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Sat, 15 Aug 2026 00:23:01 +0200 Subject: [PATCH 131/144] tests: keep pytest out of the chunkers testsuite package __init__ borg.selftest imports the *_self_test.py modules from src/borg/testsuite/chunkers/, so that package's __init__ runs on every borg invocation - and a normal borg install has no pytest. Since 8f2de9e80 the __init__ had a hard "import pytest", needed only by the chunker_with_kernel() helper, so a non-editable install without the dev dependencies died at startup: File "borg/archiver/__init__.py", line 51, in from ..selftest import selftest File "borg/selftest.py", line 24, in from .testsuite.chunkers.buzhash_self_test import ChunkerTestCase File "borg/testsuite/chunkers/__init__.py", line 4, in import pytest ModuleNotFoundError: No module named 'pytest' chunker_with_kernel() moves into a separate pytest_helpers module that only the pytest-based tests import, so the self-test modules can be imported without pytest again. The guard in selftest() could not catch this: "pytest" not in dir(module) only inspects the leaf self-test module, never a package __init__ it imports - as its own comment says. Co-Authored-By: Claude Opus 5 --- src/borg/testsuite/chunkers/__init__.py | 24 +++-------------- src/borg/testsuite/chunkers/buzhash64_test.py | 3 ++- src/borg/testsuite/chunkers/fastcdc_test.py | 3 ++- .../testsuite/chunkers/phte_chunkers_test.py | 3 ++- src/borg/testsuite/chunkers/pytest_helpers.py | 26 +++++++++++++++++++ 5 files changed, 36 insertions(+), 23 deletions(-) create mode 100644 src/borg/testsuite/chunkers/pytest_helpers.py diff --git a/src/borg/testsuite/chunkers/__init__.py b/src/borg/testsuite/chunkers/__init__.py index 7dd2d7ac9d..a48d12a3eb 100644 --- a/src/borg/testsuite/chunkers/__init__.py +++ b/src/borg/testsuite/chunkers/__init__.py @@ -1,31 +1,15 @@ +# Note: this is imported by borg.selftest via the *_self_test.py modules in this package, +# so it must not import pytest - a normal borg install does not have it. +# Helpers that need pytest belong into pytest_helpers.py. + import os import tempfile -import pytest - from borg.constants import * # noqa from ...chunkers import has_seek_hole -def chunker_with_kernel(monkeypatch, envvar, kernel, make): - """Build a chunker with set to , or skip if it cannot run here. - - Which kernels exist depends on the build (compiler support) and on the CPU, so a - kernel that cannot be selected is not a failure - the test is skipped, carrying - the reason the chunker gave, so `pytest -rs` shows which kernels a machine covered. - Requesting an unusable kernel raising (rather than falling back) is itself tested, - see test_kernel_env_rejects_unusable. - """ - monkeypatch.setenv(envvar, kernel) - try: - chunker = make() - except ValueError as err: - pytest.skip(str(err)) - assert chunker.kernel == kernel # we must have got what we asked for, not a fallback - return chunker - - def cf(chunks): """Chunk filter.""" diff --git a/src/borg/testsuite/chunkers/buzhash64_test.py b/src/borg/testsuite/chunkers/buzhash64_test.py index 66adcfc701..92e21e1eea 100644 --- a/src/borg/testsuite/chunkers/buzhash64_test.py +++ b/src/borg/testsuite/chunkers/buzhash64_test.py @@ -5,7 +5,8 @@ import pytest -from . import cf, cf_expand, chunker_with_kernel +from . import cf, cf_expand +from .pytest_helpers import chunker_with_kernel from ...chunkers import ChunkerBuzHash64 from ...chunkers.buzhash64 import buzhash64_get_table from ...constants import * # NOQA diff --git a/src/borg/testsuite/chunkers/fastcdc_test.py b/src/borg/testsuite/chunkers/fastcdc_test.py index bb8bdb6a4b..c5eef568f3 100644 --- a/src/borg/testsuite/chunkers/fastcdc_test.py +++ b/src/borg/testsuite/chunkers/fastcdc_test.py @@ -5,7 +5,8 @@ import pytest -from . import cf, cf_expand, chunker_with_kernel +from . import cf, cf_expand +from .pytest_helpers import chunker_with_kernel from ...chunkers import ChunkerFastCDC, get_chunker from ...chunkers.fastcdc import fastcdc_get_gear_table from ...constants import * # NOQA diff --git a/src/borg/testsuite/chunkers/phte_chunkers_test.py b/src/borg/testsuite/chunkers/phte_chunkers_test.py index dda8354e2b..7279f8890c 100644 --- a/src/borg/testsuite/chunkers/phte_chunkers_test.py +++ b/src/borg/testsuite/chunkers/phte_chunkers_test.py @@ -14,7 +14,8 @@ import pytest -from . import cf, cf_expand, chunker_with_kernel +from . import cf, cf_expand +from .pytest_helpers import chunker_with_kernel from ...chunkers import ChunkerRabinAES, ChunkerGoldilocksAES, ChunkerToeplitzAES, get_chunker from ...constants import * # NOQA from ...helpers import hex_to_bin diff --git a/src/borg/testsuite/chunkers/pytest_helpers.py b/src/borg/testsuite/chunkers/pytest_helpers.py new file mode 100644 index 0000000000..b1f8d31900 --- /dev/null +++ b/src/borg/testsuite/chunkers/pytest_helpers.py @@ -0,0 +1,26 @@ +"""Chunker test helpers that require pytest. + +These must not live in this package's __init__.py: borg.selftest imports the *_self_test.py +modules from this package, so the package __init__ runs on every borg invocation - and a +normal borg install has no pytest. +""" + +import pytest + + +def chunker_with_kernel(monkeypatch, envvar, kernel, make): + """Build a chunker with set to , or skip if it cannot run here. + + Which kernels exist depends on the build (compiler support) and on the CPU, so a + kernel that cannot be selected is not a failure - the test is skipped, carrying + the reason the chunker gave, so `pytest -rs` shows which kernels a machine covered. + Requesting an unusable kernel raising (rather than falling back) is itself tested, + see test_kernel_env_rejects_unusable. + """ + monkeypatch.setenv(envvar, kernel) + try: + chunker = make() + except ValueError as err: + pytest.skip(str(err)) + assert chunker.kernel == kernel # we must have got what we asked for, not a fallback + return chunker From d8abb7492acfb6de1ef57ef70162ea4c3fd9133b Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Sat, 15 Aug 2026 00:43:46 +0200 Subject: [PATCH 132/144] CI: add big-endian (s390x) test under qemu emulation All our CI machines are little-endian, so the big-endian code paths in the native code (e.g. the __builtin_bswap64 calls in the chunker kernels) are never executed and the architecture independence of the repository format is not tested at all. Build and test borg on s390x in an emulated container, running the tests of the endianness sensitive parts (chunkers, crypto, compression, hashindex, item, repository format). Additionally run a cross-architecture interoperability test: a repository written on the big-endian side is checked, extracted and compared on the little-endian side and the other way round. Both sides also have to produce identical chunk ids for the same data - if they did not, the archives would still be correct, but deduplication between machines of different endianness would silently not work. Emulating everything is slow (~35min), so this does not run for every pull request, but only for pull requests touching native or format relevant code, plus weekly and on demand. --- .github/workflows/bigendian.yml | 200 +++++++++++++++++++++++++ scripts/endian_interop_test.py | 257 ++++++++++++++++++++++++++++++++ 2 files changed, 457 insertions(+) create mode 100644 .github/workflows/bigendian.yml create mode 100644 scripts/endian_interop_test.py diff --git a/.github/workflows/bigendian.yml b/.github/workflows/bigendian.yml new file mode 100644 index 0000000000..be8d18d0b1 --- /dev/null +++ b/.github/workflows/bigendian.yml @@ -0,0 +1,200 @@ +# Big-endian test: build and test borg on s390x (big-endian) under qemu +# user-mode emulation, and check that a repository written on a big-endian +# machine can be read on a little-endian one and the other way round. +# +# Why: borg's repository format is architecture independent and the native code +# has explicit big-endian code paths (e.g. the __builtin_bswap64 calls in the +# chunker kernels), but every other machine we test on is little-endian, so +# those code paths are never executed and a bug in them would only be found by +# the users of s390x, some ppc64 and some mips machines. +# +# Everything inside the container is emulated instruction by instruction, so +# this is slow. Therefore it does not run for every pull request, but only when +# native code or format relevant code was touched (plus weekly and on demand). + +name: Big-endian + +on: + pull_request: + branches: [ master ] + paths: + - '**.pyx' + - '**.pxd' + - 'src/borg/**/*.c' + - 'src/borg/**/*.h' + - 'src/borg/chunkers/**' + - 'src/borg/crypto/**' + - 'src/borg/repository.py' + - 'src/borg/repoobj.py' + - 'src/borg/archive.py' + - 'scripts/endian_interop_test.py' + - '.github/workflows/bigendian.yml' + schedule: + - cron: '43 5 * * 3' # Wednesdays at 05:43 UTC + workflow_dispatch: # Allow manual trigger + +concurrency: + group: ${{ github.workflow }}-${{ github.head_ref || github.ref }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + +permissions: + contents: read + +env: + PY_COLORS: "1" + +jobs: + s390x: + name: s390x (big-endian, emulated) + runs-on: ubuntu-24.04 + timeout-minutes: 120 + + env: + # Debian trixie has python 3.13, and all our dependencies that ship binary + # wheels (blake3, backports-zstd, PyYAML, cffi) have s390x wheels for it - + # only the small C/Cython extensions are compiled (emulated) from source. + IMAGE: docker.io/s390x/debian:trixie-slim + CONTAINER: borg-s390x + # both sides of the interoperability test share this directory: + INTEROP: ${{ github.workspace }}/.interop + # a hung emulated test must fail instead of eating the job timeout: + PYTEST_TIMEOUT: "600" + # The endianness sensitive parts: the native code (chunkers, crypto, + # compression, hashindex, item) and the code that reads/writes the + # repository format. Extend this if it turns out to be too narrow. + PYTEST_TARGETS: >- + borg.testsuite.chunkers + borg.testsuite.crypto + borg.testsuite.compress_test + borg.testsuite.hashindex_test + borg.testsuite.item_test + borg.testsuite.repoobj_test + borg.testsuite.repository_test + borg.testsuite.archive_test + borg.testsuite.helpers.msgpack_test + + steps: + - uses: actions/checkout@v7 + with: + # Just fetching one commit is not enough for setuptools-scm, so we fetch all. + fetch-depth: 0 + fetch-tags: true + + - name: Set up Python + uses: actions/setup-python@v7 + with: + # same python version as in the container, so both sides are comparable + python-version: '3.13' + + # The wheels pip has to build from source inside the emulated container + # (msgpack, argon2-cffi-bindings, borghash, borgstore) are the expensive + # part of the container setup, so keep them from run to run. + - name: Cache pip-built wheels (s390x) + uses: actions/cache@v6 + with: + path: .pip-cache-s390x + key: s390x-pip-${{ hashFiles('pyproject.toml') }} + restore-keys: | + s390x-pip- + + - name: Install Linux packages + run: | + sudo apt-get update + sudo apt-get install -y pkg-config build-essential + sudo apt-get install -y libssl-dev libacl1-dev liblz4-dev + + - name: Build borg (native, little-endian) + run: | + set -euxo pipefail + # Note: a non-editable install on purpose - the s390x build below uses + # the same source tree and an editable install would put the extension + # modules of one architecture in there for the other one to pick up. + python -m venv "$RUNNER_TEMP/venv-native" + "$RUNNER_TEMP/venv-native/bin/pip" install --upgrade pip wheel + # Note: no pytest and no other development dependencies in this venv on + # purpose - this is what a normal borg installation looks like. + "$RUNNER_TEMP/venv-native/bin/pip" install . + "$RUNNER_TEMP/venv-native/bin/borg" --version + "$RUNNER_TEMP/venv-native/bin/python" -c 'import sys; assert sys.byteorder == "little", sys.byteorder' + # the container has no git, so tell setuptools-scm the version we got here: + echo "BORG_VERSION=$("$RUNNER_TEMP/venv-native/bin/borg" --version | cut -d' ' -f2)" >> $GITHUB_ENV + + - name: Set up QEMU + uses: docker/setup-qemu-action@v3 + with: + platforms: s390x + + - name: Start the s390x container + run: | + set -euxo pipefail + mkdir -p .pip-cache-s390x + docker run -d --name "$CONTAINER" --platform linux/s390x \ + -v "${{ github.workspace }}:/borg" -w /borg \ + -e PIP_CACHE_DIR=/borg/.pip-cache-s390x \ + -e SETUPTOOLS_SCM_PRETEND_VERSION_FOR_BORGBACKUP="$BORG_VERSION" \ + -e HOST_UID="$(id -u)" -e HOST_GID="$(id -g)" \ + "$IMAGE" sleep infinity + # if this says anything but s390x, the emulation did not kick in: + test "$(docker exec "$CONTAINER" uname -m)" = "s390x" + + - name: Build borg (s390x, big-endian) + run: | + docker exec -i "$CONTAINER" bash -s <<'EOF' + set -euxo pipefail + export DEBIAN_FRONTEND=noninteractive + apt-get update + apt-get install -y --no-install-recommends \ + python3 python3-dev python3-venv \ + build-essential pkg-config libssl-dev libacl1-dev liblz4-dev + python3 -c 'import sys; assert sys.byteorder == "big", sys.byteorder; print("byteorder:", sys.byteorder)' + python3 -m venv /venv + /venv/bin/pip install --upgrade pip wheel + /venv/bin/pip install pytest pytest-xdist pytest-benchmark pytest-timeout + /venv/bin/pip install /borg + /venv/bin/borg --version + chown -R "$HOST_UID:$HOST_GID" /borg/.pip-cache-s390x + EOF + + - name: Run the endianness sensitive tests on s390x + run: | + docker exec -i \ + -e PYTEST_TARGETS="$PYTEST_TARGETS" \ + -e PYTEST_TIMEOUT="$PYTEST_TIMEOUT" \ + -e PY_COLORS="$PY_COLORS" \ + "$CONTAINER" bash -s <<'EOF' + set -euxo pipefail + # run against the installed borg, not against the source tree: + cd /tmp + /venv/bin/python -m pytest -v -n4 -rs --benchmark-skip --pyargs $PYTEST_TARGETS + EOF + + - name: Write test data and a repository on s390x + run: | + set -euxo pipefail + "$RUNNER_TEMP/venv-native/bin/python" scripts/endian_interop_test.py testdata "$INTEROP" + docker exec -i -e INTEROP=/borg/.interop "$CONTAINER" bash -s <<'EOF' + set -euxo pipefail + BORG=/venv/bin/borg /venv/bin/python /borg/scripts/endian_interop_test.py write "$INTEROP" be + chown -R "$HOST_UID:$HOST_GID" "$INTEROP" + EOF + + - name: Read the s390x repository on x86_64 (and write to it) + run: | + set -euxo pipefail + export BORG="$RUNNER_TEMP/venv-native/bin/borg" + PY="$RUNNER_TEMP/venv-native/bin/python" + $PY scripts/endian_interop_test.py verify "$INTEROP" be --as le + $PY scripts/endian_interop_test.py write "$INTEROP" le + $PY scripts/endian_interop_test.py compare "$INTEROP" be le + + - name: Read the x86_64 archives on s390x + run: | + docker exec -i -e INTEROP=/borg/.interop "$CONTAINER" bash -s <<'EOF' + set -euxo pipefail + BORG=/venv/bin/borg /venv/bin/python /borg/scripts/endian_interop_test.py verify "$INTEROP" le --as be + chown -R "$HOST_UID:$HOST_GID" "$INTEROP" + EOF + + - name: Stop the s390x container + if: ${{ always() }} + run: docker rm -f "$CONTAINER" || true diff --git a/scripts/endian_interop_test.py b/scripts/endian_interop_test.py new file mode 100644 index 0000000000..e9935f22c0 --- /dev/null +++ b/scripts/endian_interop_test.py @@ -0,0 +1,257 @@ +#!/usr/bin/env python3 +""" +Cross-architecture repository interoperability test. + +borg's on-disk / on-the-wire format is architecture independent, but almost all +of our CI runs on little-endian machines only, so the big-endian code paths in +the native code (e.g. the __builtin_bswap64() in the chunker kernels) are never +executed. This script drives a repository from two sides, so a repository that +was written on one architecture can be verified on the other one: + + # on machine/container A (e.g. big-endian): + endian_interop_test.py testdata WORKDIR # deterministic test data + endian_interop_test.py write WORKDIR be # write archives "be-*" + + # on machine/container B (e.g. little-endian), same WORKDIR: + endian_interop_test.py verify WORKDIR be # read what A wrote + endian_interop_test.py write WORKDIR le # write archives "le-*" + endian_interop_test.py compare WORKDIR be le # same data, same chunks? + + # on machine/container A again: + endian_interop_test.py verify WORKDIR le # read what B wrote + +"verify" runs "borg check --verify-data", extracts the other side's archives and +compares the extracted files against the test data. "compare" additionally +requires that both sides cut the data into exactly the same chunks with exactly +the same chunk IDs - if they did not, the archives would still be correct, but +deduplication between machines of different endianness would silently not work. + +The environment (BORG_REPO, BORG_PASSPHRASE, cache and config dir) is set up by +this script; the borg to use can be given via the BORG environment variable. +""" + +import argparse +import hashlib +import json +import os +import random +import shutil +import subprocess +import sys + +# fastcdc is borg's default chunker (and its kernel is one of the places with a +# byte-order dependent code path). Much smaller chunks than the default: the +# test data then gets cut into thousands of chunks == thousands of chances to +# notice a byte-order dependent cut point. +# +# The second archive uses different chunker parameters on purpose: with the same +# parameters it would just deduplicate against the first one and its compression +# would never be exercised. So both sides also have to agree about the chunk +# format of more than one compression method. +ARCHIVE_SPECS = [ + ("fastcdc-zstd", ["--chunker-params", "fastcdc,10,16,12,2", "--compression", "zstd,3"]), + ("fastcdc-lz4", ["--chunker-params", "fastcdc,12,18,14,2", "--compression", "lz4"]), +] + +PASSPHRASE = "borg-endian-interop-test" + +# the test data lives in this subdirectory of the work directory +DATA = "data" + + +def borg(workdir, label, *args, cwd=None, capture=False): + """run a borg command against the test repository""" + env = os.environ.copy() + env["BORG_REPO"] = os.path.join(workdir, "repo") + env["BORG_PASSPHRASE"] = PASSPHRASE + # the KDF does not need to be strong for a throwaway test repository and + # argon2 is *slow* under emulation: + env["BORG_TESTONLY_WEAKEN_KDF"] = "1" + # keep the two sides (and the machine's own borg config) apart: + env["BORG_BASE_DIR"] = os.path.join(workdir, "home-%s" % label) + env["BORG_CACHE_DIR"] = os.path.join(workdir, "home-%s" % label, "cache") + env["BORG_CONFIG_DIR"] = os.path.join(workdir, "home-%s" % label, "config") + cmd = [os.environ.get("BORG", "borg")] + [str(a) for a in args] + print("+ %s" % " ".join(cmd), flush=True) + if capture: + return subprocess.run(cmd, env=env, cwd=cwd, check=True, stdout=subprocess.PIPE).stdout + subprocess.run(cmd, env=env, cwd=cwd, check=True) + return None + + +def datadir(workdir): + return os.path.join(workdir, DATA) + + +def cmd_testdata(args): + """create deterministic test data - content must not depend on the machine""" + path = datadir(args.workdir) + if os.path.exists(path): + shutil.rmtree(path) + os.makedirs(os.path.join(path, "nested", "dir")) + rnd = random.Random(20260814) + + # incompressible, gets cut into many chunks: + with open(os.path.join(path, "random.bin"), "wb") as f: + f.write(rnd.randbytes(8 * 1024 * 1024)) + # compressible, but not uniform - many different chunker cut points: + with open(os.path.join(path, "text.bin"), "wb") as f: + for i in range(64 * 1024): + f.write(b"line %d: %s\n" % (i, b"borg" * (i % 17))) + # all-zero data (the chunkers have a shortcut for this): + with open(os.path.join(path, "zeros.bin"), "wb") as f: + f.write(b"\0" * (2 * 1024 * 1024)) + # the same data as random.bin, but shifted by a few bytes: the rolling hash + # has to find the same cut points again after the insertion. + with open(os.path.join(path, "random.bin"), "rb") as f: + data = f.read() + with open(os.path.join(path, "nested", "shifted.bin"), "wb") as f: + f.write(b"insertion" + data) + with open(os.path.join(path, "nested", "dir", "small.txt"), "wb") as f: + f.write(b"hello borg\n") + with open(os.path.join(path, "nested", "empty"), "wb"): + pass + with open(os.path.join(path, "nested", "\u00fcmlaut-\u65e5\u672c\u8a9e.txt"), "wb") as f: + f.write("non-ascii file name and content: \u00e4\u00f6\u00fc\n".encode()) + os.symlink("dir/small.txt", os.path.join(path, "nested", "symlink")) + os.link(os.path.join(path, "nested", "dir", "small.txt"), os.path.join(path, "nested", "hardlink")) + print("test data created in %s" % path) + + +def scan(path): + """map relative path -> file content hash / symlink target, for comparing 2 trees""" + result = {} + for dirpath, dirnames, filenames in os.walk(path): + dirnames.sort() + for name in sorted(dirnames + filenames): + full = os.path.join(dirpath, name) + rel = os.path.relpath(full, path) + if os.path.islink(full): + result[rel] = "symlink:%s" % os.readlink(full) + elif os.path.isdir(full): + result[rel] = "dir" + else: + digest = hashlib.sha256() + with open(full, "rb") as f: + for block in iter(lambda: f.read(1024 * 1024), b""): + digest.update(block) + result[rel] = "file:%s" % digest.hexdigest() + return result + + +def dump_path(workdir, archive, label): + return os.path.join(workdir, "dumps", "%s.read-by-%s.json" % (archive, label)) + + +def dump_archive(workdir, label, archive): + """dump the archive metadata as read by *this* machine""" + os.makedirs(os.path.join(workdir, "dumps"), exist_ok=True) + borg(workdir, label, "debug", "dump-archive", archive, dump_path(workdir, archive, label)) + + +def chunk_list(dump_file): + """(path, size, chunks) of all items, everything that must not depend on the architecture""" + with open(dump_file) as f: + dump = json.load(f) + items = [] + for item in dump["_items"]: + items.append((item["path"], item.get("size"), item.get("target"), item.get("chunks"))) + return items + + +def cmd_write(args): + workdir, label = args.workdir, args.label + if not os.path.exists(os.path.join(workdir, "repo")): + borg(workdir, label, "repo-create", "--encryption=aes256-ocb") + for suffix, options in ARCHIVE_SPECS: + archive = "%s-%s" % (label, suffix) + # --files-cache=disabled: really chunk the files, do not trust any cache. + # Archive the relative path "data" from within the work directory: the + # two sides usually see the work directory at different absolute paths + # (e.g. one of them inside a container), but the item paths in the + # archives must be comparable. + borg(workdir, label, "create", "--files-cache=disabled", *options, archive, DATA, cwd=workdir) + dump_archive(workdir, label, archive) + borg(workdir, label, "repo-list") + + +def cmd_verify(args): + """check and extract the archives written by the *other* side""" + workdir, other, label = args.workdir, args.label, args.as_label + borg(workdir, label, "check", "--verify-data") + expected = scan(datadir(workdir)) + assert expected, "no test data found in %s" % datadir(workdir) + for suffix, _ in ARCHIVE_SPECS: + archive = "%s-%s" % (other, suffix) + extract_to = os.path.join(workdir, "extract-%s-by-%s" % (archive, label)) + if os.path.exists(extract_to): + shutil.rmtree(extract_to) + os.makedirs(extract_to) + borg(workdir, label, "extract", archive, cwd=extract_to) + got = scan(os.path.join(extract_to, DATA)) + if got != expected: + for key in sorted(set(expected) | set(got)): + if expected.get(key) != got.get(key): + print("MISMATCH %s: expected %r, got %r" % (key, expected.get(key), got.get(key))) + raise SystemExit("archive %s does not extract to the original test data!" % archive) + print("OK: %s extracted identically (%d entries)" % (archive, len(got))) + # the same archive, but read (and its metadata decoded) on this machine: + dump_archive(workdir, label, archive) + written_by_other = chunk_list(dump_path(workdir, archive, other)) + read_by_us = chunk_list(dump_path(workdir, archive, label)) + if written_by_other != read_by_us: + raise SystemExit("archive %s does not decode identically on both architectures!" % archive) + print("OK: %s metadata (incl. chunk ids) decodes identically on both architectures" % archive) + + +def cmd_compare(args): + """both sides archived the same data: they must have produced the same chunks""" + workdir, a, b = args.workdir, args.label, args.other_label + for suffix, _ in ARCHIVE_SPECS: + archive_a, archive_b = "%s-%s" % (a, suffix), "%s-%s" % (b, suffix) + items_a = chunk_list(dump_path(workdir, archive_a, a)) + items_b = chunk_list(dump_path(workdir, archive_b, b)) + if items_a != items_b: + for item_a, item_b in zip(items_a, items_b): + if item_a != item_b: + print("MISMATCH %s:\n %s: %r\n %s: %r" % (item_a[0], a, item_a[1:], b, item_b[1:])) + raise SystemExit( + "%s and %s chunked the same data differently - " + "deduplication between these architectures would not work!" % (archive_a, archive_b) + ) + chunks = sum(len(item[3] or []) for item in items_a) + print("OK: %s and %s have identical items and chunk ids (%d chunks)" % (archive_a, archive_b, chunks)) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + commands = parser.add_subparsers(dest="command", required=True) + + sub = commands.add_parser("testdata", help="create the deterministic test data") + sub.add_argument("workdir") + sub.set_defaults(func=cmd_testdata) + + sub = commands.add_parser("write", help="create the repository and this side's archives") + sub.add_argument("workdir") + sub.add_argument("label", help="label of this side, e.g. 'be'") + sub.set_defaults(func=cmd_write) + + sub = commands.add_parser("verify", help="check/extract the archives the other side wrote") + sub.add_argument("workdir") + sub.add_argument("label", help="label of the *other* side, e.g. 'be'") + sub.add_argument("--as", dest="as_label", required=True, help="label of this side, e.g. 'le'") + sub.set_defaults(func=cmd_verify) + + sub = commands.add_parser("compare", help="compare the chunk ids both sides produced") + sub.add_argument("workdir") + sub.add_argument("label") + sub.add_argument("other_label") + sub.set_defaults(func=cmd_compare) + + args = parser.parse_args() + os.makedirs(args.workdir, exist_ok=True) + args.func(args) + + +if __name__ == "__main__": + sys.exit(main()) From 49bed0c4b545f181e1393d2deb53d8e816c35614 Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Sat, 15 Aug 2026 07:56:28 +0200 Subject: [PATCH 133/144] CI: give the black workflow read-only token permissions It was the only workflow without a permissions declaration, so it got whatever the repository default is. It only checks out and runs black. --- .github/workflows/black.yaml | 3 +++ 1 file changed, 3 insertions(+) diff --git a/.github/workflows/black.yaml b/.github/workflows/black.yaml index 94c22ea57d..146c3beba3 100644 --- a/.github/workflows/black.yaml +++ b/.github/workflows/black.yaml @@ -19,6 +19,9 @@ concurrency: group: ${{ github.workflow }}-${{ github.head_ref || github.ref }} cancel-in-progress: ${{ github.event_name == 'pull_request' }} +permissions: + contents: read + jobs: lint: runs-on: ubuntu-24.04 From abd9c3d88cb92f00b735892be74d058b108131f3 Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Sat, 15 Aug 2026 08:52:14 +0200 Subject: [PATCH 134/144] CI: release automation for PyPI and GitHub releases Pushing a release tag so far only built the standalone binaries and uploaded them as workflow artifacts - everything else was manual: build and sign the sdist, upload it to PyPI, fetch the binary artifacts and create the GitHub release with them. Add a release job (.github/workflows/release.yml), called by ci.yml for tags after the tests and the binary builds succeeded - it runs inside the same workflow run, because that is where the binary artifacts are. It builds the sdist, verifies that the sdist is complete by installing it into a clean venv and running borg, attests its provenance and drafts the GitHub release with the sdist and all standalone binaries attached. A pypi job then uploads the sdist to PyPI via trusted publishing, so no API token has to be stored anywhere. It lives in ci.yml because PyPI trusted publishing does not work from a reusable workflow. It uses the "pypi" environment: configuring required reviewers for it makes the irreversible upload wait for an approval. The Windows binaries now use the same naming and layout as the binaries of the other platforms (borg-windows-x86_64-gh.exe and .tgz, with the single-directory variant and a provenance attestation), so they become release assets, too. The GitHub release is created as a draft: the release notes want a human, the detached GPG signature of the sdist can only be made locally, and the binaries should be tried out before the release becomes visible. --- .github/workflows/ci.yml | 95 ++++++++++++++++++- .github/workflows/release.yml | 171 ++++++++++++++++++++++++++++++++++ docs/development.rst | 37 ++++---- 3 files changed, 280 insertions(+), 23 deletions(-) create mode 100644 .github/workflows/release.yml diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 334ef1ca13..edbec097ce 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -685,6 +685,11 @@ jobs: timeout-minutes: 90 needs: [lint] + permissions: + contents: read + id-token: write + attestations: write + env: MSYS2_ARG_CONV_EXCL: "*" MSYS2_ENV_CONV_EXCL: "*" @@ -738,10 +743,36 @@ jobs: # build sdist and wheel in dist/... python -m build - - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + # Same layout and naming as the binaries of the other platforms, so that + # the release job picks this up as a release asset, too. The single-file + # binary keeps its .exe extension - Windows needs it to run the file. + - name: Prepare binaries (borg-windows-x86_64-gh) + run: | + pushd dist/binary + echo "single-file binary" + ./borg.exe -V + echo "single-directory binary" + ./borg-dir/borg.exe -V + tar czf borg.tgz borg-dir + popd + mkdir -p artifacts + cp dist/binary/borg.exe artifacts/borg-windows-x86_64-gh.exe + cp dist/binary/borg.tgz artifacts/borg-windows-x86_64-gh.tgz + echo "binary files" + ls -l artifacts/ + + - name: Attest binaries provenance (borg-windows-x86_64-gh) + if: ${{ startsWith(github.ref, 'refs/tags/') }} + uses: actions/attest-build-provenance@4d101475d8b20a2381f78447822ac1eab6504dd8 # v4.2.2 with: - name: borg-windows - path: dist/binary/borg.exe + subject-path: 'artifacts/*' + + - name: Upload binaries (borg-windows-x86_64-gh) + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: borg-windows-x86_64-gh + path: artifacts/* + if-no-files-found: error - name: Run tests run: | @@ -774,3 +805,61 @@ jobs: report_type: coverage env_vars: OS,python files: coverage.xml + + release: + # Build the sdist and draft the GitHub release with the binaries the jobs + # above built for this tag, see .github/workflows/release.yml. + # + # vm_tests is continue-on-error (a flaky BSD VM must not stop a release), so + # this waits for it, but only requires native_tests to have succeeded. A + # binary that did not get built is warned about while drafting the release. + if: ${{ !cancelled() && startsWith(github.ref, 'refs/tags/') && needs.native_tests.result == 'success' }} + needs: [native_tests, vm_tests] + + permissions: + contents: write + id-token: write + attestations: write + + uses: ./.github/workflows/release.yml + + pypi: + # Upload the sdist the release job built to PyPI. + # + # This job can not live in release.yml with the rest of the release code: + # PyPI trusted publishing does not work from a reusable workflow, see + # https://docs.pypi.org/trusted-publishers/troubleshooting/ + # + # One-time setup, so that no API token has to be stored anywhere: + # - on pypi.org, add a trusted publisher to the "borgbackup" project: + # owner "borgbackup", repository "borg", workflow "ci.yml", + # environment "pypi". + # - create the "pypi" environment in the repository settings. Configuring + # required reviewers for it makes this (irreversible) upload wait for an + # approval, which is the last chance to stop a release. + if: ${{ startsWith(github.ref, 'refs/tags/') && needs.release.result == 'success' }} + needs: [release] + + runs-on: ubuntu-24.04 + timeout-minutes: 30 + + environment: + name: pypi + url: https://pypi.org/project/borgbackup/ + + permissions: + contents: read + id-token: write # trusted publishing + + steps: + - name: Get the sdist built by the release job + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: sdist + path: dist + + - name: What we are about to upload + run: ls -l dist/ + + - name: Upload to PyPI + uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # v1.14.2 diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml new file mode 100644 index 0000000000..4a9a29ebb8 --- /dev/null +++ b/.github/workflows/release.yml @@ -0,0 +1,171 @@ +# Release automation: build the source distribution, publish it to PyPI and +# create a GitHub release with the standalone binaries. +# +# This is called by ci.yml when a tag is pushed, and it deliberately runs as +# part of that same workflow run: that is where the binaries for the tag are +# built, and artifacts can only be downloaded within the run that created them. +# +# The GitHub release is created as a *draft* on purpose: +# - the release notes want a human, +# - the detached GPG signature of the sdist can only be made locally +# (scripts/sdist-sign), so it has to be added by hand, +# - and the binaries should be tried out before the release becomes visible. +# Publishing the draft is a single click in the GitHub UI. +# +# The upload to PyPI is *not* done here, but by the "pypi" job in ci.yml: PyPI +# trusted publishing can not be used from a reusable workflow, see +# https://docs.pypi.org/trusted-publishers/troubleshooting/ - so that job has to +# live in a workflow that is triggered by an event. This job hands the sdist +# over to it as a workflow artifact. + +name: Release + +on: + workflow_call: + +permissions: + contents: read + +env: + # The standalone binaries expected for a release: for every platform the + # single-file binary and, as a .tgz, the single-directory variant. See the + # "binary" entries of the native_tests matrix, the "artifact_prefix" entries + # of the vm_tests matrix and the windows_tests job in ci.yml - keep in sync. + EXPECTED_ASSETS: >- + borg-linux-glibc239-x86_64-gh + borg-linux-glibc239-x86_64-gh.tgz + borg-linux-glibc239-arm64-gh + borg-linux-glibc239-arm64-gh.tgz + borg-macos-15-arm64-gh + borg-macos-15-arm64-gh.tgz + borg-macos-15-x86_64-gh + borg-macos-15-x86_64-gh.tgz + borg-freebsd-15-x86_64-gh + borg-freebsd-15-x86_64-gh.tgz + borg-windows-x86_64-gh.exe + borg-windows-x86_64-gh.tgz + +jobs: + github_release: + name: Draft the GitHub release + runs-on: ubuntu-24.04 + timeout-minutes: 30 + + permissions: + contents: write # to create the release + id-token: write # to attest the sdist + attestations: write + + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + # Just fetching one commit is not enough for setuptools-scm, so we fetch all. + fetch-depth: 0 + fetch-tags: true + + - name: Set up Python + uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 + with: + python-version: '3.13' + + - name: Install Linux packages + run: | + sudo apt-get update + sudo apt-get install -y pkg-config build-essential + sudo apt-get install -y libssl-dev libacl1-dev liblz4-dev + + - name: Build the sdist + run: | + set -euxo pipefail + python -m pip install --upgrade pip build twine + # only a sdist: we do not publish wheels, they would be platform specific. + python -m build --sdist + twine check dist/* + ls -l dist/ + + - name: Check that the sdist is complete and installable + # A release that cannot be installed from PyPI is the worst kind of + # release, and nothing else in CI ever installs borg from a sdist or + # without the development requirements. + run: | + set -euxo pipefail + python -m venv "$RUNNER_TEMP/venv-sdist" + "$RUNNER_TEMP/venv-sdist/bin/pip" install --upgrade pip + "$RUNNER_TEMP/venv-sdist/bin/pip" install dist/borgbackup-*.tar.gz + "$RUNNER_TEMP/venv-sdist/bin/borg" --version + "$RUNNER_TEMP/venv-sdist/bin/borg" --help > /dev/null + + - name: Attest the sdist provenance + uses: actions/attest-build-provenance@4d101475d8b20a2381f78447822ac1eab6504dd8 # v4.2.2 + with: + subject-path: 'dist/*.tar.gz' + + - name: Download the binaries built for this tag + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + path: assets + # the binary artifacts are all named like the binaries they contain + pattern: 'borg-*-gh' + merge-multiple: true + + - name: Check that no binary is missing + run: | + set -uo pipefail + ls -l assets/ || true + missing="" + for asset in $EXPECTED_ASSETS; do + test -f "assets/$asset" || missing="$missing $asset" + done + if [ -n "$missing" ]; then + # not an error: vm_tests is continue-on-error, so e.g. a flaky FreeBSD + # VM should not stop the release - but do not lose it silently either. + echo "::warning::binaries missing from this release:$missing" + fi + + - name: Create the draft release + env: + GH_TOKEN: ${{ github.token }} + TAG: ${{ github.ref_name }} + run: | + set -euxo pipefail + # 2.0.0b23 and friends are pre-releases, 2.1.0 is not. + prerelease="" + case "$TAG" in *a*|*b*|*rc*) prerelease="--prerelease" ;; esac + cat > release-notes.md <\`. + EOF + mkdir -p assets # there may not have been any artifact to download + if gh release view "$TAG" > /dev/null 2>&1; then + # a re-run of this job: keep the (possibly already edited) release and + # just replace its assets. + gh release upload "$TAG" --clobber dist/*.tar.gz $(find assets -type f | sort) + else + gh release create "$TAG" \ + --draft $prerelease \ + --title "borg $TAG" \ + --notes-file release-notes.md \ + dist/*.tar.gz $(find assets -type f | sort) + fi + gh release view "$TAG" --json isDraft,isPrerelease,assets + + - name: Keep the sdist for the PyPI upload + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: sdist + path: dist/*.tar.gz + if-no-files-found: error diff --git a/docs/development.rst b/docs/development.rst index 4a07ff70f9..34910e56da 100644 --- a/docs/development.rst +++ b/docs/development.rst @@ -531,22 +531,29 @@ Checklist: - Optional: run tox and/or binary builds on all supported platforms via vagrant, check for test failures. This is now optional as we do platform testing and binary building on GitHub. -- Create sdist, sign it, upload release to (test) PyPi: +- When GitHub CI looks good on the release PR, merge it and push the release tag. - :: + Pushing the tag makes CI build the standalone binaries and then release: + the ``release`` job (``.github/workflows/release.yml``) builds the sdist, + checks that it can be installed, attests its provenance and drafts the GitHub + release with the sdist and all standalone binaries attached. The ``pypi`` job + then uploads the sdist to PyPI. + + If the ``pypi`` environment has required reviewers configured, the upload to + PyPI waits for an approval - that is the last chance to stop it. Also watch + out for a warning while the release is drafted: it means that a binary is + missing because its build did not succeed. +- Sign the sdist and add the signature to the drafted GitHub release:: scripts/sdist-sign X.Y.Z - scripts/upload-pypi X.Y.Z test - scripts/upload-pypi X.Y.Z Note: the signature is not uploaded to PyPi any more, but we upload it to github releases. -- When GitHub CI looks good on the release PR, merge it and then check "Actions": - GitHub will create binary assets after the release PR is merged within the - CI testing of the merge. Check the "Upload binaries" step on Ubuntu (AMD/Intel - and ARM64) and macOS (Intel and ARM64), fetch the ZIPs with the binaries. -- Unpack the ZIPs and test the binaries, upload the binaries to the GitHub - release page (borg-OS-SPEC-ARCH-gh and borg-OS-SPEC-ARCH-gh.tgz). +- Download the binaries from the drafted release and test them. For macOS + binaries **with** FUSE support, document the macFUSE version in the release + notes: macFUSE uses a kernel extension that needs to be compatible with the + code contained in the binary. +- Review the release notes and publish the drafted GitHub release. - Close the release milestone on GitHub. - `Update borgbackup.org @@ -557,13 +564,3 @@ Checklist: - Mailing list. - Mastodon / BlueSky / X (aka Twitter). - IRC channel (change ``/topic``). - -- Create a GitHub release, include: - - - pypi dist package and signature - - Standalone binaries (see above for how to create them). - - - For macOS binaries **with** FUSE support, document the macFUSE version - in the README of the binaries. macFUSE uses a kernel extension that needs - to be compatible with the code contained in the binary. - - A link to ``CHANGES.rst``. From d0baab45d1f3236f7be205888bc3a3cac0dec909 Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Sat, 15 Aug 2026 13:20:48 +0200 Subject: [PATCH 135/144] windows: restore timestamps via SetFileTime, fixes #7269 os.utime is a bad fit for restoring timestamps on Windows: - it can not set the birthtime (creation time), so the birthtime was archived (Python >= 3.12 has st_birthtime_ns there), but never restored. - it does not support follow_symlinks=False there, so the timestamps of an extracted symlink were set on the symlink's target. - it does not support file descriptors there, so we had to use the path even though we had the file open already. Add platform.set_times() which does all that: the POSIX implementation is the os.utime code moved from Archive.restore_attrs (incl. the utimes trick to set the birthtime on the BSDs and Darwin), the win32 implementation uses CreateFileW / SetFileTime. Also, failing to set the timestamps is not silently ignored on win32 anymore, but warns and sets the warning exit code. --- docs/changes.rst | 6 ++ src/borg/archive.py | 53 +++++--------- src/borg/platform/__init__.py | 2 + src/borg/platform/base.py | 26 +++++++ src/borg/platform/windows.pyx | 69 ++++++++++++++++++ .../testsuite/archiver/extract_cmd_test.py | 22 ++++++ src/borg/testsuite/platform/windows_test.py | 72 ++++++++++++++++++- 7 files changed, 212 insertions(+), 38 deletions(-) diff --git a/docs/changes.rst b/docs/changes.rst index 5ccd15cc17..f88a6a6513 100644 --- a/docs/changes.rst +++ b/docs/changes.rst @@ -259,6 +259,12 @@ Fixes: did not notice content changes of files that kept their size and inode number. The default is ``mtime,size,inode`` there now and an explicitly given ctime based mode warns and uses the corresponding mtime based mode. +- extract: restore the timestamps using SetFileTime on Windows, #7269. + + os.utime can not set the birthtime (creation time) there, nor can it work on a file + descriptor or on a symlink itself. Thus, the birthtime was not restored at all and the + timestamps of a symlink were set on the symlink's target. Also, failing to set the + timestamps is not silently ignored anymore, but gives a warning. - re-add XXH64 to read borg 1.x integrity data, #9935 - list: add {blake3} format key, #9984 - support date: archive patterns for --from-borg1, #9949 diff --git a/src/borg/archive.py b/src/borg/archive.py index db7de5cd03..4303ee6798 100644 --- a/src/borg/archive.py +++ b/src/borg/archive.py @@ -50,7 +50,7 @@ from .patterns import PathPrefixPattern, FnmatchPattern, IECommand from .item import Item, ArchiveItem, ItemDiff from . import platform -from .platform import acl_get, acl_set, set_flags, get_flags, swidth +from .platform import acl_get, acl_set, set_flags, get_flags, set_times, swidth from .repository import Repository, NoManifestError from .repoobj import RepoObj @@ -1096,44 +1096,23 @@ def restore_attrs(self, path, item, symlink=False, fd=None): warning = xattr.set_all(fd or path, item.xattrs, follow_symlinks=False) if warning: set_ec(EXIT_WARNING) - # set timestamps rather late - mtime = item.mtime - atime = item.atime if "atime" in item else mtime - if "birthtime" in item: - birthtime = item.birthtime - try: - # This should work on FreeBSD, NetBSD, and Darwin and be harmless on other platforms. - # See utimes(2) on either of the BSDs for details. - if fd: - os.utime(fd, None, ns=(atime, birthtime)) - else: - os.utime(path, None, ns=(atime, birthtime), follow_symlinks=False) - except OSError: - # some systems don't support calling utime on a symlink - pass - try: - if fd: - os.utime(fd, None, ns=(atime, mtime)) - else: - os.utime(path, None, ns=(atime, mtime), follow_symlinks=False) - except OSError: - # some systems don't support calling utime on a symlink - pass - # bsdflags include the immutable flag and need to be set last: - if not self.noflags and "bsdflags" in item: - try: - set_flags(path, item.bsdflags, fd=fd) - except OSError: - pass - else: # win32 - # set timestamps rather late - mtime = item.mtime - atime = item.atime if "atime" in item else mtime + # set timestamps rather late + mtime = item.mtime + atime = item.atime if "atime" in item else mtime + birthtime = item.get("birthtime") + try: + set_times(path, atime_ns=atime, mtime_ns=mtime, birthtime_ns=birthtime, fd=fd, follow_symlinks=False) + except OSError as e: + # some POSIX systems don't support setting the timestamps of a symlink. + if is_win32: + # win32 can set the timestamps of a symlink itself, so this is a real problem. + logger.warning("%s: when setting timestamps: %s", remove_surrogates(item.path), e) + set_ec(EXIT_WARNING) + # bsdflags include the immutable flag and need to be set last: + if not is_win32 and not self.noflags and "bsdflags" in item: try: - # note: no fd support on win32 - os.utime(path, None, ns=(atime, mtime)) + set_flags(path, item.bsdflags, fd=fd) except OSError: - # some systems don't support calling utime on a symlink pass def set_meta(self, key, value): diff --git a/src/borg/platform/__init__.py b/src/borg/platform/__init__.py index 4ff1b77a07..4495a3444a 100644 --- a/src/borg/platform/__init__.py +++ b/src/borg/platform/__init__.py @@ -12,6 +12,7 @@ from .base import SaveFile, sync_dir, fdatasync, safe_fadvise from .base import get_process_id, fqdn, hostname, hostid, swidth from .base import acl_text_to_xattr # overridden below for platforms supporting it +from .base import set_times # overridden below for win32 # work around pyinstaller "forgetting" to include the xattr module from . import xattr # noqa: F401 @@ -84,6 +85,7 @@ from .base import acl_get, acl_set from .base import set_flags, get_flags from .windows import SyncFile + from .windows import set_times # type: ignore[no-redef] from .windows import process_alive, local_pid_alive from .windows import getosusername from . import windows_ug as platform_ug diff --git a/src/borg/platform/base.py b/src/borg/platform/base.py index 21e8bf9081..38152f82d1 100644 --- a/src/borg/platform/base.py +++ b/src/borg/platform/base.py @@ -112,6 +112,32 @@ def get_flags(path, st, fd=None): return getattr(st, "st_flags", 0) +def set_times(path, *, atime_ns, mtime_ns, birthtime_ns=None, fd=None, follow_symlinks=True): + """ + Set the timestamps of *path* (or of the open file descriptor *fd*, if given). + + *birthtime_ns* is only honoured on platforms that can set the birthtime (creation time), + it is silently ignored on the other ones. + + Raises OSError if the timestamps could not be set. + """ + if birthtime_ns is not None: + try: + # This should work on FreeBSD, NetBSD, and Darwin and be harmless on other platforms. + # See utimes(2) on either of the BSDs for details. + if fd is not None: + os.utime(fd, None, ns=(atime_ns, birthtime_ns)) + else: + os.utime(path, None, ns=(atime_ns, birthtime_ns), follow_symlinks=follow_symlinks) + except OSError: + # some systems don't support calling utime on a symlink + pass + if fd is not None: + os.utime(fd, None, ns=(atime_ns, mtime_ns)) + else: + os.utime(path, None, ns=(atime_ns, mtime_ns), follow_symlinks=follow_symlinks) + + def sync_dir(path): if is_win32: # Opening directories is not supported on Windows. diff --git a/src/borg/platform/windows.pyx b/src/borg/platform/windows.pyx index c7a192a385..075b181346 100644 --- a/src/borg/platform/windows.pyx +++ b/src/borg/platform/windows.pyx @@ -22,12 +22,24 @@ cdef extern from 'windows.h': # Win32 API constants for CreateFileW GENERIC_READ = 0x80000000 GENERIC_WRITE = 0x40000000 +FILE_WRITE_ATTRIBUTES = 0x00000100 FILE_SHARE_READ = 0x00000001 +FILE_SHARE_WRITE = 0x00000002 +FILE_SHARE_DELETE = 0x00000004 CREATE_NEW = 1 +OPEN_EXISTING = 3 FILE_ATTRIBUTE_NORMAL = 0x80 FILE_FLAG_WRITE_THROUGH = 0x80000000 +FILE_FLAG_OPEN_REPARSE_POINT = 0x00200000 +FILE_FLAG_BACKUP_SEMANTICS = 0x02000000 ERROR_FILE_EXISTS = 80 +# a FILETIME counts 100ns intervals since 1601-01-01, our timestamps count ns since 1970-01-01. +FILETIME_EPOCH_OFFSET_NS = 11644473600 * 1000000000 +# SetFileTime gives a special meaning to 0 ("do not change") and to all bits set +# ("do not update this timestamp for this handle any more"), so avoid these values. +FILETIME_MIN, FILETIME_MAX = 1, 0xFFFFFFFFFFFFFFFE + _kernel32 = ctypes.WinDLL("kernel32", use_last_error=True) _CreateFileW = _kernel32.CreateFileW _CreateFileW.restype = ctypes.wintypes.HANDLE @@ -41,6 +53,14 @@ _CreateFileW.argtypes = [ ctypes.wintypes.HANDLE, ] _CloseHandle = _kernel32.CloseHandle +_SetFileTime = _kernel32.SetFileTime +_SetFileTime.restype = ctypes.wintypes.BOOL +_SetFileTime.argtypes = [ + ctypes.wintypes.HANDLE, + ctypes.POINTER(ctypes.wintypes.FILETIME), + ctypes.POINTER(ctypes.wintypes.FILETIME), + ctypes.POINTER(ctypes.wintypes.FILETIME), +] INVALID_HANDLE_VALUE = ctypes.wintypes.HANDLE(-1).value @@ -103,6 +123,55 @@ class SyncFile(BaseSyncFile): os.fsync(self.fd) +def _ns_to_filetime(ns): + """Convert a timestamp in ns since 1970-01-01 to a Win32 FILETIME.""" + ft = (ns + FILETIME_EPOCH_OFFSET_NS) // 100 + # clamp to what a FILETIME can express (borg timestamps have a much wider range). + ft = min(max(ft, FILETIME_MIN), FILETIME_MAX) + return ctypes.wintypes.FILETIME(ft & 0xFFFFFFFF, ft >> 32) + + +def set_times(path, *, atime_ns, mtime_ns, birthtime_ns=None, fd=None, follow_symlinks=True): + """ + Set the timestamps of *path* (or of the open file descriptor *fd*, if given). + + Uses SetFileTime rather than os.utime, because os.utime on Windows can neither work on + a file descriptor nor on a symlink itself, nor can it set the birthtime (creation time). + + Raises OSError if the timestamps could not be set. + """ + if fd is not None: + handle, close_handle = msvcrt.get_osfhandle(fd), False + else: + # FILE_FLAG_BACKUP_SEMANTICS is required to get a handle for a directory. + flags = FILE_FLAG_BACKUP_SEMANTICS + if not follow_symlinks: + flags |= FILE_FLAG_OPEN_REPARSE_POINT + handle = _CreateFileW( + str(path), + FILE_WRITE_ATTRIBUTES, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + None, + OPEN_EXISTING, + flags, + None, + ) + if handle == INVALID_HANDLE_VALUE: + raise ctypes.WinError(ctypes.get_last_error()) + close_handle = True + try: + atime = _ns_to_filetime(atime_ns) + mtime = _ns_to_filetime(mtime_ns) + birthtime = _ns_to_filetime(birthtime_ns) if birthtime_ns is not None else None + if not _SetFileTime( + handle, ctypes.byref(birthtime) if birthtime is not None else None, ctypes.byref(atime), ctypes.byref(mtime) + ): + raise ctypes.WinError(ctypes.get_last_error()) + finally: + if close_handle: + _CloseHandle(handle) + + def getosusername(): """Return the OS username.""" return os.getlogin() diff --git a/src/borg/testsuite/archiver/extract_cmd_test.py b/src/borg/testsuite/archiver/extract_cmd_test.py index d2a3e302c9..0e1668d426 100644 --- a/src/borg/testsuite/archiver/extract_cmd_test.py +++ b/src/borg/testsuite/archiver/extract_cmd_test.py @@ -326,6 +326,28 @@ def test_birthtime(archivers, request): assert same_ts_ns(sto.st_mtime_ns, mtime * 10**9) +@pytest.mark.skipif(not is_win32, reason="Windows-only test") +def test_timestamps_win32(archivers, request): + """windows: extract restores atime, mtime and (with Python >= 3.12) birthtime, see #7269""" + archiver = request.getfixturevalue(archivers) + create_regular_file(archiver.input_path, "file", contents=b"stuff") + # os.utime can not set the birthtime on Windows, so use our own platform code for the setup. + # a FILETIME has a resolution of 100ns, thus only use timestamps that are a multiple of 100ns. + atime_ns, mtime_ns, birthtime_ns = 1500000000000000100, 1400000000000000200, 1300000000000000300 + platform.set_times("input/file", atime_ns=atime_ns, mtime_ns=mtime_ns, birthtime_ns=birthtime_ns) + cmd(archiver, "repo-create", RK_ENCRYPTION) + cmd(archiver, "create", "--atime", "test", "input") + sti = os.stat("input/file") + with changedir("output"): + cmd(archiver, "extract", "test") + sto = os.stat("output/input/file") + assert sto.st_mtime_ns == mtime_ns + # reading the input file for the backup might have updated its atime, so compare with the input file. + assert sto.st_atime_ns == sti.st_atime_ns + if hasattr(sti, "st_birthtime_ns"): # Python >= 3.12 + assert sto.st_birthtime_ns == birthtime_ns + + def test_sparse_file(archivers, request): archiver = request.getfixturevalue(archivers) diff --git a/src/borg/testsuite/platform/windows_test.py b/src/borg/testsuite/platform/windows_test.py index eeb6a4d0e1..4f61919689 100644 --- a/src/borg/testsuite/platform/windows_test.py +++ b/src/borg/testsuite/platform/windows_test.py @@ -1,13 +1,19 @@ +import os import tempfile import pytest from .platform_test import skipif_not_win32 -from ...platform import SyncFile +from .. import are_symlinks_supported +from ...platform import SyncFile, set_times # Set module-level skips pytestmark = [skipif_not_win32] +# timestamps used by the set_times tests. a FILETIME has a resolution of 100ns, +# so only use values that are a multiple of 100ns to be able to compare exactly. +ATIME_NS, MTIME_NS, BIRTHTIME_NS = 1500000000000000100, 1400000000000000200, 1300000000000000300 + def test_syncfile_basic(tmp_path): """Integration: SyncFile creates file and writes data correctly.""" @@ -70,3 +76,67 @@ def mock_create(*args): assert len(calls) == 1 flags_attrs = calls[0][5] # 6th arg: dwFlagsAndAttributes assert flags_attrs & windows.FILE_FLAG_WRITE_THROUGH + + +def assert_times(st, *, atime_ns=ATIME_NS, mtime_ns=MTIME_NS, birthtime_ns=BIRTHTIME_NS): + assert st.st_atime_ns == atime_ns + assert st.st_mtime_ns == mtime_ns + if hasattr(st, "st_birthtime_ns"): # Python >= 3.12 on Windows + assert st.st_birthtime_ns == birthtime_ns + + +def test_set_times_file(tmp_path): + """set_times sets atime, mtime and birthtime of a file given by path.""" + path = tmp_path / "file" + path.write_bytes(b"data") + set_times(str(path), atime_ns=ATIME_NS, mtime_ns=MTIME_NS, birthtime_ns=BIRTHTIME_NS) + assert_times(os.stat(path)) + + +def test_set_times_fd(tmp_path): + """set_times works on an open file descriptor, like borg extract uses it.""" + path = tmp_path / "file" + with open(path, "wb") as f: + f.write(b"data") + f.flush() + set_times(str(path), atime_ns=ATIME_NS, mtime_ns=MTIME_NS, birthtime_ns=BIRTHTIME_NS, fd=f.fileno()) + # the timestamps must survive closing the file we have written to. + assert_times(os.stat(path)) + + +def test_set_times_directory(tmp_path): + """set_times works on a directory (needs FILE_FLAG_BACKUP_SEMANTICS).""" + path = tmp_path / "dir" + path.mkdir() + set_times(str(path), atime_ns=ATIME_NS, mtime_ns=MTIME_NS, birthtime_ns=BIRTHTIME_NS) + assert_times(os.stat(path)) + + +def test_set_times_without_birthtime(tmp_path): + """set_times leaves the birthtime alone if we do not give one.""" + path = tmp_path / "file" + path.write_bytes(b"data") + birthtime_ns = getattr(os.stat(path), "st_birthtime_ns", None) + set_times(str(path), atime_ns=ATIME_NS, mtime_ns=MTIME_NS) + assert_times(os.stat(path), birthtime_ns=birthtime_ns) + + +@pytest.mark.skipif(not are_symlinks_supported(), reason="symlinks not supported") +def test_set_times_symlink_not_followed(tmp_path): + """set_times(follow_symlinks=False) works on the symlink itself, not on its target.""" + target = tmp_path / "target" + target.write_bytes(b"data") + target_times = os.stat(target) + link = tmp_path / "link" + os.symlink(str(target), str(link)) + set_times(str(link), atime_ns=ATIME_NS, mtime_ns=MTIME_NS, birthtime_ns=BIRTHTIME_NS, follow_symlinks=False) + assert_times(os.stat(link, follow_symlinks=False)) + st_target = os.stat(target) + assert st_target.st_mtime_ns == target_times.st_mtime_ns + assert st_target.st_atime_ns == target_times.st_atime_ns + + +def test_set_times_nonexistent(tmp_path): + """set_times raises OSError if there is no such file.""" + with pytest.raises(OSError): + set_times(str(tmp_path / "nonexistent"), atime_ns=ATIME_NS, mtime_ns=MTIME_NS) From 64c79e113fbdbc96f8ae1fa0fa66235399d43076 Mon Sep 17 00:00:00 2001 From: Mrityunjay Raj Date: Sat, 15 Aug 2026 18:44:24 +0530 Subject: [PATCH 136/144] check --repair: report index-referenced missing packs as errors, not a corrupt index --- src/borg/repository.py | 15 ++++++++------- 1 file changed, 8 insertions(+), 7 deletions(-) diff --git a/src/borg/repository.py b/src/borg/repository.py index 354f6943aa..a007f7561f 100644 --- a/src/borg/repository.py +++ b/src/borg/repository.py @@ -1195,8 +1195,6 @@ def recorded_ts(info): # run: sig_int breaks the loop early, so "no pack errors" must be paired with "all packs # scanned" (pack_files == len(pack_infos)) to not rebuild from unverified packs. if index_errors and pack_errors == 0 and not sig_int and pack_files == len(pack_infos): - from .cache import build_chunkindex_from_repo - # the exclusive check lock keeps the pack set fixed, so re-listing packs/ inside # build_chunkindex_from_repo matches this verification. write_immediately persists the # index and drops the corrupt fragments. @@ -1240,7 +1238,7 @@ def recorded_ts(info): logger.info(f"{done} {mode} repository check, no problems found{so_far}.") elif not repair: logger.error(f"{done} {mode} repository check, errors found{so_far}.") - elif index_repaired and not (pack_errors or corrupt_ids): + elif index_repaired and not (pack_errors or corrupt_ids or missing_pack_ids): # the index was the only problem and it has been rebuilt from the packs. logger.info(f"{done} {mode} repository check, repaired{so_far}.") elif pack_errors or corrupt_ids: @@ -1253,16 +1251,19 @@ def recorded_ts(info): # a full check's archives phase reads archive/item metadata (and file content with # --verify-data), so it repairs a corrupt pack holding such objects; warn rather than fail. logger.warning(f"{done} {mode} repository check, corrupt pack(s) found{so_far}.") - else: + elif index_errors and not index_repaired: # the index is corrupt but was not rebuilt, e.g. the pack verification was interrupted # before every pack was confirmed intact; the corrupt index is left in place. logger.error(f"{done} {mode} repository check, index still corrupt{so_far}.") - # in repair mode a corrupt index left unrebuilt is a failure; a corrupt pack fails only a - # repository-only run, while a full check defers it to the archives phase. + else: + # index-referenced packs are missing, so their chunks are lost. + logger.error(f"{done} {mode} repository check, errors found{so_far}.") + # in repair mode a corrupt index left unrebuilt is a failure; a corrupt or missing pack fails + # only a repository-only run, while a full check defers it to the archives phase. if repair: if index_errors and not index_repaired: return False - return not (repo_only and (pack_errors or corrupt_ids)) + return not (repo_only and (pack_errors or corrupt_ids or missing_pack_ids)) return not problems def list(self, limit=None, marker=None): From 9bf6d5ce9acf93af5217ccf8351734d869d3b8ea Mon Sep 17 00:00:00 2001 From: Mrityunjay Raj Date: Sat, 15 Aug 2026 18:54:18 +0530 Subject: [PATCH 137/144] check --repair: test the interrupted-rebuild guard and missing-pack error path --- src/borg/testsuite/repository_test.py | 47 +++++++++++++++++++++++++++ 1 file changed, 47 insertions(+) diff --git a/src/borg/testsuite/repository_test.py b/src/borg/testsuite/repository_test.py index c9020dd851..2352e5ad79 100644 --- a/src/borg/testsuite/repository_test.py +++ b/src/borg/testsuite/repository_test.py @@ -1132,6 +1132,53 @@ def test_check_repair_refuses_when_pack_corrupt(tmp_path): assert repository.check(repair=False) is False # index was not rebuilt; still corrupt +def test_check_repair_leaves_index_when_interrupted(tmp_path, caplog, monkeypatch): + # an interrupted repair (SIGINT before every pack is verified) must not rebuild the index from + # packs it did not confirm intact: it leaves the corrupt index in place and fails. + location = os.fspath(tmp_path / "repo") + ids = [H(x) for x in range(10)] + with Repository(location, exclusive=True, create=True) as repository: + for i, cid in enumerate(ids): + repository.put(cid, fchunk(bytes([i]) * 20, chunk_id=cid)) + repository.flush() # seal the pack(s) and let close() persist the index + with reopen(repository) as repository: + for info in repository.store_list("index"): # rot every fragment so repair takes the rebuild path + name = f"index/{info.name}" + data = bytearray(repository.store_load(name)) + data[0] ^= 0xFF + repository.store_store(name, bytes(data)) + with reopen(repository) as repository: + monkeypatch.setattr("borg.repository.sig_int", True) # simulate a SIGINT before the pack loop + with caplog.at_level(logging.ERROR, logger="borg.repository"): + assert repository.check(repair=True) is False # interrupted: index not rebuilt, so it fails + assert "index still corrupt" in caplog.text + with reopen(repository) as repository: + assert repository.check(repair=False) is False # repair left the index corrupt + + +def test_check_repair_reports_missing_pack_as_error(tmp_path, caplog): + # a repair with an intact index but a pack the index references missing from packs/ reports the + # loss and fails a repository-only run; a full check defers it to the archives phase (refs #9898, + # #8572). + location = os.fspath(tmp_path / "repo") + with Repository(location, exclusive=True, create=True) as repository: + for x in range(3): + repository.put(H(x), fchunk(b"DATA-%02d" % x, chunk_id=H(x))) + repository.flush() # flush before close persists the index + with reopen(repository) as repository: + pack_id = repository.chunks[H(0)].pack_id + repository.store_delete("packs/" + bin_to_hex(pack_id)) # pack gone, index entry kept + with reopen(repository) as repository: + # a repository-only repair cannot recover the lost chunks, so it fails and reports the error. + with caplog.at_level(logging.ERROR, logger="borg.repository"): + assert repository.check(repair=True, repo_only=True) is False + assert f"Missing pack: {bin_to_hex(pack_id)}" in caplog.text + assert "errors found" in caplog.text + with reopen(repository) as repository: + # a full check defers the missing pack to the archives phase, so the repository phase passes. + assert repository.check(repair=True, repo_only=False) is True + + def test_check_warns_on_invalid_chunk_index(tmp_path, caplog): # check warns about an invalid chunk index but does not fail, since the index is not part of # the repository's object integrity. From 7433e1c4c4d4fd15590a3498f7b1c98224026d35 Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Sat, 15 Aug 2026 12:42:36 +0200 Subject: [PATCH 138/144] create: follow symlinks given as recursion roots, fixes #4737 If a recursion root (a path given on the command line or via a patterns file) is a symlink, borg now follows it and archives what it points to, using the path as given. Symlinks encountered while recursing are not affected, they are archived as symlinks as before. Symlinks given via --paths-from-* are never followed either. Archiving the symlink target under the symlink's path also keeps the files cache working if the symlink target changes its name (e.g. a 'current' symlink pointing to the latest of a series of directories). A recursion root that is a symlink with a non-existing target is skipped with a warning (new BackupBrokenSymlinkError, rc 112). As a followed symlink must be opened without O_NOFOLLOW, add flags_dir_follow and flags_normal_follow and let the callers choose the flags. --- docs/internals/frontends.rst | 2 + docs/man/borg-create.1 | 50 +++++++- docs/usage/create.rst.inc | 22 +++- src/borg/archive.py | 10 +- src/borg/archiver/create_cmd.py | 106 ++++++++++++++--- src/borg/helpers/__init__.py | 4 +- src/borg/helpers/errors.py | 8 ++ src/borg/helpers/fs.py | 2 + .../testsuite/archiver/create_cmd_test.py | 110 +++++++++++++++++- 9 files changed, 288 insertions(+), 26 deletions(-) diff --git a/docs/internals/frontends.rst b/docs/internals/frontends.rst index f3162735fc..fea84ca753 100644 --- a/docs/internals/frontends.rst +++ b/docs/internals/frontends.rst @@ -819,6 +819,8 @@ Warnings {}: {} BackupTimeoutError rc: 111 {}: {} + BackupBrokenSymlinkError rc: 112 + {}: {} Operations - cache.begin_transaction diff --git a/docs/man/borg-create.1 b/docs/man/borg-create.1 index a5fc9ece74..7729fbbfad 100644 --- a/docs/man/borg-create.1 +++ b/docs/man/borg-create.1 @@ -28,7 +28,7 @@ level margin: \\n[rst2man-indent\\n[rst2man-indent-level]] .\" new: \\n[rst2man-indent\\n[rst2man-indent-level]] .in \\n[rst2man-indent\\n[rst2man-indent-level]]u .. -.TH "borg-create" "1" "2026-08-13" "" "borg backup tool" +.TH "borg-create" "1" "2026-08-15" "" "borg backup tool" .SH Name borg-create \- Creates a new archive. .SH SYNOPSIS @@ -46,6 +46,23 @@ The slashdot hack in paths (recursion roots) is triggered by using \fB/./\fP: strip the prefix on the left side of \fB\&./\fP from the archived items (in this case, \fBthis/gets/archived\fP will be the path in the archived item). .sp +If a recursion root (a path given on the command line or in a patterns file) is a +symlink, borg follows it and archives what it points to \- using the path you gave. +If \fBcurrent\fP is a symlink pointing to the directory \fB20260801\-2345\fP, +\fBborg create ARCHIVE current\fP thus archives \fBcurrent\fP as a directory (with the +metadata of \fB20260801\-2345\fP) and recurses into it, archiving the contained fs +objects as \fBcurrent/...\fP\&. As the archived paths do not change when the symlink +target changes, the files cache keeps working for such backups. +.sp +Note that the symlink itself is then not in the archive (and neither is its target +path), so restoring will create a real directory (or file) where the symlink was. +If you want the symlink archived as a symlink, do not give it as a recursion root, +but let borg find it while recursing (symlinks found that way are never followed). +A recursion root that is a symlink with a non\-existing target is skipped with a warning. +.sp +If you give both a symlink and its target as recursion roots, borg archives the fs +objects only once, under the path given first (like for any other root given twice). +.sp When specifying \(aq\-\(aq as a path, borg will read data from standard input and create a file named \(aqstdin\(aq in the created archive from that data. In some cases, it is more appropriate to use \-\-content\-from\-command. See the section \fIReading from stdin\fP @@ -71,9 +88,9 @@ the files cache. This comparison can operate in different modes as given by \fB\-\-files\-cache\fP: .INDENT 0.0 .IP \(bu 2 -ctime,size,inode (default) +ctime,size,inode (default on POSIX systems) .IP \(bu 2 -mtime,size,inode (default behaviour of borg versions older than 1.1.0rc4) +mtime,size,inode (default on Windows) .IP \(bu 2 ctime,size (ignore the inode number) .IP \(bu 2 @@ -109,6 +126,13 @@ can be arbitrarily set from userspace, e.g., to set mtime back to the same value it had before a content change happened. This can be used maliciously as well as well\-meant, but in both cases mtime\-based cache modes can be problematic. .UNINDENT +.sp +On Windows, ctime is the file \fIcreation\fP time, not the \(dqmetadata change time\(dq it is +on POSIX systems. A ctime based mode would therefore not notice content changes of a +file that keeps its size and inode number, so borg defaults to \fBmtime,size,inode\fP +there. If a ctime based mode is given explicitly on Windows, borg warns and uses the +corresponding mtime based mode instead: ctime,size,inode \-> mtime,size,inode, +ctime,size \-> mtime,size, rechunk,ctime \-> rechunk,mtime. .INDENT 0.0 .TP .B The \fB\-\-files\-changed\fP option controls how Borg detects if a file has changed during backup: @@ -134,11 +158,22 @@ The \fB\-\-progress\fP option shows (from left to right) Original and (uncompres deduplicated size (O and U respectively), then the Number of files (N) processed so far, followed by the currently processed path. .sp +Sizes of GB and above are shown with enough decimal places that even MB\-sized progress +stays visible. On a terminal, this needs a width of at least 110 columns \- on narrower +terminals, the compact format is used, so that the path stays readable. If the output +does not go to a terminal (e.g. into a logfile), the precise format is always used. +.sp When using \fB\-\-stats\fP, you will get some statistics about how much data was added \- the \(dqThis Archive\(dq deduplicated size there is most interesting as that is how much your repository will grow. Please note that the \(dqAll archives\(dq stats refer to -the state after creation. Also, the \fB\-\-stats\fP and \fB\-\-dry\-run\fP options are mutually -exclusive because the data is not actually compressed and deduplicated during a dry run. +the state after creation. +.sp +When \fB\-\-stats\fP is used together with \fB\-\-dry\-run\fP, only the number of files and the +original size are reported. They are computed from file system metadata, without reading +the file contents, so a dry run stays fast. As data is not actually read, chunked, and +deduplicated during a dry run, the deduplicated size is unknown. The sizes of data read +from standard input, from a command\(aqs output, or from special files (\fB\-\-read\-special\fP) +are also unknown in a dry run and counted as zero. .sp The \fB\-\-stats\fP output also reports the store statistics (lines prefixed with \(dqStore\(dq), taken from the storage layer after this run. These cover the backend and @@ -275,7 +310,7 @@ do not read and store xattrs into archive detect sparse holes in input (supported only by fixed chunker) .TP .BI \-\-files\-cache \ MODE -operate files cache in MODE. default: ctime,size,inode +operate files cache in MODE. default: ctime,size,inode (on Windows: mtime,size,inode, because ctime is file creation time there). .TP .BI \-\-files\-changed \ MODE specify how to detect if a file has changed during backup (ctime, mtime, disabled). default: ctime (on Windows: mtime, because ctime is file creation time there). @@ -559,6 +594,9 @@ to borg (maybe implementing your own recursion or your own rules), you can use .sp Borg supports paths with the slashdot hack to strip path prefixes here also. So, be careful not to unintentionally trigger that. +.sp +Symlinks given this way are never followed (unlike recursion roots are), they are +archived as symlinks. .SH SEE ALSO .sp \fIborg\-common(1)\fP, \fIborg\-delete(1)\fP, \fIborg\-prune(1)\fP, \fIborg\-check(1)\fP, \fIborg\-patterns(1)\fP, \fIborg\-placeholders(1)\fP, \fIborg\-compression(1)\fP, \fIborg\-repo\-create(1)\fP diff --git a/docs/usage/create.rst.inc b/docs/usage/create.rst.inc index 60a10c10c1..268b6387f1 100644 --- a/docs/usage/create.rst.inc +++ b/docs/usage/create.rst.inc @@ -196,6 +196,23 @@ The slashdot hack in paths (recursion roots) is triggered by using ``/./``: strip the prefix on the left side of ``./`` from the archived items (in this case, ``this/gets/archived`` will be the path in the archived item). +If a recursion root (a path given on the command line or in a patterns file) is a +symlink, borg follows it and archives what it points to - using the path you gave. +If ``current`` is a symlink pointing to the directory ``20260801-2345``, +``borg create ARCHIVE current`` thus archives ``current`` as a directory (with the +metadata of ``20260801-2345``) and recurses into it, archiving the contained fs +objects as ``current/...``. As the archived paths do not change when the symlink +target changes, the files cache keeps working for such backups. + +Note that the symlink itself is then not in the archive (and neither is its target +path), so restoring will create a real directory (or file) where the symlink was. +If you want the symlink archived as a symlink, do not give it as a recursion root, +but let borg find it while recursing (symlinks found that way are never followed). +A recursion root that is a symlink with a non-existing target is skipped with a warning. + +If you give both a symlink and its target as recursion roots, borg archives the fs +objects only once, under the path given first (like for any other root given twice). + When specifying '-' as a path, borg will read data from standard input and create a file named 'stdin' in the created archive from that data. In some cases, it is more appropriate to use --content-from-command. See the section *Reading from stdin* @@ -441,4 +458,7 @@ to borg (maybe implementing your own recursion or your own rules), you can use (with the latter two, borg will fail to create an archive should the command fail). Borg supports paths with the slashdot hack to strip path prefixes here also. -So, be careful not to unintentionally trigger that. \ No newline at end of file +So, be careful not to unintentionally trigger that. + +Symlinks given this way are never followed (unlike recursion roots are), they are +archived as symlinks. \ No newline at end of file diff --git a/src/borg/archive.py b/src/borg/archive.py index db7de5cd03..cd9c65f5cc 100644 --- a/src/borg/archive.py +++ b/src/borg/archive.py @@ -1454,7 +1454,7 @@ def process_dir(self, *, path, parent_fd, name, st, strip_prefix): item.update(self.metadata_collector.stat_attrs(st, path, fd=fd)) return status - def process_fifo(self, *, path, parent_fd, name, st, strip_prefix): + def process_fifo(self, *, path, parent_fd, name, st, strip_prefix, flags=flags_normal): with self.create_helper(path, st, "f", strip_prefix=strip_prefix) as ( item, status, @@ -1463,13 +1463,13 @@ def process_fifo(self, *, path, parent_fd, name, st, strip_prefix): ): # fifo if item is None: return status - with OsOpen(path=path, parent_fd=parent_fd, name=name, flags=flags_normal, noatime=True) as fd: + with OsOpen(path=path, parent_fd=parent_fd, name=name, flags=flags, noatime=True) as fd: with backup_io("fstat"): st = stat_update_check(st, os.fstat(fd)) item.update(self.metadata_collector.stat_attrs(st, path, fd=fd)) return status - def process_dev(self, *, path, parent_fd, name, st, dev_type, strip_prefix): + def process_dev(self, *, path, parent_fd, name, st, dev_type, strip_prefix, follow_symlinks=False): with self.create_helper(path, st, dev_type, strip_prefix=strip_prefix) as ( item, status, @@ -1480,7 +1480,9 @@ def process_dev(self, *, path, parent_fd, name, st, dev_type, strip_prefix): if item is None: return status with backup_io("stat"): - st = stat_update_check(st, os_stat(path=path, parent_fd=parent_fd, name=name, follow_symlinks=False)) + st = stat_update_check( + st, os_stat(path=path, parent_fd=parent_fd, name=name, follow_symlinks=follow_symlinks) + ) item.rdev = st.st_rdev item.update(self.metadata_collector.stat_attrs(st, path)) return status diff --git a/src/borg/archiver/create_cmd.py b/src/borg/archiver/create_cmd.py index 93d316063a..5dbc2ec455 100644 --- a/src/borg/archiver/create_cmd.py +++ b/src/borg/archiver/create_cmd.py @@ -21,10 +21,12 @@ from ..helpers import eval_escapes from ..helpers import timestamp, archive_ts_now from ..helpers import get_cache_dir, os_stat, get_strip_prefix, slashify +from ..helpers import BackupBrokenSymlinkError from ..helpers import dir_is_tagged from ..helpers import log_multi from ..helpers import basic_json_data, json_print, FileSize -from ..helpers import flags_dir, flags_special_follow, flags_special +from ..helpers import flags_dir, flags_dir_follow, flags_special_follow, flags_special +from ..helpers import flags_normal, flags_normal_follow from ..helpers import prepare_subprocess_env from ..helpers import sig_int, ignore_sigint from ..helpers import iter_separated @@ -40,6 +42,24 @@ logger = create_logger() +def stat_root(path): + """ + stat a recursion root, following it if it is a symlink, see #4737. + + Returns (st, followed): for a symlink, st is the stat of what it points to and followed + is True, so the caller knows that it must not use O_NOFOLLOW when opening path. + + Raises BackupBrokenSymlinkError if path is a symlink with a non-existing target. + """ + st = os_stat(path=path, parent_fd=None, name=None, follow_symlinks=False) + if not stat.S_ISLNK(st.st_mode): + return st, False + try: + return os_stat(path=path, parent_fd=None, name=None, follow_symlinks=True), True + except FileNotFoundError: + raise BackupBrokenSymlinkError("stat", "broken symlink, skipping it") from None + + class CreateMixIn: @with_repository(compatibility=(Manifest.Operation.WRITE,)) def do_create(self, args, repository, manifest): @@ -144,6 +164,7 @@ def create_inner(archive, cache, fso): path = posixpath.normpath(path) try: with backup_io("stat"): + # symlinks given this way are never followed, see #4737. st = os_stat(path=path, parent_fd=None, name=None, follow_symlinks=False) status = self._process_any( path=path, @@ -199,7 +220,7 @@ def create_inner(archive, cache, fso): path = posixpath.normpath(path) try: with backup_io("stat"): - st = os_stat(path=path, parent_fd=None, name=None, follow_symlinks=False) + st, followed = stat_root(path) restrict_dev = st.st_dev if args.one_file_system else None self._rec_walk( path=path, @@ -216,6 +237,7 @@ def create_inner(archive, cache, fso): read_special=args.read_special, dry_run=dry_run, strip_prefix=strip_prefix, + follow_symlink=followed, ) # if we get back here, we've finished recursing into , # we do not ever want to get back in there (even if path is given twice as recursion root) @@ -324,9 +346,14 @@ def create_inner(archive, cache, fso): logger=logging.getLogger("borg.output.stats"), ) - def _process_any(self, *, path, parent_fd, name, st, fso, cache, read_special, dry_run, strip_prefix): + def _process_any( + self, *, path, parent_fd, name, st, fso, cache, read_special, dry_run, strip_prefix, followed_symlink=False + ): """ Call the right method on the given FilesystemObjectProcessor. + + If followed_symlink is True, *path* is a symlink we followed (recursion root, see #4737) + and *st* is the stat of its target, so we must not use O_NOFOLLOW when opening it. """ if dry_run: @@ -348,6 +375,9 @@ def _process_any(self, *, path, parent_fd, name, st, fso, cache, read_special, d stats.nfiles += 1 # size unknown without reading the special file return "+" # included MAX_RETRIES = 10 # count includes the initial try (initial try == "retry 0") + # if we followed a symlink, we must not refuse to open its target via the symlink: + flags_file = flags_normal_follow if followed_symlink else flags_normal + flags_specialfile = flags_special_follow if followed_symlink else flags_special for retry in range(MAX_RETRIES): last_try = retry == MAX_RETRIES - 1 try: @@ -358,10 +388,12 @@ def _process_any(self, *, path, parent_fd, name, st, fso, cache, read_special, d name=name, st=st, cache=cache, + flags=flags_file, last_try=last_try, strip_prefix=strip_prefix, ) elif stat.S_ISDIR(st.st_mode): + # note: a followed symlink to a directory does not get here, _rec_walk deals with it. return fso.process_dir(path=path, parent_fd=parent_fd, name=name, st=st, strip_prefix=strip_prefix) elif stat.S_ISLNK(st.st_mode): if not read_special: @@ -393,7 +425,12 @@ def _process_any(self, *, path, parent_fd, name, st, fso, cache, read_special, d elif stat.S_ISFIFO(st.st_mode): if not read_special: return fso.process_fifo( - path=path, parent_fd=parent_fd, name=name, st=st, strip_prefix=strip_prefix + path=path, + parent_fd=parent_fd, + name=name, + st=st, + strip_prefix=strip_prefix, + flags=flags_file, ) else: return fso.process_file( @@ -402,14 +439,20 @@ def _process_any(self, *, path, parent_fd, name, st, fso, cache, read_special, d name=name, st=st, cache=cache, - flags=flags_special, + flags=flags_specialfile, last_try=last_try, strip_prefix=strip_prefix, ) elif stat.S_ISCHR(st.st_mode): if not read_special: return fso.process_dev( - path=path, parent_fd=parent_fd, name=name, st=st, dev_type="c", strip_prefix=strip_prefix + path=path, + parent_fd=parent_fd, + name=name, + st=st, + dev_type="c", + strip_prefix=strip_prefix, + follow_symlinks=followed_symlink, ) else: return fso.process_file( @@ -418,14 +461,20 @@ def _process_any(self, *, path, parent_fd, name, st, fso, cache, read_special, d name=name, st=st, cache=cache, - flags=flags_special, + flags=flags_specialfile, last_try=last_try, strip_prefix=strip_prefix, ) elif stat.S_ISBLK(st.st_mode): if not read_special: return fso.process_dev( - path=path, parent_fd=parent_fd, name=name, st=st, dev_type="b", strip_prefix=strip_prefix + path=path, + parent_fd=parent_fd, + name=name, + st=st, + dev_type="b", + strip_prefix=strip_prefix, + follow_symlinks=followed_symlink, ) else: return fso.process_file( @@ -434,7 +483,7 @@ def _process_any(self, *, path, parent_fd, name, st, fso, cache, read_special, d name=name, st=st, cache=cache, - flags=flags_special, + flags=flags_specialfile, last_try=last_try, strip_prefix=strip_prefix, ) @@ -469,7 +518,7 @@ def _process_any(self, *, path, parent_fd, name, st, fso, cache, read_special, d # mode right (which could have changed due to a race condition and is important for # dispatching) and also to get current inode number of that file. with backup_io("stat"): - st = os_stat(path=path, parent_fd=parent_fd, name=name, follow_symlinks=False) + st = os_stat(path=path, parent_fd=parent_fd, name=name, follow_symlinks=followed_symlink) def _rec_walk( self, @@ -488,10 +537,15 @@ def _rec_walk( read_special, dry_run, strip_prefix, + follow_symlink=False, ): """ Process *path* (or, preferably, parent_fd/name) recursively according to the various parameters. + follow_symlink is only given for a recursion root that is a symlink, see #4737: we then + archive what the symlink points to (using the symlink's path). Symlinks encountered while + recursing are never followed, so this is never passed on to the recursive calls. + This should only raise on critical errors. Per-item errors must be handled within this method. """ if sig_int and sig_int.action_done(): @@ -503,7 +557,7 @@ def _rec_walk( recurse_excluded_dir = False if matcher.match(path): with backup_io("stat"): - st = os_stat(path=path, parent_fd=parent_fd, name=name, follow_symlinks=False) + st = os_stat(path=path, parent_fd=parent_fd, name=name, follow_symlinks=follow_symlink) else: self.print_file_status("-", path) # excluded # get out here as quickly as possible: @@ -514,7 +568,7 @@ def _rec_walk( return recurse_excluded_dir = True with backup_io("stat"): - st = os_stat(path=path, parent_fd=parent_fd, name=name, follow_symlinks=False) + st = os_stat(path=path, parent_fd=parent_fd, name=name, follow_symlinks=follow_symlink) if not stat.S_ISDIR(st.st_mode): return @@ -547,10 +601,16 @@ def _rec_walk( read_special=read_special, dry_run=dry_run, strip_prefix=strip_prefix, + followed_symlink=follow_symlink, ) else: with OsOpen( - path=path, parent_fd=parent_fd, name=name, flags=flags_dir, noatime=True, op="dir_open" + path=path, + parent_fd=parent_fd, + name=name, + flags=flags_dir_follow if follow_symlink else flags_dir, + noatime=True, + op="dir_open", ) as child_fd: # child_fd is None for directories on windows, in that case a race condition check is not possible. if child_fd is not None: @@ -646,6 +706,23 @@ def build_parser_create(self, subparsers, common_parser, mid_common_parser): strip the prefix on the left side of ``./`` from the archived items (in this case, ``this/gets/archived`` will be the path in the archived item). + If a recursion root (a path given on the command line or in a patterns file) is a + symlink, borg follows it and archives what it points to - using the path you gave. + If ``current`` is a symlink pointing to the directory ``20260801-2345``, + ``borg create ARCHIVE current`` thus archives ``current`` as a directory (with the + metadata of ``20260801-2345``) and recurses into it, archiving the contained fs + objects as ``current/...``. As the archived paths do not change when the symlink + target changes, the files cache keeps working for such backups. + + Note that the symlink itself is then not in the archive (and neither is its target + path), so restoring will create a real directory (or file) where the symlink was. + If you want the symlink archived as a symlink, do not give it as a recursion root, + but let borg find it while recursing (symlinks found that way are never followed). + A recursion root that is a symlink with a non-existing target is skipped with a warning. + + If you give both a symlink and its target as recursion roots, borg archives the fs + objects only once, under the path given first (like for any other root given twice). + When specifying '-' as a path, borg will read data from standard input and create a file named 'stdin' in the created archive from that data. In some cases, it is more appropriate to use --content-from-command. See the section *Reading from stdin* @@ -892,6 +969,9 @@ def build_parser_create(self, subparsers, common_parser, mid_common_parser): Borg supports paths with the slashdot hack to strip path prefixes here also. So, be careful not to unintentionally trigger that. + + Symlinks given this way are never followed (unlike recursion roots are), they are + archived as symlinks. """ ) diff --git a/src/borg/helpers/__init__.py b/src/borg/helpers/__init__.py index 898e92a59a..eb2a976098 100644 --- a/src/borg/helpers/__init__.py +++ b/src/borg/helpers/__init__.py @@ -18,12 +18,14 @@ from .errors import BackupError, BackupOSError, BackupRaceConditionError, BackupItemExcluded from .errors import BackupPermissionError, BackupIOError, BackupFileNotFoundError, BackupTimeoutError from .errors import BackupSymlinkParentError, BackupPathTraversalError, BackupHardlinkSourceError +from .errors import BackupBrokenSymlinkError from .fs import ensure_dir, join_base_dir from .fs import get_security_dir, get_keys_dir, get_base_dir, get_cache_dir, get_config_dir, get_runtime_dir from .fs import dir_is_tagged, dir_is_cachedir, remove_dotdot_prefixes, make_path_safe, scandir_inorder from .fs import secure_erase, safe_unlink, dash_open, os_open, os_stat, get_strip_prefix, umount, slashify from .fs import SpecialFileReader -from .fs import O_, flags_dir, flags_special_follow, flags_special, flags_base, flags_normal, flags_noatime +from .fs import O_, flags_dir, flags_dir_follow, flags_special_follow, flags_special +from .fs import flags_base, flags_normal, flags_normal_follow, flags_noatime from .fs import HardLinkManager from .misc import sysinfo, log_multi, consume from .misc import ChunkIteratorFileWrapper, open_item, chunkit, iter_separated, ErrorIgnoringTextIOWrapper diff --git a/src/borg/helpers/errors.py b/src/borg/helpers/errors.py index 72993efb7b..b83fc43aa2 100644 --- a/src/borg/helpers/errors.py +++ b/src/borg/helpers/errors.py @@ -225,5 +225,13 @@ class BackupTimeoutError(BackupOSError): exit_mcode = 111 +class BackupBrokenSymlinkError(BackupError): + """{}: {}""" + + # A recursion root that is a symlink pointing to a non-existing target: as we follow + # symlinked recursion roots, there is nothing to archive there, see #4737. + exit_mcode = 112 + + class BackupItemExcluded(Exception): """Used internally to skip an item from processing when it is excluded.""" diff --git a/src/borg/helpers/fs.py b/src/borg/helpers/fs.py index 1412416450..653371a1b5 100644 --- a/src/borg/helpers/fs.py +++ b/src/borg/helpers/fs.py @@ -505,8 +505,10 @@ def O_(*flags): flags_special = flags_base | O_("NOFOLLOW") # BLOCK == wait when reading devices or fifos flags_special_follow = flags_base # BLOCK == wait when reading symlinked devices or fifos flags_normal = flags_base | O_("NONBLOCK", "NOFOLLOW") +flags_normal_follow = flags_base | O_("NONBLOCK") # follow a symlinked recursion root, see #4737 flags_noatime = flags_normal | O_("NOATIME") flags_dir = O_("DIRECTORY", "RDONLY", "NOFOLLOW") +flags_dir_follow = O_("DIRECTORY", "RDONLY") # follow a symlinked recursion root, see #4737 def os_open(*, flags, path=None, parent_fd=None, name=None, noatime=False): diff --git a/src/borg/testsuite/archiver/create_cmd_test.py b/src/borg/testsuite/archiver/create_cmd_test.py index c53c70a4e1..9a34c6e96d 100644 --- a/src/borg/testsuite/archiver/create_cmd_test.py +++ b/src/borg/testsuite/archiver/create_cmd_test.py @@ -16,7 +16,7 @@ from ...platform import is_win32, is_cygwin from ...platformflags import is_msystem from ...repository import Repository -from ...helpers import CommandError, BackupPermissionError, BackupTimeoutError +from ...helpers import CommandError, BackupPermissionError, BackupTimeoutError, BackupBrokenSymlinkError from .. import has_lchflags, has_mknod from .. import changedir from .. import ( @@ -1205,6 +1205,114 @@ def test_create_read_special_broken_symlink(archivers, request): assert "input/link -> somewhere does not exist" in output +@pytest.mark.skipif(not are_symlinks_supported(), reason="symlinks not supported") +def test_create_symlink_root_dir(archivers, request): + # a recursion root that is a symlink to a directory is followed, see #4737 + archiver = request.getfixturevalue(archivers) + create_regular_file(archiver.input_path, "target/file", contents=b"content") + os.symlink("target", os.path.join(archiver.input_path, "link")) + cmd(archiver, "repo-create", RK_ENCRYPTION) + cmd(archiver, "create", "test", "input/link") + output = cmd(archiver, "list", "test") + assert "input/link -> target" not in output # not archived as a symlink, but as a directory + assert "input/link/file" in output # we recursed into the symlink target + assert "input/target" not in output # the target path is not in the archive + with changedir("output"): + cmd(archiver, "extract", "test") + assert not os.path.islink("input/link") + assert os.path.isdir("input/link") + with open("input/link/file", "rb") as f: + assert f.read() == b"content" + + +@pytest.mark.skipif(not are_symlinks_supported(), reason="symlinks not supported") +def test_create_symlink_root_file(archivers, request): + # a recursion root that is a symlink to a regular file is followed, see #4737 + archiver = request.getfixturevalue(archivers) + create_regular_file(archiver.input_path, "target", contents=b"content") + os.symlink("target", os.path.join(archiver.input_path, "link")) + cmd(archiver, "repo-create", RK_ENCRYPTION) + cmd(archiver, "create", "test", "input/link") + output = cmd(archiver, "list", "test") + assert "input/link -> target" not in output # not archived as a symlink, but as a regular file + assert "input/link" in output + with changedir("output"): + cmd(archiver, "extract", "test") + assert not os.path.islink("input/link") + with open("input/link", "rb") as f: + assert f.read() == b"content" + + +@pytest.mark.skipif(not are_symlinks_supported(), reason="symlinks not supported") +def test_create_symlink_below_root_not_followed(archivers, request): + # only recursion roots are followed, symlinks found while recursing are not, see #4737 + archiver = request.getfixturevalue(archivers) + create_regular_file(archiver.input_path, "target/file", contents=b"content") + os.symlink("target", os.path.join(archiver.input_path, "link")) + cmd(archiver, "repo-create", RK_ENCRYPTION) + cmd(archiver, "create", "test", "input") + output = cmd(archiver, "list", "test") + assert "input/link -> target" in output + assert "input/link/file" not in output + assert "input/target/file" in output + + +@pytest.mark.skipif(not are_symlinks_supported(), reason="symlinks not supported") +def test_create_symlink_root_and_target(archivers, request): + # a followed symlink root and its target are the same fs objects, so they are archived + # only once, under the path given first (like any other recursion root given twice). + archiver = request.getfixturevalue(archivers) + create_regular_file(archiver.input_path, "target/file", contents=b"content") + os.symlink("target", os.path.join(archiver.input_path, "link")) + cmd(archiver, "repo-create", RK_ENCRYPTION) + cmd(archiver, "create", "test", "input/link", "input/target") + output = cmd(archiver, "list", "test") + assert "input/link/file" in output + assert "input/target/file" not in output + + +@pytest.mark.skipif(not are_symlinks_supported(), reason="symlinks not supported") +def test_create_symlink_root_broken(archivers, request): + # a recursion root that is a symlink with a non-existing target is skipped with a warning + archiver = request.getfixturevalue(archivers) + create_regular_file(archiver.input_path, "file", contents=b"content") + os.symlink("somewhere does not exist", os.path.join(archiver.input_path, "link")) + cmd(archiver, "repo-create", RK_ENCRYPTION) + expected_ec = BackupBrokenSymlinkError("stat", "broken symlink, skipping it").exit_code + if expected_ec == EXIT_ERROR: # workaround, TODO: fix it + expected_ec = EXIT_WARNING + out = cmd(archiver, "create", "test", "input/link", "input/file", exit_code=expected_ec) + assert "input/link: stat: broken symlink, skipping it" in out + output = cmd(archiver, "list", "test") + assert "input/link" not in output + assert "input/file" in output # the other recursion root was archived + + +@pytest.mark.skipif(not are_symlinks_supported(), reason="symlinks not supported") +def test_create_symlink_paths_from_stdin_not_followed(archivers, request): + # only recursion roots are followed, paths fed in via --paths-from-* are not, see #4737 + archiver = request.getfixturevalue(archivers) + create_regular_file(archiver.input_path, "target", contents=b"content") + os.symlink("target", os.path.join(archiver.input_path, "link")) + cmd(archiver, "repo-create", RK_ENCRYPTION) + cmd(archiver, "create", "--paths-from-stdin", "test", input=b"input/link") + output = cmd(archiver, "list", "test") + assert "input/link -> target" in output + + +@pytest.mark.skipif(not are_symlinks_supported(), reason="symlinks not supported") +def test_create_symlink_root_dotslash_hack(archivers, request): + # the slashdot hack also works for a recursion root that is a symlink, see #4737 + archiver = request.getfixturevalue(archivers) + create_regular_file(archiver.input_path, "target/file", contents=b"content") + os.symlink("target", os.path.join(archiver.input_path, "link")) + cmd(archiver, "repo-create", RK_ENCRYPTION) + cmd(archiver, "create", "test", "input/link/./") # hack! + output = cmd(archiver, "list", "test") + assert "input" not in output # the prefix left of the slashdot was stripped + assert "file" in output + + def test_create_dotslash_hack(archivers, request): archiver = request.getfixturevalue(archivers) os.makedirs(os.path.join(archiver.input_path, "first", "secondA", "thirdA")) From d7855ade11f8cb4484ee2cda8d0ba8616ed671f4 Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Sat, 15 Aug 2026 16:46:45 +0200 Subject: [PATCH 139/144] tests: expect the warning's exit code, not the error's The BackupErrors raised while backing up a fs object do not reach the top level, they are wrapped into a BackupWarning and only reported as a warning. Thus, the expected exit code must come from the BackupWarning (giving EXIT_WARNING for BORG_EXIT_CODES=classic and the error's specific warning rc for modern), not from the wrapped error (which would give EXIT_ERROR for classic). Removes the "workaround, TODO: fix it" that patched up EXIT_ERROR after the fact. --- src/borg/testsuite/archiver/create_cmd_test.py | 16 +++++++--------- 1 file changed, 7 insertions(+), 9 deletions(-) diff --git a/src/borg/testsuite/archiver/create_cmd_test.py b/src/borg/testsuite/archiver/create_cmd_test.py index 9a34c6e96d..c8d8077f22 100644 --- a/src/borg/testsuite/archiver/create_cmd_test.py +++ b/src/borg/testsuite/archiver/create_cmd_test.py @@ -17,6 +17,7 @@ from ...platformflags import is_msystem from ...repository import Repository from ...helpers import CommandError, BackupPermissionError, BackupTimeoutError, BackupBrokenSymlinkError +from ...helpers import BackupWarning from .. import has_lchflags, has_mknod from .. import changedir from .. import ( @@ -322,9 +323,8 @@ def test_create_no_permission_file(archivers, request): os.chmod(file_path + "2", 0o000) cmd(archiver, "repo-create", RK_ENCRYPTION) flist = "".join(f"input/file{n}\n" for n in range(1, 4)) - expected_ec = BackupPermissionError("open", OSError(13, "permission denied")).exit_code - if expected_ec == EXIT_ERROR: # workaround, TODO: fix it - expected_ec = EXIT_WARNING + exc = BackupPermissionError("open", OSError(13, "permission denied")) + expected_ec = BackupWarning("input/file2", exc).exit_code out = cmd( archiver, "create", @@ -1076,9 +1076,8 @@ def test_create_read_special_timeout_expired(archivers, request): cmd(archiver, "repo-create", RK_ENCRYPTION) create_regular_file(archiver.input_path, "file1", size=1024) os.mkfifo(os.path.join(archiver.input_path, "fifo")) - expected_ec = BackupTimeoutError("read", OSError(errno.ETIMEDOUT, "timeout")).exit_code - if expected_ec == EXIT_ERROR: # workaround, TODO: fix it - expected_ec = EXIT_WARNING + exc = BackupTimeoutError("read", OSError(errno.ETIMEDOUT, "timeout")) + expected_ec = BackupWarning("input/fifo", exc).exit_code started = time.monotonic() out = cmd( archiver, @@ -1278,9 +1277,8 @@ def test_create_symlink_root_broken(archivers, request): create_regular_file(archiver.input_path, "file", contents=b"content") os.symlink("somewhere does not exist", os.path.join(archiver.input_path, "link")) cmd(archiver, "repo-create", RK_ENCRYPTION) - expected_ec = BackupBrokenSymlinkError("stat", "broken symlink, skipping it").exit_code - if expected_ec == EXIT_ERROR: # workaround, TODO: fix it - expected_ec = EXIT_WARNING + exc = BackupBrokenSymlinkError("stat", "broken symlink, skipping it") + expected_ec = BackupWarning("input/link", exc).exit_code out = cmd(archiver, "create", "test", "input/link", "input/file", exit_code=expected_ec) assert "input/link: stat: broken symlink, skipping it" in out output = cmd(archiver, "list", "test") From 91b030e4862a10f37f52692c063b8bcea7e1dd1f Mon Sep 17 00:00:00 2001 From: Thomas Waldmann Date: Sun, 16 Aug 2026 11:28:19 +0200 Subject: [PATCH 140/144] tests: do not pin the zsh --sort-by label, shtab 1.11.0 changed it shtab 1.11.0 (released 2026-08-15) labels a zsh option argument with the argument's metavar instead of its dest, so the generated spec for "borg list --sort-by" (metavar KEYS, dest sort_by) changed from "--sort-by[...]:sort_by:_borg_complete_item_sortby" to "--sort-by[...]:KEYS:_borg_complete_item_sortby" test_zsh_sortby_wiring matched the ":sort_by:" form literally and started failing for list and repo-list (borg diff --sort-by has no metavar, so shtab still falls back to the dest there and that entry kept passing). The completions are fine: the label is only what zsh displays while completing, and --sort-by is still wired to our helper function. Match the helper and leave the label open, so either shtab behaviour passes. Note that shtab is not pinned in requirements.d/, it comes in via the shtab>=1.9.3 dependency in pyproject.toml, so CI picks up new shtab releases as they appear. Co-Authored-By: Claude Opus 5 --- src/borg/testsuite/archiver/completion_cmd_test.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/src/borg/testsuite/archiver/completion_cmd_test.py b/src/borg/testsuite/archiver/completion_cmd_test.py index 1ef5add254..e45f40001f 100644 --- a/src/borg/testsuite/archiver/completion_cmd_test.py +++ b/src/borg/testsuite/archiver/completion_cmd_test.py @@ -279,7 +279,10 @@ def test_zsh_sortby_wiring(archivers, request, command, fn): lines = completion_lines(archivers, request, "zsh") block = lines[lines.index(f"_shtab_borg_{command.replace('-', '_')}_options=(") :] block = block[: block.index(")")] - assert any(f':sort_by:{fn}"' in line for line in block) + # The ":