diff --git a/docs/faq.rst b/docs/faq.rst index bfe647293a..fcd18de522 100644 --- a/docs/faq.rst +++ b/docs/faq.rst @@ -128,8 +128,11 @@ Are there other known limitations? - borg extract supports restoring only into an empty destination. After extraction, the destination will have exactly the contents of the extracted archive. - If you extract into a non-empty destination, borg will (for example) not - remove files which are in the destination, but not in the archive. + borg refuses to extract into a non-empty destination unless ``--continue`` is + given. If you extract into a non-empty destination, borg replaces existing files + by the archived files, but it will (for example) not remove files which are in + the destination, but not in the archive: the result is a mix of existing and + extracted files. See :issue:`4598` for a workaround and more details. Why are the extracted files owned by me and not by the original owner? @@ -876,8 +879,11 @@ How can I deal with my very unstable SSH connection? If you have issues with lost connections during long-running borg commands, you could try to work around: -- Make partial extracts like ``borg extract ARCHIVE PATTERN`` to do multiple - smaller extraction runs that complete before your connection has issues. +- Use ``borg extract --continue ARCHIVE`` to continue an interrupted extraction + (in the same directory): it skips the files that are fully extracted already. +- Make partial extracts like ``borg extract --continue ARCHIVE PATTERN`` to do + multiple smaller extraction runs that complete before your connection has issues + (``--continue`` is needed as soon as the extraction directory is not empty). - Try using ``borg mount MOUNTPOINT`` and ``rsync -avH`` from ``MOUNTPOINT`` to your desired extraction directory. If the connection breaks down, just repeat that over and over again until rsync does not find anything diff --git a/docs/internals/frontends.rst b/docs/internals/frontends.rst index 07e4c8ec28..5d10ee738c 100644 --- a/docs/internals/frontends.rst +++ b/docs/internals/frontends.rst @@ -869,6 +869,8 @@ Errors Archive {} already exists Archive.DoesNotExist rc: 31 traceback: no Archive {} does not exist + Archive.ExtractionDirNotEmpty rc: 33 traceback: no + Extraction directory {} is not empty. Use --continue to extract into a non-empty directory: existing files that differ from the archived files will be replaced and the result will be a mix of existing and extracted files. Archive.IncompatibleFilesystemEncodingError rc: 32 traceback: no Failed to encode filename "{}" into file system encoding "{}". Consider configuring the LANG environment variable. diff --git a/docs/quickstart.rst b/docs/quickstart.rst index 3a6c7c3854..e901ad9c37 100644 --- a/docs/quickstart.rst +++ b/docs/quickstart.rst @@ -546,9 +546,11 @@ Example with **borg extract**: :: - # borg extract always extracts into current directory and that directory - # should be empty (borg does not support transforming a non-empty dir to - # the state as present in your backup archive). + # borg extract always extracts into the current directory and that directory + # must be empty: extracting into a non-empty directory replaces existing files + # and results in a mix of existing and extracted files, so borg refuses to do + # that unless --continue is given (and borg does not support transforming a + # non-empty dir to the state as present in your backup archive). mkdir borg_restore cd borg_restore diff --git a/docs/usage/extract.rst b/docs/usage/extract.rst index af420e5215..1e58b6929b 100644 --- a/docs/usage/extract.rst +++ b/docs/usage/extract.rst @@ -4,9 +4,12 @@ Examples ~~~~~~~~ :: - # Extract entire archive + # Extract entire archive (into the current directory, which must be empty) $ borg extract my-files + # Continue an interrupted extraction (or extract into a non-empty directory) + $ borg extract --continue my-files + # Extract entire archive and list files while processing $ borg extract --list my-files diff --git a/src/borg/archive.py b/src/borg/archive.py index 195fcb86d0..ff75dcc117 100644 --- a/src/borg/archive.py +++ b/src/borg/archive.py @@ -591,6 +591,11 @@ class IncompatibleFilesystemEncodingError(Error): exit_mcode = 32 + class ExtractionDirNotEmpty(Error): + """Extraction directory {} is not empty. Use --continue to extract into a non-empty directory: existing files that differ from the archived files will be replaced and the result will be a mix of existing and extracted files.""" + + exit_mcode = 33 + def __init__( self, manifest, diff --git a/src/borg/archiver/extract_cmd.py b/src/borg/archiver/extract_cmd.py index bca551e843..19b0e51f05 100644 --- a/src/borg/archiver/extract_cmd.py +++ b/src/borg/archiver/extract_cmd.py @@ -1,11 +1,13 @@ +import os import sys import logging import stat from ._common import with_repository, with_archive from ._common import build_filter, build_matcher -from ..archive import BackupError, format_store_stats +from ..archive import Archive, BackupError, format_store_stats from ..constants import * # NOQA +from ..helpers import Error from ..helpers import archivename_validator, PathSpec from ..helpers import remove_surrogates from ..helpers import HardLinkManager @@ -43,6 +45,16 @@ def do_extract(self, args, repository, manifest, archive): sparse = args.sparse strip_components = args.strip_components continue_extraction = args.continue_extraction + if not (dry_run or stdout or continue_extraction): + # Extracting into a non-empty directory (like a home directory that is in use) replaces existing + # files and results in a mix of existing and extracted files, so it must be asked for explicitly, + # see #10057. + try: + is_empty = not os.listdir(archive.cwd) + except OSError as e: + raise Error(f"Cannot check whether the extraction directory {archive.cwd} is empty: {e}") from None + if not is_empty: + raise Archive.ExtractionDirNotEmpty(archive.cwd) dirs = [] hlm = HardLinkManager(id_type=bytes, info_type=str) # hlid -> path @@ -146,6 +158,25 @@ def build_parser_extract(self, subparsers, common_parser, mid_common_parser): ``--progress`` can be slower than no progress display, since it makes one additional pass over the archive metadata. + Extracting into a non-empty directory replaces existing files that are in the way and + leaves all other existing files as they are. The result is a mix of the files that were + there before and the extracted files, which might not be what you want - especially not + by accident in a directory that is in use (e.g. your home directory). Thus, borg refuses + to extract into a non-empty directory unless ``--continue`` is given (``--dry-run`` and + ``--stdout`` do not write into the directory, so they always work). The usual way to + restore is to extract into a new, empty directory: after that, the directory has exactly + the contents of the extracted archive (or of the selected part of it). + + ``--continue`` extracts into a non-empty directory. It is made for continuing a previously + interrupted extraction of the same archive into the same directory: an existing regular + file that has the same type, permissions (mode), size and modification time as the + archived file is considered to be fully extracted already and is skipped. Everything else + is extracted, replacing existing files. Files that are in the directory, but not in the + archive, are left as they are. ``--continue`` is thus also needed to restore files into an + existing directory tree. Note that a file that was damaged without a change of its size + and modification time (e.g. by bit rot) is skipped, not replaced: remove it before + extracting it. + If a file's content chunks are missing from the repository or are corrupted (they fail authentication, decryption or decompression), the extraction does not abort: each such chunk is written as all-zero data of the correct size, an error naming the chunk is logged, @@ -216,7 +247,8 @@ def build_parser_extract(self, subparsers, common_parser, mid_common_parser): "--continue", dest="continue_extraction", action="store_true", - help="continue a previously interrupted extraction of the same archive", + help="extract into a non-empty directory, e.g. to continue a previously interrupted extraction of " + "the same archive: skip files that are fully extracted already, replace other existing files", ) subparser.add_argument("name", metavar="NAME", type=archivename_validator, help="specify the archive name") subparser.add_argument( diff --git a/src/borg/testsuite/archiver/check_cmd_test.py b/src/borg/testsuite/archiver/check_cmd_test.py index 9d8ec2225e..8ca228c788 100644 --- a/src/borg/testsuite/archiver/check_cmd_test.py +++ b/src/borg/testsuite/archiver/check_cmd_test.py @@ -29,6 +29,7 @@ from ...manifest import Archives, Manifest from ...repoobj import RepoObj from ...repository import PackTracker, Repository +from .. import changedir from ..repoobj_test import DATA_SIZE_OFFSET from ..repository_test import fchunk, corrupt_chunk_on_disk from . import ( @@ -1154,7 +1155,8 @@ def test_verify_data_wrong_chunk_content(archivers, request, monkeypatch): # by default, reads do not check the id/content invariant, so this is not noticed: monkeypatch.delenv("BORG_ASSERT_ID", raising=False) - cmd(archiver, "extract", "archive1", exit_code=0) + with changedir("output"): + cmd(archiver, "extract", "archive1", exit_code=0) # ... but check --verify-data always checks it: output = cmd(archiver, "check", "--archives-only", "--verify-data", exit_code=1) assert f"{bin_to_hex(chunk.id)}, integrity error" in output @@ -1163,7 +1165,9 @@ def test_verify_data_wrong_chunk_content(archivers, request, monkeypatch): # with "read" in BORG_ASSERT_ID, reads check it too: extract treats the chunk as corrupted, i.e. # it extracts all-zero data instead and reports the file with a warning. monkeypatch.setenv("BORG_ASSERT_ID", "read") - output = cmd(archiver, "extract", "archive1", exit_code=BackupDamagedChunksError.exit_mcode) + Path("output2").mkdir() + with changedir("output2"): + output = cmd(archiver, "extract", "archive1", exit_code=BackupDamagedChunksError.exit_mcode) assert "id verification failed" in output assert "1 chunk(s) missing or corrupted in the repository, replaced by all-zero data" in output diff --git a/src/borg/testsuite/archiver/checks_test.py b/src/borg/testsuite/archiver/checks_test.py index 6d9431f983..bf76ed05b3 100644 --- a/src/borg/testsuite/archiver/checks_test.py +++ b/src/borg/testsuite/archiver/checks_test.py @@ -231,10 +231,10 @@ def test_remote_repo_strip_components_doesnt_leak(remote_archiver): res = cmd(remote_archiver, "extract", "test", "--debug", "--strip-components", "2") assert marker not in res with assert_creates_file("dir/file"): - res = cmd(remote_archiver, "extract", "test", "--debug", "--strip-components", "1") + res = cmd(remote_archiver, "extract", "test", "--debug", "--continue", "--strip-components", "1") assert marker not in res with assert_creates_file("input/dir/file"): - res = cmd(remote_archiver, "extract", "test", "--debug", "--strip-components", "0") + res = cmd(remote_archiver, "extract", "test", "--debug", "--continue", "--strip-components", "0") assert marker not in res diff --git a/src/borg/testsuite/archiver/create_cmd_test.py b/src/borg/testsuite/archiver/create_cmd_test.py index 5e99a8c753..7dda8321e1 100644 --- a/src/borg/testsuite/archiver/create_cmd_test.py +++ b/src/borg/testsuite/archiver/create_cmd_test.py @@ -662,11 +662,11 @@ def test_exclude_sanitation(archivers, request): with changedir("input"): cmd(archiver, "create", "test2", ".", "--exclude=./file1") with changedir("output"): - cmd(archiver, "extract", "test2") + cmd(archiver, "extract", "test2", "--continue") assert sorted(os.listdir("output")) == ["file2"] cmd(archiver, "create", "test3", "input", "--exclude=input/./file1") with changedir("output"): - cmd(archiver, "extract", "test3") + cmd(archiver, "extract", "test3", "--continue") assert sorted(os.listdir("output/input")) == ["file2"] diff --git a/src/borg/testsuite/archiver/extract_cmd_test.py b/src/borg/testsuite/archiver/extract_cmd_test.py index 2bfb9c4a7a..fc8659461a 100644 --- a/src/borg/testsuite/archiver/extract_cmd_test.py +++ b/src/borg/testsuite/archiver/extract_cmd_test.py @@ -469,7 +469,7 @@ def test_unusual_filenames(archivers, request): cmd(archiver, "create", "test", "input") for filename in filenames: with changedir("output"): - cmd(archiver, "extract", "test", os.path.join("input", filename)) + cmd(archiver, "extract", "test", os.path.join("input", filename), "--continue") assert os.path.exists(os.path.join("output", "input", filename)) @@ -484,9 +484,9 @@ def test_strip_components(archivers, request): with assert_creates_file("file"): cmd(archiver, "extract", "test", "--strip-components", "2") with assert_creates_file("dir/file"): - cmd(archiver, "extract", "test", "--strip-components", "1") + cmd(archiver, "extract", "test", "--continue", "--strip-components", "1") with assert_creates_file("input/dir/file"): - cmd(archiver, "extract", "test", "--strip-components", "0") + cmd(archiver, "extract", "test", "--continue", "--strip-components", "0") @requires_hardlinks @@ -516,7 +516,7 @@ def test_extract_hardlinks2(archivers, request): assert os.stat("source2").st_nlink == 2 with changedir("output"): - cmd(archiver, "extract", "test", "input/dir1") + cmd(archiver, "extract", "test", "input/dir1", "--continue") assert os.stat("input/dir1/hardlink").st_nlink == 2 assert os.stat("input/dir1/subdir/hardlink").st_nlink == 2 assert open("input/dir1/subdir/hardlink", "rb").read() == b"123456" @@ -563,11 +563,11 @@ def test_extract_include_exclude(archivers, request): assert sorted(os.listdir("output/input")) == ["file1"] with changedir("output"): - cmd(archiver, "extract", "test", "--exclude=input/file2") + cmd(archiver, "extract", "test", "--continue", "--exclude=input/file2") assert sorted(os.listdir("output/input")) == ["file1", "file3"] with changedir("output"): - cmd(archiver, "extract", "test", "--exclude-from=" + archiver.exclude_file_path) + cmd(archiver, "extract", "test", "--continue", "--exclude-from=" + archiver.exclude_file_path) assert sorted(os.listdir("output/input")) == ["file1", "file3"] @@ -887,7 +887,7 @@ def test_overwrite(archivers, request): os.mkdir("output/input/file1") os.mkdir("output/input/dir2") with changedir("output"): - cmd(archiver, "extract", "test") + cmd(archiver, "extract", "test", "--continue") assert_dirs_equal("input", "output/input") # But non-empty dirs should fail @@ -898,7 +898,7 @@ def test_overwrite(archivers, request): if expected_ec == EXIT_ERROR: # workaround, TODO: fix it expected_ec = EXIT_WARNING with changedir("output"): - cmd(archiver, "extract", "test", exit_code=expected_ec) + cmd(archiver, "extract", "test", "--continue", exit_code=expected_ec) # derived from test_extract_xattrs_errors() @@ -941,6 +941,43 @@ def patched_setxattr_EACCES(*args, **kwargs): cmd(archiver, "extract", "test", exit_code=EXIT_WARNING) +def test_extract_non_empty_dir(archivers, request): + # extracting replaces existing files, thus borg refuses to extract into a non-empty directory + # unless --continue is given, see #10057. + archiver = request.getfixturevalue(archivers) + create_regular_file(archiver.input_path, "file1", contents=b"archived") + create_regular_file(archiver.input_path, "file2", contents=b"archived") + cmd(archiver, "repo-create", RK_ENCRYPTION) + cmd(archiver, "create", "test", "input") + # different size than the archived file1, so --continue never considers it as already extracted: + create_regular_file(archiver.output_path, "input/file1", contents=b"current") + + def assert_refused(*args): + if archiver.FORK_DEFAULT: + output = cmd(archiver, "extract", "test", *args, exit_code=Archive.ExtractionDirNotEmpty("x").exit_code) + assert "--continue" in output + else: + with pytest.raises(Archive.ExtractionDirNotEmpty, match="--continue"): + cmd(archiver, "extract", "test", *args) + # nothing was replaced, nothing was extracted: + assert os.listdir("input") == ["file1"] + with open("input/file1", "rb") as f: + assert f.read() == b"current" + + with changedir("output"): + assert_refused() + assert_refused("input/file2") # same for extracting only a part of the archive + # these do not write into the directory: + cmd(archiver, "extract", "test", "--dry-run") + assert cmd(archiver, "extract", "test", "input/file2", "--stdout", binary_output=True) == b"archived" + assert_refused() + # with --continue, borg extracts into the non-empty directory and replaces the existing file: + cmd(archiver, "extract", "test", "--continue") + for name in "file1", "file2": + with open(f"input/{name}", "rb") as f: + assert f.read() == b"archived" + + @pytest.mark.skipif(not are_hardlinks_supported(), reason="hardlinks not supported") def test_extract_continue(archivers, request): archiver = request.getfixturevalue(archivers) @@ -1099,7 +1136,7 @@ def test_extract_existing_directory(archivers, request): os.makedirs("input/dir", exist_ok=True) st1 = os.stat("input/dir") # extract - cmd(archiver, "extract", "test") + cmd(archiver, "extract", "test", "--continue") st2 = os.stat("input/dir") assert st1.st_ino == st2.st_ino diff --git a/src/borg/testsuite/archiver/recreate_cmd_test.py b/src/borg/testsuite/archiver/recreate_cmd_test.py index 77829f7c93..af140f7691 100644 --- a/src/borg/testsuite/archiver/recreate_cmd_test.py +++ b/src/borg/testsuite/archiver/recreate_cmd_test.py @@ -132,7 +132,8 @@ def test_recreate_subtree_hardlinks(archivers, request): assert os.stat("input/dir1/subdir/hardlink").st_nlink == 2 assert os.stat("input/dir1/aaaa").st_nlink == 2 assert os.stat("input/dir1/source2").st_nlink == 2 - with changedir("output"): + os.mkdir("output2") + with changedir("output2"): cmd(archiver, "extract", "test2") assert os.stat("input/dir1/hardlink").st_nlink == 4 diff --git a/src/borg/testsuite/archiver/return_codes_test.py b/src/borg/testsuite/archiver/return_codes_test.py index 488d4e3fc6..2a04916f65 100644 --- a/src/borg/testsuite/archiver/return_codes_test.py +++ b/src/borg/testsuite/archiver/return_codes_test.py @@ -20,14 +20,15 @@ def test_return_codes(archivers, request): cmd(archiver, "create", "archive", "input") with changedir("output"): cmd(archiver, "extract", "archive") - cmd( - archiver, - "extract", - "archive", - "does/not/match", - fork=True, - exit_code=IncludePatternNeverMatchedWarning().exit_code, - ) + cmd( + archiver, + "extract", + "archive", + "does/not/match", + "--continue", + fork=True, + exit_code=IncludePatternNeverMatchedWarning().exit_code, + ) def test_exit_codes(archivers, request, monkeypatch):