Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
90 changes: 59 additions & 31 deletions src/borg/archive.py
Original file line number Diff line number Diff line change
Expand Up @@ -2249,6 +2249,9 @@ def __init__(self):
self.chunks_modified = False
# ids of the packs repair wrote: stored by put() and flush(), or written by delete() rewriting a pack.
self.written_packs = set()
# ids of the objects with ro_type ROBJ_ARCHIVE_META that verify_data found.
# None if verify_data did not run or was interrupted.
self.archive_meta_ids = None

def record_stored(self, results):
"""Add the pack ids in results to written_packs.
Expand Down Expand Up @@ -2397,6 +2400,7 @@ def verify_data(self):
errors = 0
verified = 0 # chunks actually verified
defect_chunks = []
archive_meta_ids = set()
pi = ProgressIndicatorPercent(
total=chunks_count, msg="Verifying data %6.2f%%", step=0.01, msgid="check.verify_data"
)
Expand All @@ -2413,13 +2417,15 @@ def verify_data(self):
# we must decompress, so it'll call assert_id() in there.
# this is the audit that re-certifies the id/content invariant, so it reads at its
# own place, which always verifies and can not be switched off, see BORG_ASSERT_ID.
self.repo_objs.parse(
meta, _ = self.repo_objs.parse(
chunk_id,
encrypted_data,
decompress=True,
ro_type=ROBJ_DONTCARE,
assert_id_place="verify_data",
)
if meta["type"] == ROBJ_ARCHIVE_META:
archive_meta_ids.add(chunk_id)
except IntegrityErrorBase as integrity_error:
self.error_found = True
errors += 1
Expand Down Expand Up @@ -2456,7 +2462,7 @@ def verify_data(self):
try:
encrypted_data = self.repository.get(defect_chunk)
# we must decompress, so it'll call assert_id() in there (see above):
self.repo_objs.parse(
meta, _ = self.repo_objs.parse(
defect_chunk,
encrypted_data,
decompress=True,
Expand All @@ -2477,10 +2483,15 @@ def verify_data(self):
self.written_packs.add(new_pack_id)
else:
logger.warning("chunk %s not deleted, did not consistently fail.", bin_to_hex(defect_chunk))
if meta["type"] == ROBJ_ARCHIVE_META:
archive_meta_ids.add(defect_chunk)
else:
logger.warning("Found defect chunks. Run with --repair to remove them.")
for defect_chunk in defect_chunks:
logger.debug("chunk %s is defect.", bin_to_hex(defect_chunk))
if not sig_int:
# the ids of an interrupted pass are incomplete.
self.archive_meta_ids = archive_meta_ids
log = logger.error if errors else logger.info
if sig_int:
log(
Expand All @@ -2500,53 +2511,38 @@ def verify_data(self):
def rebuild_archives_directory(self):
"""Rebuild the archives directory, undeleting archives.

Iterates through all objects in the repository looking for archive metadata blocks.
When finding some that do not have a corresponding archives directory entry (either
a normal entry for an "existing" archive, or a soft-deleted entry for a "deleted"
archive), it will create that entry (making the archives directory consistent with
the repository).
Reads the archive metadata objects (ro_type ROBJ_ARCHIVE_META) in the repository. When
finding some that do not have a corresponding archives directory entry (either a normal
entry for an "existing" archive, or a soft-deleted entry for a "deleted" archive), it will
create that entry (making the archives directory consistent with the repository).

If self.archive_meta_ids is not None, it reads only these objects. Otherwise, it reads the
meta dict (ro_type and other object metadata, without the data) of every object to find them.
"""

def valid_archive(obj):
if not isinstance(obj, dict):
return False
return REQUIRED_ARCHIVE_KEYS.issubset(obj)

logger.info("Rebuilding missing archives directory entries, this might take some time...")
pi = ProgressIndicatorPercent(
total=len(self.chunks),
msg="Rebuilding missing archives directory entries %6.2f%%",
step=0.01,
msgid="check.rebuild_archives_directory",
)
for chunk_id, _ in self.chunks.iteritems():
if sig_int:
break
pi.show()
try:
cdata = self.repository.get(chunk_id, read_data=False) # only get metadata
meta = self.repo_objs.parse_meta(chunk_id, cdata, ro_type=ROBJ_DONTCARE)
except IntegrityErrorBase as exc:
logger.error("Skipping corrupted chunk: %s", exc)
self.error_found = True
continue
if meta["type"] != ROBJ_ARCHIVE_META:
continue
# now we know it is an archive metadata chunk, load the full object from the repo:
def check_archive_meta(chunk_id):
"""Load the archive metadata object chunk_id. If it has no archives directory entry, create one
(with --repair) or log that it would create one.
"""
cdata = self.repository.get(chunk_id)
try:
meta, data = self.repo_objs.parse(chunk_id, cdata, ro_type=ROBJ_DONTCARE)
except IntegrityErrorBase as exc:
logger.error("Skipping corrupted chunk: %s", exc)
self.error_found = True
continue
return
if meta["type"] != ROBJ_ARCHIVE_META:
continue # should never happen
return # should never happen
try:
archive = msgpack.unpackb(data)
# Ignore exceptions that might be raised when feeding msgpack with invalid data
except msgpack.UnpackException:
continue
return
if valid_archive(archive):
archive = self.key.unpack_archive(data)
archive = ArchiveItem(internal_dict=archive)
Expand All @@ -2566,6 +2562,38 @@ def valid_archive(obj):
else:
logger.warning(f"Would create archives directory entry for {name} {archive_id_hex}.")

if self.archive_meta_ids is not None:
logger.info("Rebuilding missing archives directory entries...")
logger.debug("Using the %d archive metadata objects found by verify_data.", len(self.archive_meta_ids))
# sorted, so the entries are logged in the same order on every run.
chunk_ids = sorted(self.archive_meta_ids)
total = len(chunk_ids)
else:
logger.info("Rebuilding missing archives directory entries, this might take some time...")
chunk_ids = (chunk_id for chunk_id, _ in self.chunks.iteritems())
total = len(self.chunks)
pi = ProgressIndicatorPercent(
total=total,
msg="Rebuilding missing archives directory entries %6.2f%%",
step=0.01,
msgid="check.rebuild_archives_directory",
)
for chunk_id in chunk_ids:
if sig_int:
break
pi.show()
if self.archive_meta_ids is None:
try:
cdata = self.repository.get(chunk_id, read_data=False) # only get metadata
meta = self.repo_objs.parse_meta(chunk_id, cdata, ro_type=ROBJ_DONTCARE)
except IntegrityErrorBase as exc:
logger.error("Skipping corrupted chunk: %s", exc)
self.error_found = True
continue
if meta["type"] != ROBJ_ARCHIVE_META:
continue
check_archive_meta(chunk_id)

pi.finish()
if sig_int:
logger.info("Rebuilding missing archives directory entries interrupted.")
Expand Down
15 changes: 8 additions & 7 deletions src/borg/archiver/check_cmd.py
Original file line number Diff line number Diff line change
Expand Up @@ -238,13 +238,14 @@ def build_parser_check(self, subparsers, common_parser, mid_common_parser):
which normal reads do not do by default (see ``BORG_ASSERT_ID``). Running it periodically
is therefore recommended.

The ``--find-lost-archives`` option will also scan the whole repository, but
tells Borg to search for lost archive metadata. If Borg encounters any archive
metadata that does not match an archive directory entry (including
soft-deleted archives), it means that an entry was lost.
Unless ``borg compact`` is called, these archives can be fully restored with
``--repair``. Please note that ``--find-lost-archives`` must look at every
object in the repository and is thus very time-consuming. You cannot use
The ``--find-lost-archives`` option tells Borg to search for lost archive
metadata. If Borg encounters any archive metadata that does not match an
archive directory entry (including soft-deleted archives), it means that an
entry was lost. Unless ``borg compact`` is called, these archives can be fully
restored with ``--repair``. Without ``--verify-data``, ``--find-lost-archives``
reads the metadata of every object in the repository and is thus very
time-consuming. With ``--verify-data``, it reads only the archive metadata
objects that the data verification found. You cannot use
``--find-lost-archives`` with ``--repository-only``.

You can influence how the archive part of the ``Analyzing archive ...`` output is
Expand Down
Loading
Loading