"""Documents archive (ZIP) — the one export that carries file *contents*. The JSON/YAML backup deliberately keeps document metadata only; the binary blobs ride the HA backup. That leaves a portable JSON export with dangling file docs on a fresh instance. This module adds a dedicated, self-contained documents archive: manifest.json {"version":1, "objects":[{object_id, object_name, documents:[]}]} blobs/ the raw file contents (content-addressed, dedup'd) Export gathers the selected objects' documents + their unique blobs. Import writes every blob back, then re-attaches metadata to the matching object (by id first, then by name for a cross-instance restore), skipping documents that already exist so a repeated import is idempotent. Weblinks travel too (0 bytes) so the archive is a complete documents backup on its own. """ from __future__ import annotations import io import json import logging import zipfile from typing import Any from homeassistant.core import HomeAssistant from ..const import DOMAIN, GLOBAL_UNIQUE_ID from .documents import KIND_WEBLINK, async_rewrite_doc_refs, doc_wire_dict _LOGGER = logging.getLogger(__name__) MANIFEST_NAME = "manifest.json" BLOB_DIR = "blobs/" ARCHIVE_VERSION = 1 # Cap a single archive import so a crafted ZIP can't exhaust memory/disk. A # real documents backup is dominated by the blobs, already capped at 25 MB # each × 100 docs/object — this is a coarse whole-archive ceiling on top. MAX_ARCHIVE_BYTES = 500 * 1024 * 1024 # 500 MB uncompressed-blob budget MAX_MANIFEST_BYTES = 16 * 1024 * 1024 # the manifest is metadata only MAX_ARCHIVE_MEMBERS = 20000 # ceiling on ZIP entry count (blobs cap at 100/obj) def _read_member_bounded(zf: zipfile.ZipFile, name: str, limit: int) -> bytes: """Read a ZIP member but never materialise more than ``limit`` bytes. ``ZipFile.read`` inflates the WHOLE member before returning, so a crafted member (small compressed, huge inflated — a "zip bomb") would exhaust memory before any post-hoc size check. Reading through ``open()`` with a hard byte ceiling bounds the decompression regardless of the declared/actual size. """ with zf.open(name) as fh: data = fh.read(limit + 1) if len(data) > limit: raise ValueError("archive_member_too_large") return data def _get_store(hass: HomeAssistant) -> Any: from .. import DOCUMENT_STORE_KEY return hass.data.get(DOMAIN, {}).get(DOCUMENT_STORE_KEY) def _object_name_map(hass: HomeAssistant) -> tuple[dict[str, str], dict[str, str]]: """(object_id → entry_id-object_id) is identity; return the maps the import needs: existing object_ids (set) and name → object_id for cross-instance matching.""" from ..const import CONF_OBJECT ids: dict[str, str] = {} by_name: dict[str, str] = {} for entry in hass.config_entries.async_entries(DOMAIN): if entry.unique_id == GLOBAL_UNIQUE_ID: continue obj = entry.data.get(CONF_OBJECT, {}) oid = obj.get("id") if not oid: continue ids[oid] = oid name = obj.get("name") if name: by_name.setdefault(name, oid) return ids, by_name def _object_task_ids(hass: HomeAssistant, object_id: str) -> set[str]: """The current task ids of the object whose id is ``object_id`` (empty if none). Used to keep a same-instance archive restore's task links valid.""" from ..const import CONF_OBJECT, CONF_TASKS for entry in hass.config_entries.async_entries(DOMAIN): if entry.unique_id == GLOBAL_UNIQUE_ID: continue if entry.data.get(CONF_OBJECT, {}).get("id") == object_id: return set(entry.data.get(CONF_TASKS, {})) return set() def _object_part_ids(hass: HomeAssistant, object_id: str) -> set[str]: """The current spare-part ids of the object (same role as ``_object_task_ids``, for a doc's part links).""" from ..const import CONF_OBJECT, CONF_PARTS for entry in hass.config_entries.async_entries(DOMAIN): if entry.unique_id == GLOBAL_UNIQUE_ID: continue if entry.data.get(CONF_OBJECT, {}).get("id") == object_id: return set(entry.data.get(CONF_PARTS) or {}) return set() class ArchiveTooLarge(ValueError): """The selection's files exceed what an import accepts (MAX_ARCHIVE_BYTES).""" def _gather_archive(hass: HomeAssistant, entry_ids: set[str] | None) -> tuple[Any, dict[str, Any], list[str]]: """The manifest and the blob digests of the selected objects (None = all). Metadata only — reads config entries and the document store. """ from ..const import CONF_OBJECT from ..export import object_entries store = _get_store(hass) entries = object_entries(hass, entry_ids) manifest_objects: list[dict[str, Any]] = [] blob_hashes: set[str] = set() for entry in entries: obj = entry.data.get(CONF_OBJECT, {}) object_id = obj.get("id", "") if not object_id or store is None: continue docs = [] for d in store.for_object(object_id): # The same record the JSON export writes (id included, so a # restore can re-point completion photos / part doc links). docs.append(doc_wire_dict(d, include_id=True)) h = d.get("hash") if d.get("kind") != KIND_WEBLINK and isinstance(h, str): blob_hashes.add(h) if docs: manifest_objects.append({"object_id": object_id, "object_name": obj.get("name", ""), "documents": docs}) return store, {"version": ARCHIVE_VERSION, "objects": manifest_objects}, sorted(blob_hashes) def _write_archive(store: Any, manifest: dict[str, Any], blob_hashes: list[str], target: Any) -> None: """Write the ZIP to ``target`` (a path or binary file object) — blocking. Each blob is streamed from disk into the archive (``ZipFile.write``), so memory stays flat however large the documents are. Refuses up front a selection whose files exceed MAX_ARCHIVE_BYTES: the import refuses such an archive anyway, and the export used to assemble the whole ZIP in memory with no ceiling (bug audit 2026-09-26). """ blobs: list[tuple[str, Any]] = [] total = 0 for h in blob_hashes: if store is None: break try: path = store.blob_path(h) except ValueError: continue if not path.is_file(): _LOGGER.warning("Documents archive: blob %s missing on disk, skipped", h[:12]) continue total += path.stat().st_size if total > MAX_ARCHIVE_BYTES: raise ArchiveTooLarge( f"The selected documents exceed {MAX_ARCHIVE_BYTES // (1024 * 1024)} MB, " "more than an archive import accepts — export fewer objects at a time." ) blobs.append((h, path)) with zipfile.ZipFile(target, "w", zipfile.ZIP_DEFLATED) as zf: zf.writestr(MANIFEST_NAME, json.dumps(manifest, ensure_ascii=False, indent=2)) for h, path in blobs: zf.write(path, arcname=f"{BLOB_DIR}{h}") def build_documents_archive(hass: HomeAssistant, entry_ids: set[str] | None = None) -> bytes: """Build a documents ZIP in memory for the selected objects (None = all). Blocking — call via the executor. The HTTP export streams through :func:`async_build_documents_archive_file` instead; this in-memory form stays for tooling and tests. """ store, manifest, blob_hashes = _gather_archive(hass, entry_ids) buf = io.BytesIO() _write_archive(store, manifest, blob_hashes, buf) return buf.getvalue() async def async_build_documents_archive_file(hass: HomeAssistant, entry_ids: set[str] | None = None) -> str: """Write the documents ZIP to a temporary file and return its path. The caller streams it out and deletes it. Raises :class:`ArchiveTooLarge` (nothing left behind) when the selection exceeds the import ceiling. """ import os import tempfile store, manifest, blob_hashes = _gather_archive(hass, entry_ids) def _write() -> str: fd, path = tempfile.mkstemp(prefix="maintenance-documents-", suffix=".zip") os.close(fd) try: _write_archive(store, manifest, blob_hashes, path) except BaseException: os.unlink(path) raise return path return await hass.async_add_executor_job(_write) def _doc_key(doc: dict[str, Any]) -> tuple[str | None, str | None]: """A document's identity for the idempotent re-import: kind + content hash (file) or URL (link). Only strings count — a list-valued URL in a crafted manifest was unhashable and crashed the set lookup.""" kind = doc.get("kind") ref = doc.get("hash") or doc.get("url") return (kind if isinstance(kind, str) else None, ref if isinstance(ref, str) else None) async def import_documents_archive(hass: HomeAssistant, data: bytes) -> dict[str, Any]: """Restore a documents ZIP: write blobs back, re-attach metadata. Objects are matched by id first (same instance), then by name (a cross-instance restore after a JSON import created fresh ids). Documents already present on the target object are skipped so a repeat import is idempotent. Returns counts. """ store = _get_store(hass) if store is None: return {"error": "documents store unavailable"} def _read() -> tuple[dict[str, Any], dict[str, bytes]]: blobs: dict[str, bytes] = {} manifest: dict[str, Any] = {} total = 0 with zipfile.ZipFile(io.BytesIO(data)) as zf: names = zf.namelist() if len(names) > MAX_ARCHIVE_MEMBERS: raise ValueError("archive_too_many_members") for name in names: if name == MANIFEST_NAME: # Bound the metadata member too (was read uncapped). manifest = json.loads(_read_member_bounded(zf, name, MAX_MANIFEST_BYTES).decode("utf-8")) elif name.startswith(BLOB_DIR) and not name.endswith("/"): digest = name[len(BLOB_DIR) :] if len(digest) == 64 and all(c in "0123456789abcdef" for c in digest): remaining = MAX_ARCHIVE_BYTES - total if remaining <= 0: raise ValueError("archive_too_large") # Read bounded by the remaining budget so a bomb can't # inflate past the whole-archive ceiling (checked DURING # the read, not after materialising the full member). content = _read_member_bounded(zf, name, remaining) total += len(content) blobs[digest] = content return manifest, blobs try: manifest, blobs = await hass.async_add_executor_job(_read) except (zipfile.BadZipFile, ValueError, json.JSONDecodeError, KeyError) as err: return {"error": f"invalid archive: {err}"} # The manifest is untrusted JSON: a list at the top, an "objects" that is # no list or a "documents" that is a number raised AttributeError / # TypeError past the handler above — a bare 500 instead of the clean # error (bug audit 2026-09-27). Wrong shapes are refused or skipped. if not isinstance(manifest, dict): return {"error": "invalid archive: the manifest is not an object"} raw_objects = manifest.get("objects") manifest_objects: list[tuple[dict[str, Any], list[Any]]] = [ (obj, docs) for obj in (raw_objects if isinstance(raw_objects, list) else []) if isinstance(obj, dict) and isinstance(docs := obj.get("documents", []), list) ] # 1) Write back only blobs a manifest document actually references — an # archive carrying extra blobs must not litter /config with orphans that # ride every HA backup and are never refcounted (disk-fill hardening). referenced: set[str] = set() for _obj, docs in manifest_objects: for m in docs: if isinstance(m, dict) and isinstance(m.get("hash"), str): referenced.add(m["hash"]) import hashlib written = 0 docs_created = 0 objects_matched = 0 # old → new document ids across every restored object, so history photos # and part doc links that pointed at the archived ids follow (the JSON # importer does the same for its own restore). doc_id_map: dict[str, str] = {} # Under the store's blob lock from the first blob write to the last # refcount bump: a document delete running meanwhile could remove a blob # file the restore had just written (or found) before the restored # document registered it — a document pointing at nothing (bug audit # 2026-09-27; the upload and delete paths took the lock, this one not). async with store.blob_lock: for digest, content in blobs.items(): if digest not in referenced: _LOGGER.info("Documents archive: blob %s referenced by no document, skipped", digest[:12]) continue if hashlib.sha256(content).hexdigest() != digest: _LOGGER.warning("Documents archive: blob %s failed hash check, skipped", digest[:12]) continue _, wrote_new = await hass.async_add_executor_job(store._store_blob_sync, content) if wrote_new: written += 1 store.notify_blob_added(digest) # 2) Re-attach metadata to the matching object (id, then name). ids, by_name = _object_name_map(hass) for obj, docs in manifest_objects: target = ids.get(str(obj.get("object_id") or "")) or by_name.get(str(obj.get("object_name") or "")) if target is None: _LOGGER.info("Documents archive: no object matches %r, its docs skipped", obj.get("object_name")) continue objects_matched += 1 # Skip docs already present on the target (idempotent re-import). existing = store.for_object(target) existing_keys = {_doc_key(d) for d in existing} fresh = [m for m in docs if isinstance(m, dict) and _doc_key(m) not in existing_keys] if fresh: # Keep task links that still resolve on the target (a same-instance # restore) via an identity map over the object's current task ids; # a cross-instance restore has fresh task ids, so those links drop # here and are re-established by the JSON import's remap instead. valid_task_ids = _object_task_ids(hass, target) identity = {tid: tid for tid in valid_task_ids} part_identity = {pid: pid for pid in _object_part_ids(hass, target)} docs_created += await store.async_import_documents( target, fresh, task_id_map=identity, part_id_map=part_identity, id_map=doc_id_map ) if doc_id_map: await async_rewrite_doc_refs(hass, lambda old: doc_id_map.get(old, old)) return { "blobs_written": written, "documents_created": docs_created, "objects_matched": objects_matched, }