Files
HomeAssistantVS/custom_components/maintenance_supporter/helpers/doc_archive.py
T

353 lines
15 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Documents archive (ZIP) — the one export that carries file *contents*.
The JSON/YAML backup deliberately keeps document metadata only; the binary
blobs ride the HA backup. That leaves a portable JSON export with dangling
file docs on a fresh instance. This module adds a dedicated, self-contained
documents archive:
manifest.json {"version":1, "objects":[{object_id, object_name,
documents:[<metadata>]}]}
blobs/<sha256> the raw file contents (content-addressed, dedup'd)
Export gathers the selected objects' documents + their unique blobs. Import
writes every blob back, then re-attaches metadata to the matching object
(by id first, then by name for a cross-instance restore), skipping documents
that already exist so a repeated import is idempotent. Weblinks travel too
(0 bytes) so the archive is a complete documents backup on its own.
"""
from __future__ import annotations
import io
import json
import logging
import zipfile
from typing import Any
from homeassistant.core import HomeAssistant
from ..const import DOMAIN, GLOBAL_UNIQUE_ID
from .documents import KIND_WEBLINK, async_rewrite_doc_refs, doc_wire_dict
_LOGGER = logging.getLogger(__name__)
MANIFEST_NAME = "manifest.json"
BLOB_DIR = "blobs/"
ARCHIVE_VERSION = 1
# Cap a single archive import so a crafted ZIP can't exhaust memory/disk. A
# real documents backup is dominated by the blobs, already capped at 25 MB
# each × 100 docs/object — this is a coarse whole-archive ceiling on top.
MAX_ARCHIVE_BYTES = 500 * 1024 * 1024 # 500 MB uncompressed-blob budget
MAX_MANIFEST_BYTES = 16 * 1024 * 1024 # the manifest is metadata only
MAX_ARCHIVE_MEMBERS = 20000 # ceiling on ZIP entry count (blobs cap at 100/obj)
def _read_member_bounded(zf: zipfile.ZipFile, name: str, limit: int) -> bytes:
"""Read a ZIP member but never materialise more than ``limit`` bytes.
``ZipFile.read`` inflates the WHOLE member before returning, so a crafted
member (small compressed, huge inflated — a "zip bomb") would exhaust memory
before any post-hoc size check. Reading through ``open()`` with a hard byte
ceiling bounds the decompression regardless of the declared/actual size.
"""
with zf.open(name) as fh:
data = fh.read(limit + 1)
if len(data) > limit:
raise ValueError("archive_member_too_large")
return data
def _get_store(hass: HomeAssistant) -> Any:
from .. import DOCUMENT_STORE_KEY
return hass.data.get(DOMAIN, {}).get(DOCUMENT_STORE_KEY)
def _object_name_map(hass: HomeAssistant) -> tuple[dict[str, str], dict[str, str]]:
"""(object_id → entry_id-object_id) is identity; return the maps the import
needs: existing object_ids (set) and name → object_id for cross-instance
matching."""
from ..const import CONF_OBJECT
ids: dict[str, str] = {}
by_name: dict[str, str] = {}
for entry in hass.config_entries.async_entries(DOMAIN):
if entry.unique_id == GLOBAL_UNIQUE_ID:
continue
obj = entry.data.get(CONF_OBJECT, {})
oid = obj.get("id")
if not oid:
continue
ids[oid] = oid
name = obj.get("name")
if name:
by_name.setdefault(name, oid)
return ids, by_name
def _object_task_ids(hass: HomeAssistant, object_id: str) -> set[str]:
"""The current task ids of the object whose id is ``object_id`` (empty if
none). Used to keep a same-instance archive restore's task links valid."""
from ..const import CONF_OBJECT, CONF_TASKS
for entry in hass.config_entries.async_entries(DOMAIN):
if entry.unique_id == GLOBAL_UNIQUE_ID:
continue
if entry.data.get(CONF_OBJECT, {}).get("id") == object_id:
return set(entry.data.get(CONF_TASKS, {}))
return set()
def _object_part_ids(hass: HomeAssistant, object_id: str) -> set[str]:
"""The current spare-part ids of the object (same role as
``_object_task_ids``, for a doc's part links)."""
from ..const import CONF_OBJECT, CONF_PARTS
for entry in hass.config_entries.async_entries(DOMAIN):
if entry.unique_id == GLOBAL_UNIQUE_ID:
continue
if entry.data.get(CONF_OBJECT, {}).get("id") == object_id:
return set(entry.data.get(CONF_PARTS) or {})
return set()
class ArchiveTooLarge(ValueError):
"""The selection's files exceed what an import accepts (MAX_ARCHIVE_BYTES)."""
def _gather_archive(hass: HomeAssistant, entry_ids: set[str] | None) -> tuple[Any, dict[str, Any], list[str]]:
"""The manifest and the blob digests of the selected objects (None = all).
Metadata only — reads config entries and the document store.
"""
from ..const import CONF_OBJECT
from ..export import object_entries
store = _get_store(hass)
entries = object_entries(hass, entry_ids)
manifest_objects: list[dict[str, Any]] = []
blob_hashes: set[str] = set()
for entry in entries:
obj = entry.data.get(CONF_OBJECT, {})
object_id = obj.get("id", "")
if not object_id or store is None:
continue
docs = []
for d in store.for_object(object_id):
# The same record the JSON export writes (id included, so a
# restore can re-point completion photos / part doc links).
docs.append(doc_wire_dict(d, include_id=True))
h = d.get("hash")
if d.get("kind") != KIND_WEBLINK and isinstance(h, str):
blob_hashes.add(h)
if docs:
manifest_objects.append({"object_id": object_id, "object_name": obj.get("name", ""), "documents": docs})
return store, {"version": ARCHIVE_VERSION, "objects": manifest_objects}, sorted(blob_hashes)
def _write_archive(store: Any, manifest: dict[str, Any], blob_hashes: list[str], target: Any) -> None:
"""Write the ZIP to ``target`` (a path or binary file object) — blocking.
Each blob is streamed from disk into the archive (``ZipFile.write``), so
memory stays flat however large the documents are. Refuses up front a
selection whose files exceed MAX_ARCHIVE_BYTES: the import refuses such
an archive anyway, and the export used to assemble the whole ZIP in
memory with no ceiling (bug audit 2026-09-26).
"""
blobs: list[tuple[str, Any]] = []
total = 0
for h in blob_hashes:
if store is None:
break
try:
path = store.blob_path(h)
except ValueError:
continue
if not path.is_file():
_LOGGER.warning("Documents archive: blob %s missing on disk, skipped", h[:12])
continue
total += path.stat().st_size
if total > MAX_ARCHIVE_BYTES:
raise ArchiveTooLarge(
f"The selected documents exceed {MAX_ARCHIVE_BYTES // (1024 * 1024)} MB, "
"more than an archive import accepts — export fewer objects at a time."
)
blobs.append((h, path))
with zipfile.ZipFile(target, "w", zipfile.ZIP_DEFLATED) as zf:
zf.writestr(MANIFEST_NAME, json.dumps(manifest, ensure_ascii=False, indent=2))
for h, path in blobs:
zf.write(path, arcname=f"{BLOB_DIR}{h}")
def build_documents_archive(hass: HomeAssistant, entry_ids: set[str] | None = None) -> bytes:
"""Build a documents ZIP in memory for the selected objects (None = all).
Blocking — call via the executor. The HTTP export streams through
:func:`async_build_documents_archive_file` instead; this in-memory form
stays for tooling and tests.
"""
store, manifest, blob_hashes = _gather_archive(hass, entry_ids)
buf = io.BytesIO()
_write_archive(store, manifest, blob_hashes, buf)
return buf.getvalue()
async def async_build_documents_archive_file(hass: HomeAssistant, entry_ids: set[str] | None = None) -> str:
"""Write the documents ZIP to a temporary file and return its path.
The caller streams it out and deletes it. Raises :class:`ArchiveTooLarge`
(nothing left behind) when the selection exceeds the import ceiling.
"""
import os
import tempfile
store, manifest, blob_hashes = _gather_archive(hass, entry_ids)
def _write() -> str:
fd, path = tempfile.mkstemp(prefix="maintenance-documents-", suffix=".zip")
os.close(fd)
try:
_write_archive(store, manifest, blob_hashes, path)
except BaseException:
os.unlink(path)
raise
return path
return await hass.async_add_executor_job(_write)
def _doc_key(doc: dict[str, Any]) -> tuple[str | None, str | None]:
"""A document's identity for the idempotent re-import: kind + content
hash (file) or URL (link). Only strings count — a list-valued URL in a
crafted manifest was unhashable and crashed the set lookup."""
kind = doc.get("kind")
ref = doc.get("hash") or doc.get("url")
return (kind if isinstance(kind, str) else None, ref if isinstance(ref, str) else None)
async def import_documents_archive(hass: HomeAssistant, data: bytes) -> dict[str, Any]:
"""Restore a documents ZIP: write blobs back, re-attach metadata.
Objects are matched by id first (same instance), then by name (a
cross-instance restore after a JSON import created fresh ids). Documents
already present on the target object are skipped so a repeat import is
idempotent. Returns counts.
"""
store = _get_store(hass)
if store is None:
return {"error": "documents store unavailable"}
def _read() -> tuple[dict[str, Any], dict[str, bytes]]:
blobs: dict[str, bytes] = {}
manifest: dict[str, Any] = {}
total = 0
with zipfile.ZipFile(io.BytesIO(data)) as zf:
names = zf.namelist()
if len(names) > MAX_ARCHIVE_MEMBERS:
raise ValueError("archive_too_many_members")
for name in names:
if name == MANIFEST_NAME:
# Bound the metadata member too (was read uncapped).
manifest = json.loads(_read_member_bounded(zf, name, MAX_MANIFEST_BYTES).decode("utf-8"))
elif name.startswith(BLOB_DIR) and not name.endswith("/"):
digest = name[len(BLOB_DIR) :]
if len(digest) == 64 and all(c in "0123456789abcdef" for c in digest):
remaining = MAX_ARCHIVE_BYTES - total
if remaining <= 0:
raise ValueError("archive_too_large")
# Read bounded by the remaining budget so a bomb can't
# inflate past the whole-archive ceiling (checked DURING
# the read, not after materialising the full member).
content = _read_member_bounded(zf, name, remaining)
total += len(content)
blobs[digest] = content
return manifest, blobs
try:
manifest, blobs = await hass.async_add_executor_job(_read)
except (zipfile.BadZipFile, ValueError, json.JSONDecodeError, KeyError) as err:
return {"error": f"invalid archive: {err}"}
# The manifest is untrusted JSON: a list at the top, an "objects" that is
# no list or a "documents" that is a number raised AttributeError /
# TypeError past the handler above — a bare 500 instead of the clean
# error (bug audit 2026-09-27). Wrong shapes are refused or skipped.
if not isinstance(manifest, dict):
return {"error": "invalid archive: the manifest is not an object"}
raw_objects = manifest.get("objects")
manifest_objects: list[tuple[dict[str, Any], list[Any]]] = [
(obj, docs)
for obj in (raw_objects if isinstance(raw_objects, list) else [])
if isinstance(obj, dict) and isinstance(docs := obj.get("documents", []), list)
]
# 1) Write back only blobs a manifest document actually references — an
# archive carrying extra blobs must not litter /config with orphans that
# ride every HA backup and are never refcounted (disk-fill hardening).
referenced: set[str] = set()
for _obj, docs in manifest_objects:
for m in docs:
if isinstance(m, dict) and isinstance(m.get("hash"), str):
referenced.add(m["hash"])
import hashlib
written = 0
docs_created = 0
objects_matched = 0
# old → new document ids across every restored object, so history photos
# and part doc links that pointed at the archived ids follow (the JSON
# importer does the same for its own restore).
doc_id_map: dict[str, str] = {}
# Under the store's blob lock from the first blob write to the last
# refcount bump: a document delete running meanwhile could remove a blob
# file the restore had just written (or found) before the restored
# document registered it — a document pointing at nothing (bug audit
# 2026-09-27; the upload and delete paths took the lock, this one not).
async with store.blob_lock:
for digest, content in blobs.items():
if digest not in referenced:
_LOGGER.info("Documents archive: blob %s referenced by no document, skipped", digest[:12])
continue
if hashlib.sha256(content).hexdigest() != digest:
_LOGGER.warning("Documents archive: blob %s failed hash check, skipped", digest[:12])
continue
_, wrote_new = await hass.async_add_executor_job(store._store_blob_sync, content)
if wrote_new:
written += 1
store.notify_blob_added(digest)
# 2) Re-attach metadata to the matching object (id, then name).
ids, by_name = _object_name_map(hass)
for obj, docs in manifest_objects:
target = ids.get(str(obj.get("object_id") or "")) or by_name.get(str(obj.get("object_name") or ""))
if target is None:
_LOGGER.info("Documents archive: no object matches %r, its docs skipped", obj.get("object_name"))
continue
objects_matched += 1
# Skip docs already present on the target (idempotent re-import).
existing = store.for_object(target)
existing_keys = {_doc_key(d) for d in existing}
fresh = [m for m in docs if isinstance(m, dict) and _doc_key(m) not in existing_keys]
if fresh:
# Keep task links that still resolve on the target (a same-instance
# restore) via an identity map over the object's current task ids;
# a cross-instance restore has fresh task ids, so those links drop
# here and are re-established by the JSON import's remap instead.
valid_task_ids = _object_task_ids(hass, target)
identity = {tid: tid for tid in valid_task_ids}
part_identity = {pid: pid for pid in _object_part_ids(hass, target)}
docs_created += await store.async_import_documents(
target, fresh, task_id_map=identity, part_id_map=part_identity, id_map=doc_id_map
)
if doc_id_map:
await async_rewrite_doc_refs(hass, lambda old: doc_id_map.get(old, old))
return {
"blobs_written": written,
"documents_created": docs_created,
"objects_matched": objects_matched,
}