"""Google Photos shared album client that bypasses the ~300 item HTML cap. Strategy -------- 1. Fetch the share URL HTML (with browser User-Agent so Google serves the real page, not a JS-only shim). 2. Extract the album media key and auth key from the embedded ``AF_dataServiceRequests`` blob - the same parameters Google's JS uses when it scrolls and needs more pages. 3. Page through the album by POSTing to the public ``batchexecute`` endpoint with the ``snAcKc`` RPC (the same one googlephotos.com calls). Each page carries up to 300 items plus a continuation token; we loop until the token is empty. This mirrors the approach used by community projects like ``xob0t/google-photos-toolkit``. It uses only undocumented public endpoints and no auth. Brittleness ----------- Google occasionally reshuffles the per-item array layout. We keep parsing positional but verify each field at access time and skip malformed entries rather than failing the whole batch. """ from __future__ import annotations import json import logging import re from typing import Any from urllib.parse import quote from .coordinator import MediaItem _LOGGER = logging.getLogger(__name__) _BROWSER_UA = ( "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) " "Chrome/124.0.0.0 Safari/537.36" ) # AF_dataServiceRequests block in the HTML carries the snAcKc request payload: # "snAcKc",ext: ... ,request:["",null,null,""] _REQUEST_RE = re.compile( r"snAcKc[^}]*?request:\s*\[\s*\"([A-Za-z0-9_-]+)\"\s*,\s*null\s*,\s*null\s*,\s*\"([A-Za-z0-9_-]+)\"", re.DOTALL, ) # XSSI prefix Google prepends to batchexecute responses. We strip it before parsing. _XSSI = ")]}'" # Hard ceiling - matches the upstream Google album item limit. _MAX_ITEMS = 20_000 # Per-page item count Google returns. Used only for sanity logging. _PAGE_SIZE = 300 # Strip any existing ``=w...-h...`` suffix from a Google CDN URL so we can # attach our own size hint based on the photo's known dimensions. _SIZE_SUFFIX_RE = re.compile(r"=[wh]\d+(?:-[a-z0-9]+)*$", re.IGNORECASE) _VIDEO_DURATION_KEY = 76647426 # presence indicates a video; we skip those _LIVEPHOTO_KEY = 146008172 class _AlbumKeys: __slots__ = ("album_key", "auth_key") def __init__(self, album_key: str, auth_key: str) -> None: self.album_key = album_key self.auth_key = auth_key async def fetch_album(session, share_url: str, *, timeout: float = 30.0) -> tuple[str | None, list[MediaItem]]: """Fetch a shared album in full. Returns (title, items). The HTML page is fetched once - just to recover the album/auth keys and title. All actual photo enumeration goes through Google's ``batchexecute`` endpoint, which is the only way to reach photos beyond the first ~300. """ keys, title = await _fetch_album_keys(session, share_url, timeout=timeout) if keys is None: return None, [] items: list[MediaItem] = [] seen_urls: set[str] = set() page_id: str | None = None page_no = 0 while True: page_no += 1 try: page_items, page_id = await _fetch_album_page( session, keys, page_id, timeout=timeout ) except Exception as err: _LOGGER.warning( "Album scrape: page %d batchexecute failed (%s); returning %d items so far", page_no, err, len(items), ) break added = 0 for it in page_items: if it.url in seen_urls: continue seen_urls.add(it.url) items.append(it) added += 1 if len(items) >= _MAX_ITEMS: break _LOGGER.debug( "Album scrape: page %d returned %d items (%d new), running total %d", page_no, len(page_items), added, len(items), ) if not page_id or len(items) >= _MAX_ITEMS or added == 0: break _LOGGER.info( "Album scraper: batchexecute fetched %d photos in %d page(s)", len(items), page_no, ) return title, items # -- Internals --------------------------------------------------------------- async def _fetch_album_keys( session, share_url: str, *, timeout: float ) -> tuple[_AlbumKeys | None, str | None]: """Fetch the share URL HTML and extract the album/auth keys + title.""" headers = { "User-Agent": _BROWSER_UA, "Accept": "text/html,application/xhtml+xml", "Accept-Language": "en-US,en;q=0.9", } try: async with session.get( share_url, headers=headers, timeout=timeout, allow_redirects=True ) as resp: resp.raise_for_status() ct = resp.headers.get("Content-Type", "").lower() if "html" not in ct and "text" not in ct: _LOGGER.debug("Album scrape: unexpected content-type %r", ct) return None, None html = await resp.text() except Exception as err: _LOGGER.debug("Album scrape: failed to fetch %s: %s", share_url, err) return None, None keys = _extract_keys(html) if keys is None: _LOGGER.debug("Album scrape: could not locate album keys in HTML") return None, None return keys, _extract_title(html) def _extract_keys(html: str) -> _AlbumKeys | None: m = _REQUEST_RE.search(html) if not m: return None return _AlbumKeys(album_key=m.group(1), auth_key=m.group(2)) def _extract_title(html: str) -> str | None: m = re.search(r"]*>(.*?)", html, re.IGNORECASE | re.DOTALL) if not m: return None title = m.group(1).strip() suffix = " - Google Photos" if title.endswith(suffix): title = title[: -len(suffix)].strip() return title or None def _extract_first_page_items(html: str) -> list[MediaItem]: """Pull the first 300 items out of the AF_initDataCallback blocks. Returns an empty list if the embedded data can't be parsed - the caller will still get the rest via batchexecute pagination. """ candidates: list[list[Any]] = [] for blob in _iter_af_data_blobs(html): try: tree = json.loads(blob) except json.JSONDecodeError: continue candidates.extend(_collect_album_item_lists(tree)) if not candidates: return [] best = max(candidates, key=len) items: list[MediaItem] = [] seen: set[str] = set() for raw in best: item = _parse_album_item(raw) if item is None or item.url in seen: continue seen.add(item.url) items.append(item) return items def _next_page_token_for_first_page( first_items: list[MediaItem], first_page_size: int ) -> str | None: """The first page's nextPageId isn't easy to find in the HTML AF blob. Strategy: if the first page is exactly the standard page size we use a sentinel empty token so the caller fetches page 2 with ``pageId=None``. The batchexecute endpoint, given ``pageId=None``, returns page 1 again along with the real continuation token, which we then use for the rest. Slightly wasteful (we re-fetch page 1) but robust to layout changes. """ if first_page_size >= _PAGE_SIZE: return "" # sentinel: drives the first batchexecute call return None async def _fetch_album_page( session, keys: _AlbumKeys, page_id: str | None, *, timeout: float, ) -> tuple[list[MediaItem], str | None]: """Call snAcKc once. ``page_id=""`` is treated as ``None`` (initial fetch).""" pid = page_id or None inner = json.dumps([keys.album_key, pid, None, keys.auth_key]) envelope = json.dumps([[["snAcKc", inner, None, "generic"]]]) form = f"f.req={quote(envelope)}" url = ( "https://photos.google.com/u/0/_/PhotosUi/data/batchexecute" f"?rpcids=snAcKc&source-path=/share/{quote(keys.album_key)}" ) headers = { "User-Agent": _BROWSER_UA, "Content-Type": "application/x-www-form-urlencoded;charset=UTF-8", "Accept": "*/*", "Origin": "https://photos.google.com", "Referer": f"https://photos.google.com/share/{keys.album_key}?key={keys.auth_key}", } async with session.post(url, data=form, headers=headers, timeout=timeout) as resp: resp.raise_for_status() body = await resp.text() return _parse_batchexecute_album_page(body) def _parse_batchexecute_album_page(body: str) -> tuple[list[MediaItem], str | None]: """Parse a batchexecute response for one snAcKc call. Format (per line, after the XSSI prefix): [["wrb.fr", "snAcKc", "", null, null, "generic"], ...] The inner data shape (per gptk-toolkit): data[1] = list of album items data[2] = nextPageId (str) or null """ text = body.lstrip() if text.startswith(_XSSI): text = text[len(_XSSI):] text = text.lstrip() line = text.split("\n", 1)[0].strip() if not line: return [], None try: outer = json.loads(line) except json.JSONDecodeError: return [], None inner_json: str | None = None for entry in outer: if isinstance(entry, list) and len(entry) >= 3 and entry[0] == "wrb.fr": inner_json = entry[2] break if not isinstance(inner_json, str): return [], None try: inner = json.loads(inner_json) except json.JSONDecodeError: return [], None raw_items = inner[1] if len(inner) > 1 and isinstance(inner[1], list) else [] next_page = inner[2] if len(inner) > 2 and isinstance(inner[2], str) else None if next_page == "": next_page = None items: list[MediaItem] = [] for raw in raw_items: item = _parse_album_item(raw) if item is not None: items.append(item) return items, next_page def _parse_album_item(raw: Any) -> MediaItem | None: """Parse a single album item array. Layout used in both AF blocks and snAcKc responses: [mediaKey, [url, w, h, ..., [byte_size]], captured_ms, dedupKey, timezoneOffsetMin, uploaded_ms, ..., {: ...}] """ if not isinstance(raw, list) or len(raw) < 2: return None visual = raw[1] if not isinstance(visual, list) or len(visual) < 3: return None url = visual[0] if not isinstance(url, str) or not url.startswith("http"): return None width = visual[1] if isinstance(visual[1], int) else None height = visual[2] if isinstance(visual[2], int) else None # Skip videos: their last element is a dict with key 76647426 (duration). if raw and isinstance(raw[-1], dict): if _VIDEO_DURATION_KEY in raw[-1] or "76647426" in raw[-1]: # Note: live photos also carry a duration but are still images; # we treat the presence of duration as "video". If a user reports # missing live photos we can revisit by checking 146008172 too. return None captured_at = raw[2] if len(raw) > 2 and _looks_like_timestamp_ms(raw[2]) else None uploaded_at = raw[5] if len(raw) > 5 and _looks_like_timestamp_ms(raw[5]) else None # File size, when present, lives at visual[-1][0] as a single int. byte_size: int | None = None if visual and isinstance(visual[-1], list) and visual[-1]: candidate = visual[-1][0] if isinstance(candidate, int) and candidate > 0: byte_size = candidate return MediaItem( url=_normalise_size(url, width, height), width=width, height=height, mime_type=None, filename=None, captured_at=captured_at, uploaded_at=uploaded_at, byte_size=byte_size, ) # Plausible epoch-ms range: 2000-01-01 to 2100-01-01. _MIN_TS_MS = 946_684_800_000 _MAX_TS_MS = 4_102_444_800_000 def _looks_like_timestamp_ms(value: Any) -> bool: return isinstance(value, int) and _MIN_TS_MS <= value <= _MAX_TS_MS # -- AF block parsing (for the initial 300 items embedded in the HTML) ------ _AF_BLOCK_RE = re.compile(r"AF_initDataCallback\s*\(\s*\{", re.DOTALL) _DATA_KEY_RE = re.compile(r"[\s,{]data\s*:\s*\[", re.DOTALL) def _iter_af_data_blobs(html: str): """Yield the JSON text of every AF_initDataCallback ``data:`` value.""" for m in _AF_BLOCK_RE.finditer(html): block_end = _balanced_close(html, m.end() - 1, "{", "}") if block_end is None: continue block = html[m.end() - 1: block_end + 1] data_match = _DATA_KEY_RE.search(block) if data_match is None: continue open_pos = data_match.end() - 1 close_pos = _balanced_close(block, open_pos, "[", "]") if close_pos is None: continue yield block[open_pos: close_pos + 1] def _balanced_close(s: str, open_idx: int, open_char: str, close_char: str) -> int | None: if open_idx >= len(s) or s[open_idx] != open_char: return None depth = 0 in_string: str | None = None i = open_idx n = len(s) while i < n: c = s[i] if in_string is not None: if c == "\\": i += 2 continue if c == in_string: in_string = None i += 1 continue if c in ("'", '"'): in_string = c i += 1 continue if c == open_char: depth += 1 elif c == close_char: depth -= 1 if depth == 0: return i i += 1 return None def _collect_album_item_lists(node: Any, _out: list[list[Any]] | None = None) -> list[list[Any]]: """Walk the AF data tree, collecting lists of album-item-shaped entries.""" out = _out if _out is not None else [] if isinstance(node, list): if _list_looks_like_album_items(node): out.append(node) for child in node: _collect_album_item_lists(child, out) elif isinstance(node, dict): for child in node.values(): _collect_album_item_lists(child, out) return out def _list_looks_like_album_items(lst: list[Any]) -> bool: if not lst: return False sample = lst[:20] for item in sample: if _parse_album_item(item) is None: return False return True # -- URL normalisation ------------------------------------------------------- def _normalise_size(url: str, width: int | None, height: int | None) -> str: """Strip any existing size suffix and request a 4K-capped version.""" base = _SIZE_SUFFIX_RE.sub("", url) if width and height: try: w = int(width) h = int(height) except (TypeError, ValueError): return f"{base}=w1920-h1080" longest = max(w, h) if longest > 3840: scale = 3840 / longest w = max(1, int(round(w * scale))) h = max(1, int(round(h * scale))) return f"{base}=w{w}-h{h}" return f"{base}=w1920-h1080" # Backwards-compatible names so the existing tests still find the helpers. def parse_album_html(html: str) -> list[MediaItem]: """Parse only the first-page items embedded in the HTML. Retained for backwards compatibility with the 0.5.0-rc1 surface; new callers should use ``fetch_album`` for the full paginated result. """ return _extract_first_page_items(html) _PHOTO_HOST_RE = re.compile( r"^https?://(?:[a-z0-9-]+\.)?(?:googleusercontent\.com|google\.com)/", re.IGNORECASE, ) def _is_dimension(v: Any) -> bool: return isinstance(v, int) and 16 <= v <= 20_000