pushframe 5.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. pushframe/__init__.py +4 -0
  2. pushframe/api/__init__.py +0 -0
  3. pushframe/api/accountApi.py +72 -0
  4. pushframe/api/activityApi.py +87 -0
  5. pushframe/api/assetApi.py +194 -0
  6. pushframe/api/baseApi.py +6 -0
  7. pushframe/api/frameApi.py +271 -0
  8. pushframe/api/notificationApi.py +15 -0
  9. pushframe/api/peopleApi.py +25 -0
  10. pushframe/api/playlistApi.py +9 -0
  11. pushframe/aura.py +182 -0
  12. pushframe/aws/__init__.py +0 -0
  13. pushframe/aws/awsclient.py +23 -0
  14. pushframe/aws/s3client.py +40 -0
  15. pushframe/aws/sqsclient.py +33 -0
  16. pushframe/cache.py +50 -0
  17. pushframe/cli.py +1134 -0
  18. pushframe/client.py +267 -0
  19. pushframe/exif.py +147 -0
  20. pushframe/export.py +53 -0
  21. pushframe/google/__init__.py +43 -0
  22. pushframe/google/bootstrap.py +134 -0
  23. pushframe/google/cache.py +167 -0
  24. pushframe/google/client.py +140 -0
  25. pushframe/google/enumerate.py +270 -0
  26. pushframe/google/manifest.py +111 -0
  27. pushframe/google/parsers.py +345 -0
  28. pushframe/google/redaction.py +33 -0
  29. pushframe/google/vault.py +126 -0
  30. pushframe/gsync.py +463 -0
  31. pushframe/migration.py +86 -0
  32. pushframe/models/__init__.py +0 -0
  33. pushframe/models/activity.py +79 -0
  34. pushframe/models/asset.py +159 -0
  35. pushframe/models/frame.py +105 -0
  36. pushframe/models/meta.py +11 -0
  37. pushframe/models/person.py +24 -0
  38. pushframe/models/user.py +22 -0
  39. pushframe/ratelimit.py +222 -0
  40. pushframe/reconcile.py +384 -0
  41. pushframe/sync.py +1105 -0
  42. pushframe/utils/dt.py +15 -0
  43. pushframe/utils/io.py +23 -0
  44. pushframe/utils/settings.py +59 -0
  45. pushframe-5.0.0.dist-info/METADATA +53 -0
  46. pushframe-5.0.0.dist-info/RECORD +49 -0
  47. pushframe-5.0.0.dist-info/WHEEL +4 -0
  48. pushframe-5.0.0.dist-info/entry_points.txt +2 -0
  49. pushframe-5.0.0.dist-info/licenses/LICENSE +31 -0
@@ -0,0 +1,167 @@
1
+ """Pruned-disk download cache for Google album items (phase 18 plan 18-01).
2
+
3
+ The cache is a STAGING area only — `cache_dir/<google_media_id>` files hold
4
+ the exact `=d` bytes (never re-encoded; LGS-06: the bytes decide) while they
5
+ wait for their confirmed frame upload, and are pruned afterwards (CSE-04).
6
+ Nothing in the pipeline ever diffs by walking it — demand is rebuilt from the
7
+ album listing plus the manifest (CSE-03, plan 18-02 builds that side).
8
+
9
+ SAFE-04 discipline: a download that fails (HTTP error, transport error, or a
10
+ Content-Length/expected-size mismatch) is recorded in CacheOutcome.failed and
11
+ its bytes are NEVER written, NEVER hashed — a failed or partial download can
12
+ never become uploadable junk bytes.
13
+
14
+ Concurrency is confined to this module (D-06 / CSE-01): a bounded worker pool
15
+ over the GoogleSession's httpx client. It imports nothing from the sync/apply
16
+ side of the project and never touches an Aura client, so concurrent code
17
+ structurally cannot reach the frame-write seam.
18
+ """
19
+ from __future__ import annotations
20
+
21
+ from concurrent.futures import ThreadPoolExecutor
22
+ from dataclasses import dataclass, field
23
+ from pathlib import Path
24
+
25
+ from pushframe.aws.s3client import get_md5
26
+
27
+
28
+ class CacheError(RuntimeError):
29
+ """Cache staging failed in a way that must abort the sync run."""
30
+
31
+
32
+ @dataclass
33
+ class CacheOutcome:
34
+ """The result of one download_to_cache run over an album listing.
35
+
36
+ staged entries carry {"google_media_id", "path", "md5_hash",
37
+ "size_bytes"} — verified, hash-annotated files ready for upload planning.
38
+ `failed` carries (google_media_id, error) pairs; those ids are excluded
39
+ from any plan (SAFE-04) and reported, never silently dropped.
40
+ """
41
+
42
+ staged: list[dict] = field(default_factory=list)
43
+ failed: list[tuple[str, str]] = field(default_factory=list)
44
+ skipped_manifest: int = 0
45
+ videos_skipped: int | None = None
46
+
47
+ @property
48
+ def staged_by_id(self) -> dict[str, dict]:
49
+ return {s["google_media_id"]: s for s in self.staged}
50
+
51
+
52
+ def _download_one(session, item: dict, cache_dir: Path,
53
+ expected_size: int | None) -> dict:
54
+ """Fetch ONE item's `=d` bytes, verify length, write, hash.
55
+
56
+ Raises on any failure — the caller records it in CacheOutcome.failed and
57
+ guarantees no file was left behind for the failed id.
58
+ """
59
+ google_media_id = item["id"]
60
+ resp = session.http.get(f"{item['base_url']}=d")
61
+ if resp.status_code != 200:
62
+ raise RuntimeError(f"=d download returned HTTP {resp.status_code}")
63
+ body = resp.content
64
+ declared = resp.headers.get("content-length")
65
+ # A stream transport may not populate Content-Length even for complete
66
+ # bodies; when the header is absent, the expected_size (phase 16's
67
+ # Range-GET measurement) is the length check.
68
+ if declared is not None and int(declared) != len(body):
69
+ raise RuntimeError(
70
+ f"truncated download: Content-Length {declared} != received {len(body)}"
71
+ )
72
+ if declared is None and expected_size is not None and expected_size != len(body):
73
+ raise RuntimeError(
74
+ f"truncated download: expected {expected_size} != received {len(body)}"
75
+ )
76
+ path = cache_dir / google_media_id
77
+ path.write_bytes(body)
78
+ return {
79
+ "google_media_id": google_media_id,
80
+ "path": str(path),
81
+ "md5_hash": get_md5(body),
82
+ "size_bytes": len(body),
83
+ }
84
+
85
+
86
+ def download_to_cache(session, listing, cache_dir: Path, *,
87
+ expected_sizes: dict[str, int] | None = None,
88
+ manifest=None, workers: int = 4,
89
+ progress=None) -> CacheOutcome:
90
+ """Download every listing item that the manifest does not already cover.
91
+
92
+ Manifest members are skipped entirely (D-07): steady state = zero
93
+ downloads, zero cache residency — their md5 lives in the manifest and the
94
+ frame already holds the bytes. Concurrency is a bounded pool (default 4,
95
+ injectable down to 1); results are merged after shutdown, so no shared
96
+ mutable state is touched during flight. `progress(google_media_id, ok)`
97
+ is invoked per RESOLVED item (success or failure) after the merge, so a
98
+ CLI can drive a progress bar (CSE-08) without thread races.
99
+ """
100
+ cache_dir = Path(cache_dir)
101
+ cache_dir.mkdir(parents=True, exist_ok=True)
102
+ expected_sizes = expected_sizes or {}
103
+ outcome = CacheOutcome()
104
+
105
+ todo: list[dict] = []
106
+ for item in listing.items:
107
+ if manifest is not None and manifest.entry_for(item["id"]) is not None:
108
+ outcome.skipped_manifest += 1
109
+ else:
110
+ todo.append(item)
111
+
112
+ if todo:
113
+ with ThreadPoolExecutor(max_workers=max(1, workers)) as pool:
114
+ futures = {
115
+ pool.submit(_download_one, session, item, cache_dir,
116
+ expected_sizes.get(item["id"])): item["id"]
117
+ for item in todo
118
+ }
119
+ for fut, google_media_id in futures.items():
120
+ try:
121
+ outcome.staged.append(fut.result())
122
+ except Exception as exc: # per-item isolation: one bad download
123
+ outcome.failed.append((google_media_id, str(exc)))
124
+ # A crash or transport error mid-write could strand a partial file;
125
+ # a failed id must leave NOTHING staged-looking behind (SAFE-04).
126
+ staged_ids = {s["google_media_id"] for s in outcome.staged}
127
+ for google_media_id, _err in outcome.failed:
128
+ orphan = cache_dir / google_media_id
129
+ if google_media_id not in staged_ids and orphan.exists():
130
+ orphan.unlink()
131
+
132
+ if progress is not None:
133
+ for s in outcome.staged:
134
+ progress(s["google_media_id"], True)
135
+ for google_media_id, _err in outcome.failed:
136
+ progress(google_media_id, False)
137
+
138
+ return outcome
139
+
140
+
141
+ def prune_cache(cache_dir: Path, manifest_ids: set[str]) -> int:
142
+ """Delete staged files whose google_media_id is manifest-backed (CSE-04).
143
+
144
+ Targeted deletions BY NAME — never a directory walk (CSE-03 discipline
145
+ applies to the whole module, pruning included). Files of un-manifested
146
+ ids (failed downloads awaiting retry) are untouched. An already-missing
147
+ file counts as pruned: pruning is idempotent and never raises.
148
+ """
149
+ cache_dir = Path(cache_dir)
150
+ if not manifest_ids:
151
+ return 0
152
+ pruned = 0
153
+ for google_media_id in manifest_ids:
154
+ f = cache_dir / google_media_id
155
+ if f.exists():
156
+ f.unlink()
157
+ pruned += 1
158
+ return pruned
159
+
160
+
161
+ def videos_skipped(listing, metadata_item_count: int | None) -> int | None:
162
+ """Named video count via the live-proven metadata delta (CSE-07 / D-10):
163
+ album metadata item count minus the photo count the walker enumerates.
164
+ Returns None when the metadata count is unknown (no delta to report)."""
165
+ if metadata_item_count is None:
166
+ return None
167
+ return metadata_item_count - len(listing.items)
@@ -0,0 +1,140 @@
1
+ """Google session client: full-cookie-jar httpx over photos.google.com.
2
+
3
+ Migrated from probes/browser_bootstrap.py's live-proven session builder
4
+ (phase 16). Two live-proven requirements drive the shape:
5
+
6
+ 1. photos.google.com treats a FLATTENED name→value cookie dict as an
7
+ anonymous visitor (redirect to the marketing page), while the complete
8
+ jar — domain+path preserved from the harvest — yields the logged-in app
9
+ page with the SNlM0e at-token. The jar is therefore built from the
10
+ vault's FULL records, never from `cookies_for_httpx()`.
11
+ 2. The User-Agent must match the harvesting browser's (Chrome); a
12
+ mismatched UA gets redirected to `about/`.
13
+
14
+ The `transport=` seam mirrors `pushframe.client.Client`'s DI: offline tests
15
+ inject an httpx.MockTransport and never touch the network (TEST-02).
16
+
17
+ NOTE: the vault denylist refuses `pushframe.cli` — the CLI reaches the
18
+ session only through this package's `GoogleSession.from_vault`, which is the
19
+ boundary's intended enforcement point.
20
+ """
21
+ from __future__ import annotations
22
+
23
+ import hashlib
24
+ import re
25
+ import time
26
+
27
+ import httpx
28
+
29
+ from pushframe.google import vault
30
+
31
+ PHOTOS_HOME = "https://photos.google.com/"
32
+
33
+ # UA of the harvesting browser (live-proven: required, else redirect to about/).
34
+ HARVEST_UA = ("Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
35
+ "(KHTML, like Gecko) Chrome/153.0.0.0 Safari/537.36")
36
+
37
+ _AUTH_COOKIE_NAMES = ("SAPISID", "__Secure-1PAPISID", "__Secure-3PAPISID")
38
+
39
+ # Account email on the logged-in home page: the "oPEP7c" key with the
40
+ # account address as value (whitespace-tolerant on the JSON separator).
41
+ _EMAIL_RE = re.compile(r'"oPEP7c"\s*[,:]\s*"([^"@]+@[^"]+)"')
42
+
43
+ # The SNlM0e at-token (XSRF boilerplate batchexecute needs).
44
+ _SNLMOE_RE = re.compile(r'"SNlM0e"\s*:\s*"([^"]+)"')
45
+
46
+
47
+ class GoogleSessionError(RuntimeError):
48
+ """Session-level failure: vault unusable or a page fetch failing loud."""
49
+
50
+
51
+ class GoogleSession:
52
+ """An httpx session over photos.google.com built from the full cookie jar.
53
+
54
+ Exposes `transport=` for offline tests (httpx.MockTransport), the
55
+ session-usable check (`is_linked`), the SNlM0e at-token extraction, and
56
+ the account email (for `status` — email only, never cookie values, D-02).
57
+ """
58
+
59
+ def __init__(self, cookie_records: list[dict], *, transport: httpx.BaseTransport | None = None,
60
+ user_agent: str = HARVEST_UA, timeout: float = 30.0) -> None:
61
+ self.cookie_records = cookie_records
62
+ self._sapisid: str | None = next(
63
+ (c["value"] for c in cookie_records if c.get("name") in _AUTH_COOKIE_NAMES), None)
64
+ # FULL jar: domain+path preserved (live-proven requirement — a
65
+ # flattened dict reads as anonymous).
66
+ jar = httpx.Cookies()
67
+ for c in cookie_records:
68
+ if "name" in c and "value" in c:
69
+ jar.set(c["name"], c["value"],
70
+ domain=c.get("domain", ".google.com"), path=c.get("path", "/"))
71
+ self.http = httpx.Client(
72
+ cookies=jar,
73
+ timeout=timeout,
74
+ follow_redirects=True,
75
+ transport=transport,
76
+ headers={"User-Agent": user_agent,
77
+ "Accept-Language": "fr-FR,fr;q=0.9,en;q=0.8"},
78
+ )
79
+ self._home_text: str | None = None
80
+
81
+ @classmethod
82
+ def from_vault(cls, *, path: str | None = None,
83
+ **kwargs) -> "GoogleSession":
84
+ """Build a session from the cookie vault (denylist check fires inside
85
+ vault.load()). Falls back to the legacy probe vault path (soft migration).
86
+ """
87
+ return cls(vault.load(path=path), **kwargs)
88
+
89
+ @property
90
+ def sapisid(self) -> str | None:
91
+ """SAPISID value from the harvest, for SAPISIDHASH Authorization."""
92
+ return self._sapisid
93
+
94
+ def home_text(self) -> str:
95
+ """The logged-in photos home page text (fetched once, then cached)."""
96
+ if self._home_text is None:
97
+ resp = self.http.get(PHOTOS_HOME)
98
+ if resp.status_code != 200:
99
+ raise GoogleSessionError(
100
+ f"photos.google.com home returned HTTP {resp.status_code} "
101
+ f"with session cookies attached — failing loud"
102
+ )
103
+ self._home_text = resp.text
104
+ return self._home_text
105
+
106
+ def is_linked(self) -> bool:
107
+ """Session-usable check (no exception on a dead session).
108
+
109
+ Linked iff the photos home answers 200 with the SNlM0e at-token. A
110
+ dead/expired session gets redirected to the marketing page (no
111
+ at-token) — the live-proven anonymous signal.
112
+ """
113
+ try:
114
+ m = _SNLMOE_RE.search(self.home_text())
115
+ except GoogleSessionError:
116
+ return False
117
+ return m is not None
118
+
119
+ def at_token(self) -> str | None:
120
+ """The SNlM0e at-token (XSRF boilerplate for batchexecute calls)."""
121
+ m = _SNLMOE_RE.search(self.home_text())
122
+ return m.group(1) if m else None
123
+
124
+ def account_email(self) -> str | None:
125
+ """The linked account's email address (for `status`; D-02: email only,
126
+ never cookie values). None when the page shape carries no email row."""
127
+ m = _EMAIL_RE.search(self.home_text())
128
+ return m.group(1) if m else None
129
+
130
+ def authorization_header(self, origin: str = "https://photos.google.com") -> str | None:
131
+ """SAPISIDHASH Authorization value, or None when no SAPISID cookie —
132
+ some batchexecute calls expect it (xob0t/Google-Photos-Toolkit shape)."""
133
+ if not self._sapisid:
134
+ return None
135
+ now_ms = int(time.time() * 1000)
136
+ digest = hashlib.sha1(f"{now_ms} {self._sapisid} {origin}".encode()).hexdigest()
137
+ return f"SAPISIDHASH {now_ms}_{digest}"
138
+
139
+ def close(self) -> None:
140
+ self.http.close()
@@ -0,0 +1,270 @@
1
+ """Album enumeration and disk-weight measurement over photos.google.com.
2
+
3
+ RPC-FIRST protocol (live-proven 2026-09-28, this session — supersedes the
4
+ phase-16 share-page bootstrap for the linked flow):
5
+
6
+ - Batch-1 IS a snAcKc call with a NULL continuation:
7
+ `snAcKc(album_id, null, null, page_key)` answers HTTP 200 with the first
8
+ 300 items — the share page is never fetched. (The authenticated share
9
+ page is an SPA shell: no ds:1 block, no AH_ cursor — the phase-16
10
+ share-page batch-1 only ever worked for the ANONYMOUS flow.)
11
+ - `page_key` (4th argument) is OPTIONAL: a null page_key still answers
12
+ (300 items on the live 1096-item album). The /albums listing carries it,
13
+ so it is sent when known.
14
+ - Continuation: the LAST AH_ token in each payload is the cursor; no token
15
+ = exhausted. Live proof: 300+300+300+194 = 1094, clean exhaustion.
16
+ - The album's item-count METADATA (the /albums ds:5 value) can exceed the
17
+ photo count the walker returns (live: 1096 metadata vs 1094 photos — the
18
+ delta matches the album's videos, which the §1b photo walker skips).
19
+ The authoritative photo count is the enumeration's.- scalars (SNlM0e at-token, FdrFJe f.sid, cfb2h bl) come from the logged-in
20
+ photos home page; the f.req envelope is TRIPLE-nested (double nesting
21
+ answers HTTP 400 — the phase-16 finding that started this all).
22
+ - ONE bounded retry (post-phase-17 hardening, live smoke 2026-09-28):
23
+ Google occasionally answers HTTP 200 with a well-formed wrb.fr/snAcKc
24
+ entry whose inner payload is null. The exact call is re-issued once
25
+ (same cursor — a null answer consumed no page); a null again fails
26
+ loud. The budget is one retry per enumeration, not per page.
27
+
28
+ Fail-loud discipline (T-17-02): HTTP != 200, a malformed envelope, a
29
+ persisting null payload or a page-cap overrun raises — never a silent
30
+ partial listing.
31
+ """
32
+ from __future__ import annotations
33
+
34
+ import json
35
+ from dataclasses import dataclass, field
36
+ from urllib.parse import quote
37
+
38
+ import httpx
39
+
40
+ from pushframe.google.parsers import (
41
+ ProbeParseError,
42
+ parse_batchexecute,
43
+ parse_snackc_payload,
44
+ )
45
+ from pushframe.google.redaction import redact_tokens
46
+
47
+ BATCHEXECUTE_URL = "https://photos.google.com/_/PhotosUi/data/batchexecute"
48
+ PHOTOS_HOME = "https://photos.google.com/"
49
+ PHOTOS_ALBUMS = "https://photos.google.com/albums"
50
+
51
+ MAX_PAGES = 60 # 60 x 300 = 18000, far beyond any real album (T-17-04 cap)
52
+ PAGE_SIZE = 300 # live-proven page size
53
+
54
+ # ONE bounded retry for the live-observed transient (2026-09-28 smoke):
55
+ # HTTP 200 carrying a well-formed snAcKc entry whose inner payload is
56
+ # null. The budget is GLOBAL to one enumerate_album run — one recovery
57
+ # per enumeration, not per page — and a null answer never consumed a
58
+ # page, so the retry re-issues the exact same call (same cursor).
59
+ _retry_budget = {"snAcKc_null_payload": 1}
60
+
61
+ # WIZ_global_data scalars on the home page.
62
+ _FSID_RE = _FSID_RE = None # replaced below (kept name stable for tests)
63
+ _FSID_RE = __import__("re").compile(r'"FdrFJe"\s*:\s*"([^"]+)"')
64
+ _BL_RE = __import__("re").compile(r'"cfb2h"\s*:\s*"([^"]+)"')
65
+ _AT_RE = __import__("re").compile(r'"SNlM0e"\s*:\s*"([^"]+)"')
66
+
67
+
68
+ class EnumerateError(RuntimeError):
69
+ """Album enumeration failed — fail loud, never a silent partial listing."""
70
+
71
+
72
+ @dataclass
73
+ class AlbumListing:
74
+ """The complete, exhausted album listing (D-06: every item, disk weight
75
+ via measure_disk_weight, photo count cross-checkable against the UI)."""
76
+
77
+ items: list[dict] = field(default_factory=list)
78
+ page_count: int = 0
79
+ exhausted_cleanly: bool = False
80
+ album_id: str | None = None
81
+
82
+ @property
83
+ def total_bytes(self) -> int:
84
+ return sum(i.get("bytes", 0) for i in self.items)
85
+
86
+
87
+ @dataclass
88
+ class _SessionScalars:
89
+ at: str
90
+ fsid: str
91
+ bl: str
92
+
93
+
94
+ def _session_scalars(session) -> _SessionScalars:
95
+ """Extract at-token / f.sid / bl from the logged-in photos home page."""
96
+ text = session.home_text()
97
+ at_m = _AT_RE.search(text)
98
+ fsid_m = _FSID_RE.search(text)
99
+ bl_m = _BL_RE.search(text)
100
+ if not (at_m and fsid_m and bl_m):
101
+ missing = [n for n, m in (("SNlM0e", at_m), ("FdrFJe", fsid_m),
102
+ ("cfb2h", bl_m)) if not m]
103
+ raise EnumerateError(
104
+ f"photos.google.com home lacks required batchexecute scalars: "
105
+ f"{missing} — session dead or page shape changed; failing loud"
106
+ )
107
+ return _SessionScalars(at=at_m.group(1), fsid=fsid_m.group(1), bl=bl_m.group(1))
108
+
109
+
110
+ def _freq_envelope(rpcid: str, args: list) -> str:
111
+ """The TRIPLE-nested compact f.req envelope (live-proven: double nesting
112
+ answers HTTP 400). Shape: [[ [rpcid, <stringified args>, null, "generic"] ]] —
113
+ the 4-element entry sits at the third nesting level, its 2nd member being
114
+ the JSON-encoded args STRING."""
115
+ entry = [rpcid, json.dumps(args, separators=(",", ":")), None, "generic"]
116
+ return json.dumps([[entry]], separators=(",", ":"))
117
+
118
+
119
+ def _rpc_headers(session) -> dict[str, str]:
120
+ headers = {
121
+ "Content-Type": "application/x-www-form-urlencoded;charset=UTF-8",
122
+ "Origin": PHOTOS_HOME.rstrip("/"),
123
+ "Referer": PHOTOS_HOME,
124
+ }
125
+ auth = session.authorization_header()
126
+ if auth:
127
+ headers["Authorization"] = auth
128
+ return headers
129
+
130
+
131
+ def _snackc_page(session, scalars: _SessionScalars, album_id: str,
132
+ page_key: str | None,
133
+ continuation_token: str | None) -> tuple[list[dict], str | None]:
134
+ """Issue ONE snAcKc batchexecute call; return (items, next_token_or_None).
135
+
136
+ continuation_token=None is the live-proven batch-1 form. A null inner
137
+ payload on HTTP 200 (live-observed transient) is retried exactly once
138
+ with the same cursor while the global budget lasts; a persisting null
139
+ fails loud rather than emitting a partial listing."""
140
+ args = [album_id, continuation_token, None, page_key]
141
+ freq = _freq_envelope("snAcKc", args)
142
+ url = (
143
+ f"{BATCHEXECUTE_URL}?rpcids=snAcKc"
144
+ f"&source-path={quote('/share/' + album_id, safe='')}"
145
+ f"&f.sid={quote(scalars.fsid, safe='')}"
146
+ f"&bl={quote(scalars.bl, safe='')}"
147
+ f"&hl=fr&soc-app=165&soc-platform=1&soc-device=1"
148
+ )
149
+ body = f"f.req={quote(freq, safe='')}&at={quote(scalars.at, safe='')}&"
150
+ headers = _rpc_headers(session)
151
+
152
+ def _post() -> httpx.Response:
153
+ return session.http.post(url, content=body.encode("utf-8"),
154
+ headers=headers)
155
+
156
+ def _snackc_entries(resp: httpx.Response) -> list:
157
+ if resp.status_code != 200:
158
+ raise EnumerateError(
159
+ f"snAcKc batchexecute returned HTTP {resp.status_code} "
160
+ f"(album {album_id[:12]}…, continuation sent: "
161
+ f"{continuation_token is not None}) — failing loud; body head: "
162
+ f"{redact_tokens(resp.text[:200])}"
163
+ )
164
+ entries = [e for e in parse_batchexecute(resp.text)
165
+ if e.rpcid == "snAcKc"]
166
+ if not entries:
167
+ raise EnumerateError(
168
+ "snAcKc batchexecute answer carries no wrb.fr/snAcKc entry — "
169
+ "malformed envelope; refusing to guess"
170
+ )
171
+ return entries
172
+
173
+ entries = _snackc_entries(_post())
174
+ if entries[0].payload is None and _retry_budget["snAcKc_null_payload"] > 0:
175
+ _retry_budget["snAcKc_null_payload"] -= 1
176
+ entries = _snackc_entries(_post())
177
+ if entries[0].payload is None:
178
+ raise EnumerateError(
179
+ "snAcKc carries a null inner payload on HTTP 200 (live-observed "
180
+ "transient); the single bounded retry did not recover it — "
181
+ "failing loud, never a partial listing"
182
+ )
183
+ page = parse_snackc_payload(entries[0].payload)
184
+ return page.items, page.continuation_token
185
+
186
+
187
+ def enumerate_album(session, album_id: str, *, page_key: str | None = None,
188
+ max_pages: int = MAX_PAGES) -> AlbumListing:
189
+ """Enumerate EVERY photo of an album (D-06) via the RPC-first protocol:
190
+
191
+ batch-1 = snAcKc(album_id, None, None, page_key) → 300 items; swap the
192
+ LAST AH_ cursor per page until no token or a 0-item page. Fail-loud on
193
+ HTTP != 200 or a malformed envelope (never a silent partial listing).
194
+
195
+ `album_id` is the SHARE token (the /share/<id> path segment — what the
196
+ /albums listing carries and what the live proof used). `page_key` is
197
+ the share URL's ?key= value when known (optional, live-proven).
198
+
199
+ A null-payload snAcKc answer on HTTP 200 (live-observed transient) is
200
+ retried once per run with the same cursor — see _snackc_page.
201
+ """
202
+ if not album_id or not isinstance(album_id, str):
203
+ raise EnumerateError("album_id is required — refusing to guess")
204
+ _retry_budget["snAcKc_null_payload"] = 1 # one bounded retry per run
205
+ scalars = _session_scalars(session)
206
+
207
+ listing = AlbumListing(album_id=album_id)
208
+ token: str | None = None
209
+ while listing.page_count < max_pages:
210
+ page_items, next_token = _snackc_page(session, scalars, album_id,
211
+ page_key, token)
212
+ listing.page_count += 1
213
+ known = {i["id"] for i in listing.items}
214
+ fresh = [i for i in page_items if i["id"] not in known]
215
+ listing.items.extend(fresh)
216
+ if next_token is None or not page_items:
217
+ listing.exhausted_cleanly = True
218
+ return listing
219
+ token = next_token
220
+
221
+ raise EnumerateError(
222
+ f"album enumeration hit the {max_pages}-page cap (T-17-04) with "
223
+ f"{len(listing.items)} items and a live continuation token — "
224
+ f"listing is INCOMPLETE; raise max_pages or investigate"
225
+ )
226
+
227
+
228
+ def list_shared_albums(session) -> list:
229
+ """The account's shared albums, from photos.google.com/albums' ds:5 block
230
+ (live-proven: 3 albums with titles, share tokens, base64 page_keys and
231
+ metadata item counts; the 'Cadre' row's count = 24 matched the UI).
232
+
233
+ Note: the home page's ds:5 is the photo feed — the /albums page is the
234
+ one that carries the album cards.
235
+ """
236
+ from pushframe.google.parsers import extract_initdata, parse_album_summaries
237
+
238
+ resp = session.http.get(PHOTOS_ALBUMS)
239
+ if resp.status_code != 200:
240
+ raise EnumerateError(
241
+ f"photos.google.com/albums returned HTTP {resp.status_code} "
242
+ f"— failing loud"
243
+ )
244
+ try:
245
+ ds5 = extract_initdata(resp.text, "ds:5")
246
+ except ProbeParseError:
247
+ # An account with no albums renders a ds:5 without card rows — a
248
+ # legitimate empty listing, not a parse failure.
249
+ return []
250
+ return parse_album_summaries(ds5)
251
+
252
+
253
+ def measure_disk_weight(session, base_urls: list[str]) -> list[int]:
254
+ """Exact per-item byte sizes via 1-octet Range GETs on `{baseUrl}=d`.
255
+
256
+ Google answers `Content-Range: bytes 0-0/TOTAL` — the album's total disk
257
+ weight is measurable without downloading any photo (live-proven: album C,
258
+ 24 items → 86.6 MiB exact). Items whose answer carries neither
259
+ Content-Range nor Content-Length report 0 (defensive; live never seen).
260
+ """
261
+ sizes: list[int] = []
262
+ for base in base_urls:
263
+ r = session.http.get(f"{base}=d", headers={"Range": "bytes=0-0"})
264
+ cr = r.headers.get("content-range", "")
265
+ size = int(cr.rsplit("/", 1)[-1]) if "/" in cr else None
266
+ if size is None:
267
+ cl = r.headers.get("content-length")
268
+ size = int(cl) if cl else 0
269
+ sizes.append(size)
270
+ return sizes
@@ -0,0 +1,111 @@
1
+ """Persistent google_media_id → md5_hash manifest (phase 18 plan 18-01).
2
+
3
+ This file is the sole memory that survives cache pruning: a manifest entry
4
+ MEANS "this photo lives on the frame" (its md5_hash was confirmed uploaded).
5
+ The mirror plan is rebuilt from the album listing plus THIS file — never from
6
+ a directory walk of the (pruned) cache (CSE-02/CSE-03).
7
+
8
+ Privacy posture: ids, hashes and sizes only — no cookie material, no
9
+ capability URLs. The cookie vault's structural denylist is therefore not
10
+ implicated by this module, but the vault's FILE discipline is mirrored:
11
+ 0600 permissions and an atomic temp+rename write so a crash mid-save can
12
+ never leave a truncated manifest behind.
13
+
14
+ Entry shape is closed (fail loud on anything else):
15
+
16
+ {"md5_hash": str, "size_bytes": int,
17
+ "album_share_token": str, "first_synced": iso8601}
18
+ """
19
+ from __future__ import annotations
20
+
21
+ import json
22
+ import os
23
+ from dataclasses import dataclass, field
24
+ from datetime import datetime, timezone
25
+ from pathlib import Path
26
+
27
+ DEFAULT_MANIFEST_PATH = Path("~/.config/pushframe/google-manifest.json")
28
+
29
+ # The ONLY keys a manifest entry may carry. Anything else fails loud on load —
30
+ # this file must never become a bag of unvetted per-id state (and can never
31
+ # accumulate secrets without tripping this guard).
32
+ _ENTRY_KEYS = frozenset(
33
+ {"md5_hash", "size_bytes", "album_share_token", "first_synced"}
34
+ )
35
+
36
+
37
+ class ManifestError(RuntimeError):
38
+ """Manifest absent-malformed, or an entry violating the closed shape."""
39
+
40
+
41
+ def _now_iso() -> str:
42
+ return datetime.now(timezone.utc).isoformat(timespec="seconds")
43
+
44
+
45
+ def _validate_entry(google_media_id: str, entry: object) -> dict:
46
+ if not isinstance(entry, dict):
47
+ raise ManifestError(
48
+ f"manifest entry for {google_media_id!r} is not an object — "
49
+ f"refusing to guess"
50
+ )
51
+ keys = set(entry)
52
+ if keys != set(_ENTRY_KEYS):
53
+ unexpected = keys - _ENTRY_KEYS
54
+ missing = _ENTRY_KEYS - keys
55
+ raise ManifestError(
56
+ f"manifest entry for {google_media_id!r} violates the closed "
57
+ f"shape — unexpected keys: {sorted(unexpected)}, "
58
+ f"missing: {sorted(missing)}"
59
+ )
60
+ if not isinstance(entry["md5_hash"], str) or not entry["md5_hash"]:
61
+ raise ManifestError(f"entry {google_media_id!r}: md5_hash must be a non-empty string")
62
+ if not isinstance(entry["size_bytes"], int) or isinstance(entry["size_bytes"], bool):
63
+ raise ManifestError(f"entry {google_media_id!r}: size_bytes must be an int")
64
+ return entry
65
+
66
+
67
+ @dataclass
68
+ class GoogleManifest:
69
+ """The in-memory manifest map with load/save and entry access."""
70
+
71
+ entries: dict[str, dict] = field(default_factory=dict)
72
+
73
+ @classmethod
74
+ def load(cls, path: str | Path | None = None) -> "GoogleManifest":
75
+ """Read the manifest; an ABSENT file is a legitimate empty manifest
76
+ (first run), while a present-but-malformed one fails loud."""
77
+ p = Path(path).expanduser() if path else DEFAULT_MANIFEST_PATH.expanduser()
78
+ if not p.exists():
79
+ return cls()
80
+ data = json.loads(p.read_text())
81
+ if not isinstance(data, dict):
82
+ raise ManifestError(f"manifest at {p} is not a JSON object — refusing to guess")
83
+ return cls(entries={mid: _validate_entry(mid, e) for mid, e in data.items()})
84
+
85
+ def entry_for(self, google_media_id: str) -> dict | None:
86
+ return self.entries.get(google_media_id)
87
+
88
+ def add(self, google_media_id: str, *, md5_hash: str, size_bytes: int,
89
+ album_share_token: str, first_synced: str | None = None) -> None:
90
+ self.entries[google_media_id] = {
91
+ "md5_hash": md5_hash,
92
+ "size_bytes": size_bytes,
93
+ "album_share_token": album_share_token,
94
+ "first_synced": first_synced or _now_iso(),
95
+ }
96
+
97
+ def save(self, path: str | Path | None = None) -> Path:
98
+ """Persist atomically (temp file + os.replace) with 0600 permissions —
99
+ a crash mid-save can never leave a truncated manifest behind."""
100
+ p = Path(path).expanduser() if path else DEFAULT_MANIFEST_PATH.expanduser()
101
+ p.parent.mkdir(parents=True, exist_ok=True)
102
+ payload = json.dumps(self.entries, indent=2, sort_keys=True).encode("utf-8")
103
+ tmp = p.with_name(p.name + ".tmp")
104
+ fd = os.open(tmp, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
105
+ try:
106
+ os.write(fd, payload)
107
+ finally:
108
+ os.close(fd)
109
+ os.replace(tmp, p)
110
+ os.chmod(p, 0o600) # in case the file pre-existed with looser mode
111
+ return p