pushframe 5.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pushframe/__init__.py +4 -0
- pushframe/api/__init__.py +0 -0
- pushframe/api/accountApi.py +72 -0
- pushframe/api/activityApi.py +87 -0
- pushframe/api/assetApi.py +194 -0
- pushframe/api/baseApi.py +6 -0
- pushframe/api/frameApi.py +271 -0
- pushframe/api/notificationApi.py +15 -0
- pushframe/api/peopleApi.py +25 -0
- pushframe/api/playlistApi.py +9 -0
- pushframe/aura.py +182 -0
- pushframe/aws/__init__.py +0 -0
- pushframe/aws/awsclient.py +23 -0
- pushframe/aws/s3client.py +40 -0
- pushframe/aws/sqsclient.py +33 -0
- pushframe/cache.py +50 -0
- pushframe/cli.py +1134 -0
- pushframe/client.py +267 -0
- pushframe/exif.py +147 -0
- pushframe/export.py +53 -0
- pushframe/google/__init__.py +43 -0
- pushframe/google/bootstrap.py +134 -0
- pushframe/google/cache.py +167 -0
- pushframe/google/client.py +140 -0
- pushframe/google/enumerate.py +270 -0
- pushframe/google/manifest.py +111 -0
- pushframe/google/parsers.py +345 -0
- pushframe/google/redaction.py +33 -0
- pushframe/google/vault.py +126 -0
- pushframe/gsync.py +463 -0
- pushframe/migration.py +86 -0
- pushframe/models/__init__.py +0 -0
- pushframe/models/activity.py +79 -0
- pushframe/models/asset.py +159 -0
- pushframe/models/frame.py +105 -0
- pushframe/models/meta.py +11 -0
- pushframe/models/person.py +24 -0
- pushframe/models/user.py +22 -0
- pushframe/ratelimit.py +222 -0
- pushframe/reconcile.py +384 -0
- pushframe/sync.py +1105 -0
- pushframe/utils/dt.py +15 -0
- pushframe/utils/io.py +23 -0
- pushframe/utils/settings.py +59 -0
- pushframe-5.0.0.dist-info/METADATA +53 -0
- pushframe-5.0.0.dist-info/RECORD +49 -0
- pushframe-5.0.0.dist-info/WHEEL +4 -0
- pushframe-5.0.0.dist-info/entry_points.txt +2 -0
- pushframe-5.0.0.dist-info/licenses/LICENSE +31 -0
|
@@ -0,0 +1,345 @@
|
|
|
1
|
+
"""Parsers for Google Photos share pages and snAcKc batchexecute payloads.
|
|
2
|
+
|
|
3
|
+
Migrated from probes/shared_link_probe.py (phase 16, live-proven against a
|
|
4
|
+
24-item and a 794-item album). The media-item shape is shared by BOTH the
|
|
5
|
+
share page's ds:1 payload and the snAcKc RPC inner payload (ALBUM-ACCESS.md
|
|
6
|
+
§1b):
|
|
7
|
+
|
|
8
|
+
[mediaItemId, [baseUrl, width, height, ...], uploadTimestampMs, ...]
|
|
9
|
+
|
|
10
|
+
Fail-loud convention: ProbeParseError names the missing structure; a partial
|
|
11
|
+
or silent item list is never emitted (T-17-02).
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import re
|
|
17
|
+
import sys
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
|
|
20
|
+
_DS1_RE = re.compile(r"AF_initDataCallback\(\s*\{\s*key:\s*['\"]ds:1['\"]")
|
|
21
|
+
|
|
22
|
+
# Any AF_initDataCallback block, key captured (ds:0 album headers, ds:1 media,
|
|
23
|
+
# ds:N anything else the frontend carries).
|
|
24
|
+
_ANY_DS_RE = re.compile(r"AF_initDataCallback\(\s*\{\s*key:\s*['\"](ds:\d+)['\"]")
|
|
25
|
+
|
|
26
|
+
# snAcKc continuation cursors: AH_ followed by 40+ URL-safe chars (live-proven).
|
|
27
|
+
AH_TOKEN_RE = re.compile(r"AH_[A-Za-z0-9_-]{40,}")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class ProbeParseError(RuntimeError):
|
|
31
|
+
"""Raised when a Google page/payload lacks the expected structure (fail-loud)."""
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def extract_initdata(html: str, key: str) -> list:
|
|
35
|
+
"""Extract and JSON-parse the `data` argument of the given ds:N AF_initDataCallback.
|
|
36
|
+
|
|
37
|
+
Regex-locates the `key: 'ds:N'` occurrences, then performs balanced-bracket
|
|
38
|
+
extraction of the `data:[...]` argument and parses it as a JS array literal.
|
|
39
|
+
Raises ProbeParseError (naming the missing key) when the page has no such
|
|
40
|
+
block — fail loud, never emit a partial item list silently.
|
|
41
|
+
"""
|
|
42
|
+
matches = [m for m in _ANY_DS_RE.finditer(html) if m.group(1) == key]
|
|
43
|
+
if not matches:
|
|
44
|
+
raise ProbeParseError(
|
|
45
|
+
f"page has no AF_initDataCallback with key '{key}' — page shape changed "
|
|
46
|
+
"or the link did not resolve to an album page; refusing to guess"
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
# Try every occurrence (pages can carry several; non-payload lookalikes
|
|
50
|
+
# — e.g. prose or comments naming the structure — fail their parse and the
|
|
51
|
+
# walk continues to the next occurrence). First parseable payload wins.
|
|
52
|
+
last_error: ProbeParseError | None = None
|
|
53
|
+
for match in matches:
|
|
54
|
+
try:
|
|
55
|
+
return _extract_data_at(html, match, key=key)
|
|
56
|
+
except ProbeParseError as exc:
|
|
57
|
+
last_error = exc
|
|
58
|
+
continue
|
|
59
|
+
raise ProbeParseError(
|
|
60
|
+
f"no parseable {key} data block among {len(matches)} occurrence(s): "
|
|
61
|
+
f"{last_error} — refusing to guess"
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def extract_ds1_data(html: str) -> list:
|
|
66
|
+
"""Extract and JSON-parse the `data` argument of the ds:1 AF_initDataCallback
|
|
67
|
+
(the shared-album media payload). Thin wrapper over `extract_initdata`."""
|
|
68
|
+
return extract_initdata(html, "ds:1")
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _extract_data_at(html: str, match: re.Match, *, key: str = "ds:1") -> list:
|
|
72
|
+
# From the match, find the `data:` argument's opening bracket.
|
|
73
|
+
tail = html[match.end():]
|
|
74
|
+
data_m = re.search(r"\bdata\s*:", tail)
|
|
75
|
+
if not data_m:
|
|
76
|
+
raise ProbeParseError(
|
|
77
|
+
"ds:1 callback found but has no 'data:' argument — refusing to guess"
|
|
78
|
+
)
|
|
79
|
+
after = tail[data_m.end():]
|
|
80
|
+
bracket_m = re.search(r"\[", after)
|
|
81
|
+
if not bracket_m:
|
|
82
|
+
raise ProbeParseError("data argument carries no opening '[' — refusing to guess")
|
|
83
|
+
|
|
84
|
+
start = match.end() + data_m.end() + bracket_m.start()
|
|
85
|
+
depth = 0
|
|
86
|
+
end = None
|
|
87
|
+
in_str = False
|
|
88
|
+
esc = False
|
|
89
|
+
quote = ""
|
|
90
|
+
for i in range(start, len(html)):
|
|
91
|
+
ch = html[i]
|
|
92
|
+
if in_str:
|
|
93
|
+
if esc:
|
|
94
|
+
esc = False
|
|
95
|
+
elif ch == "\\":
|
|
96
|
+
esc = True
|
|
97
|
+
elif ch == quote:
|
|
98
|
+
in_str = False
|
|
99
|
+
continue
|
|
100
|
+
if ch in ("'", '"'):
|
|
101
|
+
in_str = True
|
|
102
|
+
quote = ch
|
|
103
|
+
elif ch == "[":
|
|
104
|
+
depth += 1
|
|
105
|
+
elif ch == "]":
|
|
106
|
+
depth -= 1
|
|
107
|
+
if depth == 0:
|
|
108
|
+
end = i + 1
|
|
109
|
+
break
|
|
110
|
+
if end is None:
|
|
111
|
+
raise ProbeParseError("unbalanced brackets in ds:1 data payload — truncated page?")
|
|
112
|
+
|
|
113
|
+
literal = html[start:end]
|
|
114
|
+
return _parse_array_literal(literal, key=key)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _parse_array_literal(literal: str, *, key: str = "ds:1") -> list:
|
|
118
|
+
"""Parse a JS array literal: strict json.loads first, tolerant fallback second.
|
|
119
|
+
|
|
120
|
+
The payload is normally plain JSON (double-quoted). When Google emits JS-isms
|
|
121
|
+
(single quotes, bare keys, trailing commas), a conservative sanitizer normalizes
|
|
122
|
+
only those cases — no bespoke parser beyond balanced extraction, per the plan.
|
|
123
|
+
"""
|
|
124
|
+
try:
|
|
125
|
+
parsed = json.loads(literal)
|
|
126
|
+
except json.JSONDecodeError as first_error:
|
|
127
|
+
sanitized = literal
|
|
128
|
+
# Strip // line comments if any leaked in.
|
|
129
|
+
sanitized = re.sub(r"^\s*//.*$", "", sanitized, flags=re.MULTILINE)
|
|
130
|
+
# Quote bare object keys: {foo: 1} -> {"foo": 1}
|
|
131
|
+
sanitized = re.sub(r"([{,]\s*)([A-Za-z_][A-Za-z0-9_]*)(\s*):", r'\1"\2"\3:', sanitized)
|
|
132
|
+
# Trailing commas: [1,2,] -> [1,2]
|
|
133
|
+
sanitized = re.sub(r",\s*([\]}])", r"\1", sanitized)
|
|
134
|
+
try:
|
|
135
|
+
parsed = json.loads(sanitized)
|
|
136
|
+
except json.JSONDecodeError: raise ProbeParseError(
|
|
137
|
+
f"{key} data is neither strict JSON nor tolerantly sanitizable "
|
|
138
|
+
f"(first error: {first_error.msg} at {first_error.pos}) — refusing to guess"
|
|
139
|
+
) from first_error
|
|
140
|
+
print("parse note: ds:1 payload needed tolerant sanitization (JS-isms present)",
|
|
141
|
+
file=sys.stderr)
|
|
142
|
+
if not isinstance(parsed, list):
|
|
143
|
+
raise ProbeParseError(f"{key} data is not an array — page shape changed")
|
|
144
|
+
return parsed
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _walk_items(node, items):
|
|
148
|
+
"""Depth-first walk collecting media items with the §1b structure:
|
|
149
|
+
[mediaItemId, [baseUrl, width, height, ...], uploadTimestampMs, ...]."""
|
|
150
|
+
if not isinstance(node, list):
|
|
151
|
+
return
|
|
152
|
+
if (len(node) >= 3 and isinstance(node[0], str)
|
|
153
|
+
and node[0].startswith("AF1Qip")
|
|
154
|
+
and isinstance(node[1], list) and node[1]
|
|
155
|
+
and isinstance(node[1][0], str)
|
|
156
|
+
and node[1][0].startswith("http")):
|
|
157
|
+
base = node[1]
|
|
158
|
+
items.append({
|
|
159
|
+
"id": node[0],
|
|
160
|
+
"base_url": base[0],
|
|
161
|
+
"width": base[1] if len(base) > 1 else None,
|
|
162
|
+
"height": base[2] if len(base) > 2 else None,
|
|
163
|
+
"ts_ms": node[2] if isinstance(node[2], (int, float)) else None,
|
|
164
|
+
})
|
|
165
|
+
return
|
|
166
|
+
for child in node:
|
|
167
|
+
_walk_items(child, items)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _dedupe(items: list[dict]) -> list[dict]:
|
|
171
|
+
seen: set[str] = set()
|
|
172
|
+
out: list[dict] = []
|
|
173
|
+
for item in items:
|
|
174
|
+
if item["id"] not in seen:
|
|
175
|
+
seen.add(item["id"])
|
|
176
|
+
out.append(item)
|
|
177
|
+
return out
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def parse_af_initdata(html: str) -> list[dict]:
|
|
181
|
+
"""Parse a shared-album page's HTML into a deduped list of media items."""
|
|
182
|
+
data = extract_ds1_data(html)
|
|
183
|
+
items: list[dict] = []
|
|
184
|
+
_walk_items(data, items)
|
|
185
|
+
return _dedupe(items)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
@dataclass
|
|
189
|
+
class SnackcPage:
|
|
190
|
+
"""One snAcKc page: walked items + the freshest continuation token.
|
|
191
|
+
|
|
192
|
+
`continuation_token is None` means the album is exhausted — no further
|
|
193
|
+
snAcKc call should be issued (live-proven exhaustion signal).
|
|
194
|
+
"""
|
|
195
|
+
|
|
196
|
+
items: list[dict]
|
|
197
|
+
continuation_token: str | None
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def parse_snackc_payload(payload_str: str | None) -> SnackcPage:
|
|
201
|
+
"""Parse one snAcKc `wrb.fr` inner payload string (the wire's JSON-in-JSON).
|
|
202
|
+
|
|
203
|
+
Runs the same item walker as the share page (the payload's item shape is
|
|
204
|
+
identical) and pulls the continuation cursor: the LAST AH_ token in the
|
|
205
|
+
payload is the freshest one (live-proven). No token = album exhausted.
|
|
206
|
+
"""
|
|
207
|
+
if not payload_str:
|
|
208
|
+
raise ProbeParseError(
|
|
209
|
+
"snAcKc entry carries no inner payload (null) — malformed envelope; "
|
|
210
|
+
"refusing to emit a partial listing"
|
|
211
|
+
)
|
|
212
|
+
try:
|
|
213
|
+
inner = json.loads(payload_str)
|
|
214
|
+
except json.JSONDecodeError as exc:
|
|
215
|
+
raise ProbeParseError(
|
|
216
|
+
f"snAcKc inner payload is not JSON (error at position {exc.pos}) — "
|
|
217
|
+
f"refusing to guess"
|
|
218
|
+
) from exc
|
|
219
|
+
items: list[dict] = []
|
|
220
|
+
_walk_items(inner, items)
|
|
221
|
+
tokens = AH_TOKEN_RE.findall(payload_str)
|
|
222
|
+
return SnackcPage(items=_dedupe(items), continuation_token=tokens[-1] if tokens else None)
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
@dataclass
|
|
226
|
+
class BatchexecuteEntry:
|
|
227
|
+
"""One `wrb.fr` entry of a batchexecute response body."""
|
|
228
|
+
|
|
229
|
+
rpcid: str | None
|
|
230
|
+
payload: str | None
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def parse_batchexecute(text: str) -> list[BatchexecuteEntry]:
|
|
234
|
+
"""Split a batchexecute response body into its `wrb.fr` entries.
|
|
235
|
+
|
|
236
|
+
The body is `)]}'`-prefixed, then one JSON array per line. Non-JSON lines
|
|
237
|
+
are skipped (the wire carries rpcids other than the requested one, e.g.
|
|
238
|
+
`di` heartbeats); only entries shaped ["wrb.fr", rpcid, payload, ...] are
|
|
239
|
+
returned. A body with ZERO parseable lines raises — a truncated response
|
|
240
|
+
must fail loud, not read as an empty album (T-17-02).
|
|
241
|
+
"""
|
|
242
|
+
entries: list[BatchexecuteEntry] = []
|
|
243
|
+
saw_json_line = False
|
|
244
|
+
for line in text.split("\n"):
|
|
245
|
+
line = line.strip()
|
|
246
|
+
if not line or line.startswith(")]}'"):
|
|
247
|
+
continue
|
|
248
|
+
try:
|
|
249
|
+
arr = json.loads(line)
|
|
250
|
+
except json.JSONDecodeError:
|
|
251
|
+
continue
|
|
252
|
+
saw_json_line = True
|
|
253
|
+
if not isinstance(arr, list):
|
|
254
|
+
continue
|
|
255
|
+
for entry in arr:
|
|
256
|
+
if isinstance(entry, list) and entry and entry[0] == "wrb.fr":
|
|
257
|
+
entries.append(BatchexecuteEntry(
|
|
258
|
+
rpcid=entry[1] if len(entry) > 1 else None,
|
|
259
|
+
payload=entry[2] if len(entry) > 2 else None,
|
|
260
|
+
))
|
|
261
|
+
if not saw_json_line:
|
|
262
|
+
raise ProbeParseError(
|
|
263
|
+
"batchexecute response carries no JSON lines after the )]}\\' prefix "
|
|
264
|
+
"— truncated or reshaped envelope; refusing to guess"
|
|
265
|
+
)
|
|
266
|
+
return entries
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
@dataclass
|
|
270
|
+
class AlbumSummary:
|
|
271
|
+
"""One shared album, as surfaced by photos.google.com/albums' ds:5 block
|
|
272
|
+
(live-proven row shape; see parse_album_summaries).
|
|
273
|
+
|
|
274
|
+
`item_count` is the album's METADATA count — it can exceed the photo
|
|
275
|
+
count the media walker returns when the album carries videos (live:
|
|
276
|
+
1096 vs 1094 photos). The authoritative PHOTO count is enumerate_album's.
|
|
277
|
+
"""
|
|
278
|
+
|
|
279
|
+
album_id: str | None
|
|
280
|
+
title: str | None
|
|
281
|
+
share_url: str | None = None
|
|
282
|
+
item_count: int | None = None
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def _walk_album_summaries(node, out):
|
|
286
|
+
"""Depth-first walk of the /albums page's ds:5 payload collecting
|
|
287
|
+
shared-album entries (live-proven 2026-09-28 row shape):
|
|
288
|
+
|
|
289
|
+
[album_cover_id, [cover_url, w, h, …], …, …, {<key>: ENTRY}]
|
|
290
|
+
|
|
291
|
+
ENTRY = [4, <title:str>, [dates…], <item_count:int>, 1,
|
|
292
|
+
<page_key_b64:str>, …, <share_token AF1Qip…:str>, …]
|
|
293
|
+
|
|
294
|
+
The album's identity token is the SHARE token (the /share/<id> path
|
|
295
|
+
segment); the base64-decoded field 5 is the share URL's ?key= page_key
|
|
296
|
+
(both live-proven by enumerating through the constructed URL).
|
|
297
|
+
"""
|
|
298
|
+
if not isinstance(node, list):
|
|
299
|
+
return
|
|
300
|
+
if node and isinstance(node[0], str) and node[0].startswith("AF1Qip"):
|
|
301
|
+
for field_ in node:
|
|
302
|
+
if not isinstance(field_, dict):
|
|
303
|
+
continue
|
|
304
|
+
for entry in field_.values():
|
|
305
|
+
if (isinstance(entry, list) and len(entry) >= 9
|
|
306
|
+
and isinstance(entry[1], str)
|
|
307
|
+
and isinstance(entry[3], int)
|
|
308
|
+
and isinstance(entry[5], str)
|
|
309
|
+
and isinstance(entry[8], str)
|
|
310
|
+
and entry[8].startswith("AF1Qip")):
|
|
311
|
+
out.append(AlbumSummary(
|
|
312
|
+
album_id=entry[8],
|
|
313
|
+
title=entry[1],
|
|
314
|
+
item_count=entry[3],
|
|
315
|
+
share_url=(f"https://photos.google.com/share/{entry[8]}"
|
|
316
|
+
f"?key={_decode_page_key(entry[5])}"),
|
|
317
|
+
))
|
|
318
|
+
return
|
|
319
|
+
for child in node:
|
|
320
|
+
_walk_album_summaries(child, out)
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def _decode_page_key(b64: str) -> str:
|
|
324
|
+
"""Decode the ds:5 entry's base64 share key into the ?key= page_key
|
|
325
|
+
(live-proven: `UX20fm…` base64 == the ?key= the constructed URL used)."""
|
|
326
|
+
import base64
|
|
327
|
+
|
|
328
|
+
try:
|
|
329
|
+
return base64.b64decode(b64).decode("ascii", "replace")
|
|
330
|
+
except Exception:
|
|
331
|
+
return ""
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def parse_album_summaries(ds5_data: list) -> list[AlbumSummary]:
|
|
335
|
+
"""Parse the /albums page's ds:5 payload into deduped AlbumSummary rows
|
|
336
|
+
(title + share token + page_key + metadata item count)."""
|
|
337
|
+
out: list[AlbumSummary] = []
|
|
338
|
+
_walk_album_summaries(ds5_data, out)
|
|
339
|
+
seen: set[str] = set()
|
|
340
|
+
deduped = []
|
|
341
|
+
for s in out:
|
|
342
|
+
if s.album_id and s.album_id not in seen:
|
|
343
|
+
seen.add(s.album_id)
|
|
344
|
+
deduped.append(s)
|
|
345
|
+
return deduped
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
"""Redaction helpers for every Google-facing print/log surface (T-17-03).
|
|
2
|
+
|
|
3
|
+
Migrated from probes/common.py. Capability URLs and AH_ cursors are
|
|
4
|
+
secret-like: they grant album/read access to anyone holding them. Every
|
|
5
|
+
downstream print site uses these helpers — the full shape is never echoed
|
|
6
|
+
to stdout, never written to any committed file.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
|
|
12
|
+
# A full share-link/album token: AF1Qip followed by the ~48-char id portion.
|
|
13
|
+
_FULL_TOKEN_RE = re.compile(r"AF1Qip[A-Za-z0-9_-]{40,}")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def redact_link(url: str | None) -> str:
|
|
17
|
+
"""Return the truncated capability-URL shape for any URL string.
|
|
18
|
+
|
|
19
|
+
`https://photos.google.com/share/AF1QipXXXXXXXX...YYYY` becomes
|
|
20
|
+
`photos.google.com/share/AF1Qip…YYYY` — enough to correlate two links
|
|
21
|
+
as different, never enough to resolve either. Idempotent: a string with
|
|
22
|
+
no full token passes through with the scheme stripped.
|
|
23
|
+
"""
|
|
24
|
+
if not url:
|
|
25
|
+
return "(none)"
|
|
26
|
+
text = _FULL_TOKEN_RE.sub(lambda m: f"AF1Qip…{m.group(0)[-4:]}", url)
|
|
27
|
+
return text.replace("https://", "")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def redact_tokens(text: str) -> str:
|
|
31
|
+
"""Redact every full capability token embedded in arbitrary text (e.g. a
|
|
32
|
+
raw error body we are about to print or record)."""
|
|
33
|
+
return _FULL_TOKEN_RE.sub(lambda m: f"AF1Qip…{m.group(0)[-4:]}", text)
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""Cookie vault for harvested Google sessions (migrated from
|
|
2
|
+
probes/cookie_vault.py, phase 17 plan 17-01 — D-03/D-06).
|
|
3
|
+
|
|
4
|
+
Harvested Google session cookies are wide-scope secrets. This module is the
|
|
5
|
+
ONLY reader of the vault, and the boundary is structural, not conventional:
|
|
6
|
+
`load()` inspects the caller's module name and refuses any sync/apply path
|
|
7
|
+
outright (D-06's denylist), the file lives outside the repo (refusing
|
|
8
|
+
repo-inside paths at save time), and permissions are 0600 via os.open.
|
|
9
|
+
|
|
10
|
+
Soft migration (17-CONTEXT discretion): the default path is now the
|
|
11
|
+
production location `~/.config/pushframe/google-cookies.json`; when it is
|
|
12
|
+
absent, `load()` falls back to the legacy probe vault
|
|
13
|
+
`~/.config/pushframe/probes/google-cookies.json` so an operator's existing
|
|
14
|
+
session keeps working. The boundary travels with the module: the denylist,
|
|
15
|
+
the 0600 mode and the repo-inside refusal are carried over verbatim.
|
|
16
|
+
|
|
17
|
+
NOTE: the denylist includes `pushframe.cli` — the CLI must reach the vault
|
|
18
|
+
only through `pushframe.google` (GoogleSession.from_vault), never by
|
|
19
|
+
importing this module directly. That routing is the boundary's enforcement
|
|
20
|
+
point, not a formality.
|
|
21
|
+
"""
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import json
|
|
25
|
+
import os
|
|
26
|
+
import sys
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
|
|
29
|
+
DEFAULT_VAULT_PATH = Path("~/.config/pushframe/google-cookies.json")
|
|
30
|
+
LEGACY_VAULT_PATH = Path("~/.config/pushframe/probes/google-cookies.json")
|
|
31
|
+
|
|
32
|
+
# D-06: sync/apply code paths can never hold session cookies. Enforced at
|
|
33
|
+
# load() via frame inspection — an import from these namespaces raises
|
|
34
|
+
# before any file is read.
|
|
35
|
+
_DENYLIST_PREFIXES = ("pushframe.sync", "pushframe.reconcile", "pushframe.cli")
|
|
36
|
+
|
|
37
|
+
_REPO_ROOT = Path(__file__).resolve().parents[2]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class CookieVaultError(RuntimeError):
|
|
41
|
+
"""Vault absent, mis-located, or reached from a denied code path."""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _deny_check(caller_module_name: str) -> None:
|
|
45
|
+
for prefix in _DENYLIST_PREFIXES:
|
|
46
|
+
if caller_module_name.startswith(prefix):
|
|
47
|
+
raise CookieVaultError(
|
|
48
|
+
f"cookie vault refused: module '{caller_module_name}' matches the "
|
|
49
|
+
f"sync/apply denylist prefix '{prefix}' — sync paths can never read "
|
|
50
|
+
f"harvested session cookies (D-06)"
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _caller_module_name() -> str:
|
|
55
|
+
frame = sys._getframe(2) # load() <- caller
|
|
56
|
+
return frame.f_globals.get("__name__", "") if frame else ""
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _resolve(path: str | Path | None) -> Path:
|
|
60
|
+
return Path(path).expanduser().resolve() if path else DEFAULT_VAULT_PATH.expanduser().resolve()
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _vault_candidates(path: str | Path | None) -> list[Path]:
|
|
64
|
+
"""Vault paths to try, in order: explicit > production > legacy probe vault.
|
|
65
|
+
|
|
66
|
+
The soft migration (17-CONTEXT): an operator whose session still lives in
|
|
67
|
+
the legacy probe vault keeps working until the next `google-link` writes
|
|
68
|
+
the production vault.
|
|
69
|
+
"""
|
|
70
|
+
if path is not None:
|
|
71
|
+
return [_resolve(path)]
|
|
72
|
+
prod = DEFAULT_VAULT_PATH.expanduser().resolve()
|
|
73
|
+
if prod.exists():
|
|
74
|
+
return [prod]
|
|
75
|
+
legacy = LEGACY_VAULT_PATH.expanduser().resolve()
|
|
76
|
+
if legacy.exists():
|
|
77
|
+
return [legacy]
|
|
78
|
+
return [prod]
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def save(cookies: list[dict], *, path: str | Path | None = None) -> Path:
|
|
82
|
+
"""Persist cookie records atomically with 0600 permissions.
|
|
83
|
+
|
|
84
|
+
Refuses to write when the resolved path is inside the git repo — the vault
|
|
85
|
+
lives outside it by construction (T-16-05). Defaults to the production
|
|
86
|
+
vault path; `google-link` re-links always write there.
|
|
87
|
+
"""
|
|
88
|
+
vault = _resolve(path)
|
|
89
|
+
if vault.is_relative_to(_REPO_ROOT):
|
|
90
|
+
raise CookieVaultError(
|
|
91
|
+
f"cookie vault refused: resolved path {vault} is inside the git repo — "
|
|
92
|
+
f"harvested cookies are never tracked (D-06)"
|
|
93
|
+
)
|
|
94
|
+
vault.parent.mkdir(parents=True, exist_ok=True)
|
|
95
|
+
payload = json.dumps(cookies, indent=2).encode("utf-8")
|
|
96
|
+
fd = os.open(vault, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
|
|
97
|
+
try:
|
|
98
|
+
os.write(fd, payload)
|
|
99
|
+
finally:
|
|
100
|
+
os.close(fd)
|
|
101
|
+
os.chmod(vault, 0o600) # in case the file pre-existed with looser mode
|
|
102
|
+
return vault
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def load(*, path: str | Path | None = None) -> list[dict]:
|
|
106
|
+
"""Read the vault, refusing sync/apply-named callers structurally."""
|
|
107
|
+
_deny_check(_caller_module_name())
|
|
108
|
+
vault = _resolve(_vault_candidates(path)[0])
|
|
109
|
+
if not vault.exists():
|
|
110
|
+
raise CookieVaultError(
|
|
111
|
+
f"no session — run the bootstrap first (vault not found at {vault})"
|
|
112
|
+
)
|
|
113
|
+
records = json.loads(vault.read_text())
|
|
114
|
+
if not isinstance(records, list) or not records:
|
|
115
|
+
raise CookieVaultError(f"vault at {vault} is empty or malformed")
|
|
116
|
+
return records
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def cookies_for_httpx(*, path: str | Path | None = None) -> dict[str, str]:
|
|
120
|
+
"""Return an httpx.Cookies-shaped name→value dict from the vault.
|
|
121
|
+
|
|
122
|
+
FLATTENED — diagnostics only. The session client must build its jar from
|
|
123
|
+
the full `load()` records instead: photos.google.com treats a flattened
|
|
124
|
+
dict as an anonymous visitor (live-proven phase 16).
|
|
125
|
+
"""
|
|
126
|
+
return {c["name"]: c["value"] for c in load(path=path) if "name" in c and "value" in c}
|