@hraness/message-like-me 0.8.0 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,467 +0,0 @@
1
- #!/usr/bin/env python3
2
- """Prepare a bounded Ensoul source packet from account-authored X posts.
3
-
4
- Only allowlisted public-post members are opened. Direct messages, address books,
5
- advertising data, media, and every other archive member remain unread.
6
- """
7
-
8
- from __future__ import annotations
9
-
10
- import argparse
11
- import datetime as dt
12
- import email.utils
13
- import hashlib
14
- import json
15
- import os
16
- from pathlib import Path, PurePosixPath
17
- import re
18
- import stat
19
- import sys
20
- import uuid
21
- import zipfile
22
-
23
-
24
- SCHEMA_VERSION = "ensoul.source-packet.v1"
25
- PAYLOAD_SCHEMA = "ensoul.x-authored-posts-source.v1"
26
- MAX_ARCHIVE_MEMBERS = 100_000
27
- MAX_SELECTED_MEMBER_BYTES = 256 * 1024 * 1024
28
- MAX_SELECTED_TOTAL_BYTES = 512 * 1024 * 1024
29
- MAX_COMPRESSION_RATIO = 1_000
30
- MAX_POSTS = 2_000
31
- MAX_TEXT_CHARS = 50_000
32
- MAX_RECORD_CONTENT_BYTES = 32 * 1024
33
- MAX_TOTAL_CONTENT_BYTES = MAX_POSTS * MAX_RECORD_CONTENT_BYTES
34
- MAX_PACKET_BYTES = 128 * 1024 * 1024
35
- TWEET_MEMBER = re.compile(r"(?:^|/)data/tweets(?:-part\d+)?\.js$")
36
- POST_ID = re.compile(r"^[0-9]{1,20}$", re.ASCII)
37
- JS_PREFIX = re.compile(
38
- r"^\s*(?:window\.)?YTD\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\s*=\s*",
39
- re.ASCII,
40
- )
41
-
42
-
43
- class ArchiveError(ValueError):
44
- pass
45
-
46
-
47
- def canonical_bytes(value: object) -> bytes:
48
- """Encode the packet's JSON subset according to RFC 8785 JCS."""
49
-
50
- def encode(item: object) -> str:
51
- if item is None:
52
- return "null"
53
- if item is True:
54
- return "true"
55
- if item is False:
56
- return "false"
57
- if isinstance(item, int):
58
- if abs(item) > 9_007_199_254_740_991:
59
- raise ArchiveError("integer exceeds the interoperable JSON range")
60
- return str(item)
61
- if isinstance(item, float):
62
- raise ArchiveError("floating-point values are not supported in source packets")
63
- if isinstance(item, str):
64
- if any(0xD800 <= ord(character) <= 0xDFFF for character in item):
65
- raise ArchiveError("unpaired Unicode surrogate in source packet")
66
- return json.dumps(item, ensure_ascii=False, separators=(",", ":"))
67
- if isinstance(item, list):
68
- return "[" + ",".join(encode(member) for member in item) + "]"
69
- if isinstance(item, dict):
70
- if not all(isinstance(key, str) for key in item):
71
- raise ArchiveError("source packet object keys must be strings")
72
- keys = sorted(item, key=lambda key: key.encode("utf-16be"))
73
- return "{" + ",".join(f"{encode(key)}:{encode(item[key])}" for key in keys) + "}"
74
- raise ArchiveError(f"unsupported source packet value: {type(item).__name__}")
75
-
76
- return encode(value).encode("utf-8")
77
-
78
-
79
- def sha256_hex(value: bytes) -> str:
80
- return hashlib.sha256(value).hexdigest()
81
-
82
-
83
- def parse_bound(value: str | None, flag: str) -> dt.datetime | None:
84
- if value is None:
85
- return None
86
- try:
87
- parsed = dt.datetime.fromisoformat(value.replace("Z", "+00:00"))
88
- except ValueError as exc:
89
- raise ArchiveError(f"{flag} must be an ISO-8601 timestamp") from exc
90
- if parsed.tzinfo is None:
91
- raise ArchiveError(f"{flag} must include a timezone")
92
- return parsed.astimezone(dt.timezone.utc)
93
-
94
-
95
- def parse_created_at(value: object) -> str | None:
96
- if not isinstance(value, str) or not value.strip():
97
- return None
98
- try:
99
- parsed = email.utils.parsedate_to_datetime(value)
100
- except (TypeError, ValueError):
101
- try:
102
- parsed = dt.datetime.fromisoformat(value.replace("Z", "+00:00"))
103
- except ValueError:
104
- return None
105
- if parsed.tzinfo is None:
106
- parsed = parsed.replace(tzinfo=dt.timezone.utc)
107
- return parsed.astimezone(dt.timezone.utc).isoformat().replace("+00:00", "Z")
108
-
109
-
110
- def validate_member(info: zipfile.ZipInfo) -> None:
111
- name = info.filename
112
- if "\\" in name or "\x00" in name:
113
- raise ArchiveError("archive contains an unsafe member name")
114
- path = PurePosixPath(name)
115
- if path.is_absolute() or ".." in path.parts:
116
- raise ArchiveError("archive contains a path-traversal member")
117
- unix_type = (info.external_attr >> 16) & 0o170000
118
- if unix_type == stat.S_IFLNK:
119
- raise ArchiveError("archive contains a symbolic-link member")
120
-
121
-
122
- def selected_members(archive: zipfile.ZipFile) -> list[zipfile.ZipInfo]:
123
- infos = archive.infolist()
124
- if len(infos) > MAX_ARCHIVE_MEMBERS:
125
- raise ArchiveError("archive has too many members")
126
- selected: list[zipfile.ZipInfo] = []
127
- names: set[str] = set()
128
- total = 0
129
- for info in infos:
130
- validate_member(info)
131
- if not TWEET_MEMBER.search(info.filename):
132
- continue
133
- if info.filename in names:
134
- raise ArchiveError("archive contains a duplicate posts member")
135
- names.add(info.filename)
136
- if info.file_size > MAX_SELECTED_MEMBER_BYTES:
137
- raise ArchiveError("posts member exceeds the safety limit")
138
- ratio = info.file_size / max(info.compress_size, 1)
139
- if ratio > MAX_COMPRESSION_RATIO:
140
- raise ArchiveError("posts member has an unsafe compression ratio")
141
- total += info.file_size
142
- if total > MAX_SELECTED_TOTAL_BYTES:
143
- raise ArchiveError("combined posts members exceed the safety limit")
144
- selected.append(info)
145
- if not selected:
146
- raise ArchiveError("archive contains no supported data/tweets*.js member")
147
- return sorted(selected, key=lambda item: item.filename)
148
-
149
-
150
- def read_bounded(archive: zipfile.ZipFile, info: zipfile.ZipInfo) -> bytes:
151
- with archive.open(info, "r") as member:
152
- data = member.read(info.file_size + 1)
153
- if len(data) != info.file_size:
154
- raise ArchiveError("posts member size changed while reading")
155
- return data
156
-
157
-
158
- def parse_js_array(data: bytes, member_name: str) -> list[object]:
159
- try:
160
- text = data.decode("utf-8-sig")
161
- except UnicodeDecodeError as exc:
162
- raise ArchiveError(f"{member_name} is not UTF-8") from exc
163
- prefix = JS_PREFIX.match(text)
164
- if prefix is None:
165
- raise ArchiveError(f"{member_name} has an unsupported JavaScript wrapper")
166
- payload = text[prefix.end() :].strip()
167
- if payload.endswith(";"):
168
- payload = payload[:-1].rstrip()
169
- try:
170
- value = json.loads(payload)
171
- except json.JSONDecodeError as exc:
172
- raise ArchiveError(f"{member_name} does not contain valid JSON") from exc
173
- if not isinstance(value, list):
174
- raise ArchiveError(f"{member_name} must contain an array")
175
- return value
176
-
177
-
178
- def classify_post(tweet: dict[str, object], text: str) -> tuple[str, str]:
179
- if text.startswith("RT @"):
180
- return "repost", "mixed"
181
- if tweet.get("in_reply_to_status_id") or tweet.get("in_reply_to_status_id_str"):
182
- return "reply", "subject"
183
- if tweet.get("quoted_status_id") or tweet.get("quoted_status_id_str"):
184
- return "quote_post", "subject"
185
- return "post", "subject"
186
-
187
-
188
- def bounded_content(text: str) -> dict[str, object]:
189
- candidate = text[:MAX_TEXT_CHARS]
190
- truncated = len(candidate) != len(text)
191
-
192
- def content(prefix: str, was_truncated: bool) -> dict[str, object]:
193
- return {"text": prefix, "truncated": was_truncated}
194
-
195
- if len(canonical_bytes(content(candidate, truncated))) <= MAX_RECORD_CONTENT_BYTES:
196
- return content(candidate, truncated)
197
- low = 0
198
- high = len(candidate)
199
- while low < high:
200
- middle = (low + high + 1) // 2
201
- if len(canonical_bytes(content(candidate[:middle], True))) <= MAX_RECORD_CONTENT_BYTES:
202
- low = middle
203
- else:
204
- high = middle - 1
205
- return content(candidate[:low], True)
206
-
207
-
208
- def post_record(raw: object, member_name: str) -> dict[str, object] | None:
209
- if not isinstance(raw, dict):
210
- return None
211
- candidate = raw.get("tweet", raw)
212
- if not isinstance(candidate, dict):
213
- return None
214
- post_id = candidate.get("id_str", candidate.get("id"))
215
- text = candidate.get("full_text", candidate.get("text"))
216
- if not isinstance(post_id, (str, int)) or not isinstance(text, str):
217
- return None
218
- post_id_text = str(post_id).strip()
219
- if POST_ID.fullmatch(post_id_text) is None or not text.strip():
220
- return None
221
- content = bounded_content(text)
222
- bounded_text = str(content["text"])
223
- kind, author_role = classify_post(candidate, bounded_text)
224
- occurred_at = parse_created_at(candidate.get("created_at"))
225
- if occurred_at is None:
226
- return None
227
- content_hash = sha256_hex(canonical_bytes(content))
228
- semantic: dict[str, object] = {
229
- "id": f"x:{post_id_text}",
230
- "kind": kind,
231
- "authorRole": author_role,
232
- "contentRole": "forwarded" if kind == "repost" else "original",
233
- "authorshipConfidence": "strong",
234
- "sentStatus": "published",
235
- "visibility": "public",
236
- "sourceClass": "polished_self_presentation",
237
- "content": content,
238
- "provenance": {
239
- "provider": "x-archive",
240
- "operation": "account-authored-public-post-export",
241
- "sourceId": f"x-post:{post_id_text}",
242
- "policyVersion": PAYLOAD_SCHEMA,
243
- "contentSha256": content_hash,
244
- },
245
- }
246
- semantic["occurredAt"] = occurred_at
247
- semantic["digest"] = "sha256:" + sha256_hex(canonical_bytes(semantic))
248
- semantic["_member"] = member_name
249
- return semantic
250
-
251
-
252
- def choose_evenly(records: list[dict[str, object]], limit: int) -> list[dict[str, object]]:
253
- if len(records) <= limit:
254
- return records
255
- if limit == 1:
256
- return [records[-1]]
257
- indexes = [round(index * (len(records) - 1) / (limit - 1)) for index in range(limit)]
258
- return [records[index] for index in indexes]
259
-
260
-
261
- def build_packet(
262
- archive_path: Path,
263
- *,
264
- limit: int,
265
- after: dt.datetime | None,
266
- before: dt.datetime | None,
267
- ) -> tuple[dict[str, object], dict[str, object]]:
268
- selected_hash = hashlib.sha256()
269
- records: list[dict[str, object]] = []
270
- malformed = 0
271
- duplicate_ids = 0
272
- input_records = 0
273
- selected_member_count = 0
274
- with zipfile.ZipFile(archive_path, "r") as archive:
275
- members = selected_members(archive)
276
- selected_member_count = len(members)
277
- for info in members:
278
- data = read_bounded(archive, info)
279
- selected_hash.update(info.filename.encode("utf-8"))
280
- selected_hash.update(b"\x00")
281
- selected_hash.update(data)
282
- raw_records = parse_js_array(data, info.filename)
283
- input_records += len(raw_records)
284
- for raw in raw_records:
285
- record = post_record(raw, info.filename)
286
- if record is None:
287
- malformed += 1
288
- continue
289
- records.append(record)
290
-
291
- by_id: dict[str, dict[str, object]] = {}
292
- for record in records:
293
- record_id = str(record["id"])
294
- if record_id in by_id:
295
- previous = dict(by_id[record_id])
296
- current = dict(record)
297
- previous.pop("_member", None)
298
- current.pop("_member", None)
299
- if previous != current:
300
- raise ArchiveError("archive contains conflicting records for one post ID")
301
- duplicate_ids += 1
302
- continue
303
- by_id[record_id] = record
304
- records = list(by_id.values())
305
- records.sort(key=lambda value: (str(value.get("occurredAt", "")), str(value["id"])))
306
-
307
- def in_bounds(record: dict[str, object]) -> bool:
308
- value = record.get("occurredAt")
309
- if not isinstance(value, str):
310
- return after is None and before is None
311
- timestamp = dt.datetime.fromisoformat(value.replace("Z", "+00:00"))
312
- if after is not None and timestamp < after:
313
- return False
314
- if before is not None and timestamp >= before:
315
- return False
316
- return True
317
-
318
- records = [record for record in records if in_bounds(record)]
319
- eligible_count = len(records)
320
- records = choose_evenly(records, limit)
321
- for record in records:
322
- record.pop("_member", None)
323
-
324
- content_bytes = sum(len(canonical_bytes(record["content"])) for record in records)
325
- if content_bytes > MAX_TOTAL_CONTENT_BYTES:
326
- raise ArchiveError("selected post content exceeds the aggregate safety limit")
327
-
328
- revision = selected_hash.hexdigest()
329
- generated_at = dt.datetime.now(dt.timezone.utc).isoformat().replace("+00:00", "Z")
330
- latest = max((str(record.get("occurredAt")) for record in records if record.get("occurredAt")), default=None)
331
- if latest is not None and dt.datetime.fromisoformat(latest.replace("Z", "+00:00")) > dt.datetime.fromisoformat(
332
- generated_at.replace("Z", "+00:00")
333
- ):
334
- raise ArchiveError("selected posts contain an occurrence time later than packet generation")
335
- limits: dict[str, object] = {
336
- "maxRecords": limit,
337
- "eligibleRecords": eligible_count,
338
- "selection": "chronological-even-sample",
339
- "afterInclusive": after.isoformat().replace("+00:00", "Z") if after else None,
340
- "beforeExclusive": before.isoformat().replace("+00:00", "Z") if before else None,
341
- "contentBytes": content_bytes,
342
- "maxContentBytes": MAX_TOTAL_CONTENT_BYTES,
343
- "selectedMembers": selected_member_count,
344
- "inputRecords": input_records,
345
- "malformedRecordsSkipped": malformed,
346
- "exactDuplicateRecordsSkipped": duplicate_ids,
347
- }
348
- packet: dict[str, object] = {
349
- "schemaVersion": SCHEMA_VERSION,
350
- "digestCanonicalization": "JCS-RFC8785",
351
- "packetId": f"ensoul_x_{revision[:24]}",
352
- "generatedAt": generated_at,
353
- "subject": {
354
- "localId": f"x-archive-owner:{revision[:16]}",
355
- "kind": "owner",
356
- "identityBasis": "User-authorized official account archive; the user must confirm archive ownership.",
357
- },
358
- "scope": {
359
- "adapter": "x-archive",
360
- "payloadSchema": PAYLOAD_SCHEMA,
361
- "completeness": (
362
- "sampled" if eligible_count > limit
363
- else "bounded" if malformed > 0 or duplicate_ids > 0
364
- else "complete"
365
- ),
366
- "sourceRevision": "sha256:" + revision,
367
- "limits": limits,
368
- },
369
- "records": records,
370
- "limitations": [
371
- "Only allowlisted data/tweets*.js members were opened; direct messages, address books, advertising data, media, deleted posts, and community posts were not accessed.",
372
- "Archive membership supports account authorship but does not prove that every embedded or quoted phrase was written by the subject; reposts are marked mixed.",
373
- "Public visibility describes the original post context and is not permission to republish content that may since have been deleted or restricted.",
374
- "Selection is bounded and chronological; counts do not measure importance, motive, or stable personality.",
375
- ],
376
- }
377
- if latest:
378
- packet["scope"]["sourceCutoff"] = latest # type: ignore[index]
379
- packet["packetDigest"] = "sha256:" + sha256_hex(canonical_bytes(packet))
380
- receipt = {
381
- "schemaVersion": SCHEMA_VERSION,
382
- "packetDigest": packet["packetDigest"],
383
- "records": len(records),
384
- "eligibleRecords": eligible_count,
385
- "malformedRecordsSkipped": malformed,
386
- "duplicateIdsSkipped": duplicate_ids,
387
- "contentBytes": content_bytes,
388
- "maxContentBytes": MAX_TOTAL_CONTENT_BYTES,
389
- "selectedMembers": selected_member_count,
390
- "inputRecords": input_records,
391
- "exactDuplicateRecordsSkipped": duplicate_ids,
392
- "sourceRevision": packet["scope"]["sourceRevision"], # type: ignore[index]
393
- }
394
- return packet, receipt
395
-
396
-
397
- def write_private_atomic(path: Path, data: bytes) -> None:
398
- if not path.is_absolute():
399
- raise ArchiveError("--output must be an absolute path")
400
- parent = path.parent
401
- resolved_parent = parent.resolve(strict=True)
402
- if resolved_parent != parent:
403
- raise ArchiveError("--output parent must not contain symbolic links")
404
- if path.exists() or path.is_symlink():
405
- raise ArchiveError("--output already exists")
406
- temp = parent / f".{path.name}.tmp-{uuid.uuid4().hex}"
407
- flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL
408
- if hasattr(os, "O_NOFOLLOW"):
409
- flags |= os.O_NOFOLLOW
410
- descriptor = os.open(temp, flags, 0o600)
411
- try:
412
- with os.fdopen(descriptor, "wb", closefd=True) as stream:
413
- stream.write(data)
414
- stream.flush()
415
- os.fsync(stream.fileno())
416
- os.chmod(temp, 0o600)
417
- os.link(temp, path, follow_symlinks=False)
418
- directory = os.open(parent, os.O_RDONLY)
419
- try:
420
- os.fsync(directory)
421
- finally:
422
- os.close(directory)
423
- finally:
424
- try:
425
- temp.unlink()
426
- except FileNotFoundError:
427
- pass
428
-
429
-
430
- def parse_args(argv: list[str]) -> argparse.Namespace:
431
- parser = argparse.ArgumentParser(description=__doc__)
432
- parser.add_argument("archive", type=Path, help="absolute path to an official X archive ZIP")
433
- parser.add_argument("--output", required=True, type=Path, help="new absolute private packet path")
434
- parser.add_argument("--limit", type=int, default=2000, help="maximum evenly sampled posts (default: 2000)")
435
- parser.add_argument("--after", help="inclusive ISO-8601 timestamp")
436
- parser.add_argument("--before", help="exclusive ISO-8601 timestamp")
437
- return parser.parse_args(argv)
438
-
439
-
440
- def main(argv: list[str] | None = None) -> int:
441
- args = parse_args(sys.argv[1:] if argv is None else argv)
442
- try:
443
- if not args.archive.is_absolute():
444
- raise ArchiveError("archive path must be absolute")
445
- if not args.archive.is_file():
446
- raise ArchiveError("archive path must be an existing file")
447
- if args.limit < 1 or args.limit > MAX_POSTS:
448
- raise ArchiveError(f"--limit must be between 1 and {MAX_POSTS}")
449
- after = parse_bound(args.after, "--after")
450
- before = parse_bound(args.before, "--before")
451
- if after and before and after >= before:
452
- raise ArchiveError("--after must be earlier than --before")
453
- packet, receipt = build_packet(args.archive, limit=args.limit, after=after, before=before)
454
- payload = json.dumps(packet, ensure_ascii=False, indent=2).encode("utf-8") + b"\n"
455
- if len(payload) > MAX_PACKET_BYTES:
456
- raise ArchiveError("prepared packet exceeds the 128 MiB output safety limit")
457
- write_private_atomic(args.output, payload)
458
- receipt["output"] = str(args.output)
459
- print(json.dumps(receipt, sort_keys=True, separators=(",", ":")))
460
- return 0
461
- except (ArchiveError, OSError, zipfile.BadZipFile) as exc:
462
- print(f"error: {exc}", file=sys.stderr)
463
- return 2
464
-
465
-
466
- if __name__ == "__main__":
467
- raise SystemExit(main())