@hraness/message-like-me 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +151 -0
- package/LICENSE +21 -0
- package/README.md +698 -0
- package/SECURITY.md +318 -0
- package/dist/agentic-messaging-v1.d.ts +179 -0
- package/dist/agentic-messaging-v1.js +52 -0
- package/dist/canonical-json.d.ts +3 -0
- package/dist/cli-bs3db5jr.js +643 -0
- package/dist/cli-d7qv38ab.js +485 -0
- package/dist/cli-kw20gkk3.js +5 -0
- package/dist/cli-qqafdvz9.js +5 -0
- package/dist/cli-ry4128kz.js +584 -0
- package/dist/cli-ththzwja.js +20 -0
- package/dist/cli-x1qncxm7.js +1078 -0
- package/dist/cli.js +9436 -0
- package/dist/ensoul-source-v1.d.ts +121 -0
- package/dist/ensoul-source-v1.js +24 -0
- package/dist/index.d.ts +3 -0
- package/dist/index.js +47 -0
- package/dist/message-bundle-v1-identity.d.ts +2 -0
- package/dist/message-bundle-v1.d.ts +205 -0
- package/dist/message-bundle-v1.js +37 -0
- package/dist/message-bundle-v2-identity.d.ts +2 -0
- package/dist/message-bundle-v2.d.ts +215 -0
- package/dist/message-bundle-v2.js +43 -0
- package/dist/metrics.d.ts +41 -0
- package/dist/types.d.ts +568 -0
- package/docs/local-message-bundle-v1.md +245 -0
- package/docs/local-message-bundle-v2.md +167 -0
- package/docs/methodology.md +322 -0
- package/docs/research.md +164 -0
- package/package.json +82 -0
- package/schema/ensoul-messages-source-v1.schema.json +248 -0
- package/schema/local-message-bundle-v1.schema.json +449 -0
- package/schema/local-message-bundle-v2.schema.json +462 -0
- package/schema/style-profile-v1.schema.json +223 -0
- package/schema/style-profile-v2.schema.json +202 -0
- package/skills/ensoul/LICENSE +23 -0
- package/skills/ensoul/NOTICE.md +7 -0
- package/skills/ensoul/SKILL.md +226 -0
- package/skills/ensoul/VENDORED_FROM.md +7 -0
- package/skills/ensoul/agents/openai.yaml +4 -0
- package/skills/ensoul/references/ensoul-source-packet-v1.schema.json +187 -0
- package/skills/ensoul/references/evidence-method.md +148 -0
- package/skills/ensoul/references/output-blueprint.md +143 -0
- package/skills/ensoul/references/source-packets.md +139 -0
- package/skills/ensoul/scripts/prepare_x_archive.py +467 -0
- package/skills/ensoul/scripts/validate_source_packet.py +477 -0
- package/skills/message-like-me/SKILL.md +229 -0
- package/skills/message-like-me/agents/openai.yaml +4 -0
- package/skills/message-like-me/references/analysis.md +159 -0
- package/skills/message-like-me/references/drafting.md +86 -0
- package/skills/message-like-me/references/ensoul.md +94 -0
- package/skills/message-like-me/references/evaluation.md +81 -0
- package/skills/message-like-me/references/privacy.md +106 -0
- package/skills/message-like-me/references/profile-schema.md +148 -0
|
@@ -0,0 +1,467 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Prepare a bounded Ensoul source packet from account-authored X posts.
|
|
3
|
+
|
|
4
|
+
Only allowlisted public-post members are opened. Direct messages, address books,
|
|
5
|
+
advertising data, media, and every other archive member remain unread.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import argparse
|
|
11
|
+
import datetime as dt
|
|
12
|
+
import email.utils
|
|
13
|
+
import hashlib
|
|
14
|
+
import json
|
|
15
|
+
import os
|
|
16
|
+
from pathlib import Path, PurePosixPath
|
|
17
|
+
import re
|
|
18
|
+
import stat
|
|
19
|
+
import sys
|
|
20
|
+
import uuid
|
|
21
|
+
import zipfile
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
SCHEMA_VERSION = "ensoul.source-packet.v1"
|
|
25
|
+
PAYLOAD_SCHEMA = "ensoul.x-authored-posts-source.v1"
|
|
26
|
+
MAX_ARCHIVE_MEMBERS = 100_000
|
|
27
|
+
MAX_SELECTED_MEMBER_BYTES = 256 * 1024 * 1024
|
|
28
|
+
MAX_SELECTED_TOTAL_BYTES = 512 * 1024 * 1024
|
|
29
|
+
MAX_COMPRESSION_RATIO = 1_000
|
|
30
|
+
MAX_POSTS = 2_000
|
|
31
|
+
MAX_TEXT_CHARS = 50_000
|
|
32
|
+
MAX_RECORD_CONTENT_BYTES = 32 * 1024
|
|
33
|
+
MAX_TOTAL_CONTENT_BYTES = MAX_POSTS * MAX_RECORD_CONTENT_BYTES
|
|
34
|
+
MAX_PACKET_BYTES = 128 * 1024 * 1024
|
|
35
|
+
TWEET_MEMBER = re.compile(r"(?:^|/)data/tweets(?:-part\d+)?\.js$")
|
|
36
|
+
POST_ID = re.compile(r"^[0-9]{1,20}$", re.ASCII)
|
|
37
|
+
JS_PREFIX = re.compile(
|
|
38
|
+
r"^\s*(?:window\.)?YTD\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\s*=\s*",
|
|
39
|
+
re.ASCII,
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class ArchiveError(ValueError):
|
|
44
|
+
pass
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def canonical_bytes(value: object) -> bytes:
|
|
48
|
+
"""Encode the packet's JSON subset according to RFC 8785 JCS."""
|
|
49
|
+
|
|
50
|
+
def encode(item: object) -> str:
|
|
51
|
+
if item is None:
|
|
52
|
+
return "null"
|
|
53
|
+
if item is True:
|
|
54
|
+
return "true"
|
|
55
|
+
if item is False:
|
|
56
|
+
return "false"
|
|
57
|
+
if isinstance(item, int):
|
|
58
|
+
if abs(item) > 9_007_199_254_740_991:
|
|
59
|
+
raise ArchiveError("integer exceeds the interoperable JSON range")
|
|
60
|
+
return str(item)
|
|
61
|
+
if isinstance(item, float):
|
|
62
|
+
raise ArchiveError("floating-point values are not supported in source packets")
|
|
63
|
+
if isinstance(item, str):
|
|
64
|
+
if any(0xD800 <= ord(character) <= 0xDFFF for character in item):
|
|
65
|
+
raise ArchiveError("unpaired Unicode surrogate in source packet")
|
|
66
|
+
return json.dumps(item, ensure_ascii=False, separators=(",", ":"))
|
|
67
|
+
if isinstance(item, list):
|
|
68
|
+
return "[" + ",".join(encode(member) for member in item) + "]"
|
|
69
|
+
if isinstance(item, dict):
|
|
70
|
+
if not all(isinstance(key, str) for key in item):
|
|
71
|
+
raise ArchiveError("source packet object keys must be strings")
|
|
72
|
+
keys = sorted(item, key=lambda key: key.encode("utf-16be"))
|
|
73
|
+
return "{" + ",".join(f"{encode(key)}:{encode(item[key])}" for key in keys) + "}"
|
|
74
|
+
raise ArchiveError(f"unsupported source packet value: {type(item).__name__}")
|
|
75
|
+
|
|
76
|
+
return encode(value).encode("utf-8")
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def sha256_hex(value: bytes) -> str:
|
|
80
|
+
return hashlib.sha256(value).hexdigest()
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def parse_bound(value: str | None, flag: str) -> dt.datetime | None:
|
|
84
|
+
if value is None:
|
|
85
|
+
return None
|
|
86
|
+
try:
|
|
87
|
+
parsed = dt.datetime.fromisoformat(value.replace("Z", "+00:00"))
|
|
88
|
+
except ValueError as exc:
|
|
89
|
+
raise ArchiveError(f"{flag} must be an ISO-8601 timestamp") from exc
|
|
90
|
+
if parsed.tzinfo is None:
|
|
91
|
+
raise ArchiveError(f"{flag} must include a timezone")
|
|
92
|
+
return parsed.astimezone(dt.timezone.utc)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def parse_created_at(value: object) -> str | None:
|
|
96
|
+
if not isinstance(value, str) or not value.strip():
|
|
97
|
+
return None
|
|
98
|
+
try:
|
|
99
|
+
parsed = email.utils.parsedate_to_datetime(value)
|
|
100
|
+
except (TypeError, ValueError):
|
|
101
|
+
try:
|
|
102
|
+
parsed = dt.datetime.fromisoformat(value.replace("Z", "+00:00"))
|
|
103
|
+
except ValueError:
|
|
104
|
+
return None
|
|
105
|
+
if parsed.tzinfo is None:
|
|
106
|
+
parsed = parsed.replace(tzinfo=dt.timezone.utc)
|
|
107
|
+
return parsed.astimezone(dt.timezone.utc).isoformat().replace("+00:00", "Z")
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def validate_member(info: zipfile.ZipInfo) -> None:
|
|
111
|
+
name = info.filename
|
|
112
|
+
if "\\" in name or "\x00" in name:
|
|
113
|
+
raise ArchiveError("archive contains an unsafe member name")
|
|
114
|
+
path = PurePosixPath(name)
|
|
115
|
+
if path.is_absolute() or ".." in path.parts:
|
|
116
|
+
raise ArchiveError("archive contains a path-traversal member")
|
|
117
|
+
unix_type = (info.external_attr >> 16) & 0o170000
|
|
118
|
+
if unix_type == stat.S_IFLNK:
|
|
119
|
+
raise ArchiveError("archive contains a symbolic-link member")
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def selected_members(archive: zipfile.ZipFile) -> list[zipfile.ZipInfo]:
|
|
123
|
+
infos = archive.infolist()
|
|
124
|
+
if len(infos) > MAX_ARCHIVE_MEMBERS:
|
|
125
|
+
raise ArchiveError("archive has too many members")
|
|
126
|
+
selected: list[zipfile.ZipInfo] = []
|
|
127
|
+
names: set[str] = set()
|
|
128
|
+
total = 0
|
|
129
|
+
for info in infos:
|
|
130
|
+
validate_member(info)
|
|
131
|
+
if not TWEET_MEMBER.search(info.filename):
|
|
132
|
+
continue
|
|
133
|
+
if info.filename in names:
|
|
134
|
+
raise ArchiveError("archive contains a duplicate posts member")
|
|
135
|
+
names.add(info.filename)
|
|
136
|
+
if info.file_size > MAX_SELECTED_MEMBER_BYTES:
|
|
137
|
+
raise ArchiveError("posts member exceeds the safety limit")
|
|
138
|
+
ratio = info.file_size / max(info.compress_size, 1)
|
|
139
|
+
if ratio > MAX_COMPRESSION_RATIO:
|
|
140
|
+
raise ArchiveError("posts member has an unsafe compression ratio")
|
|
141
|
+
total += info.file_size
|
|
142
|
+
if total > MAX_SELECTED_TOTAL_BYTES:
|
|
143
|
+
raise ArchiveError("combined posts members exceed the safety limit")
|
|
144
|
+
selected.append(info)
|
|
145
|
+
if not selected:
|
|
146
|
+
raise ArchiveError("archive contains no supported data/tweets*.js member")
|
|
147
|
+
return sorted(selected, key=lambda item: item.filename)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def read_bounded(archive: zipfile.ZipFile, info: zipfile.ZipInfo) -> bytes:
|
|
151
|
+
with archive.open(info, "r") as member:
|
|
152
|
+
data = member.read(info.file_size + 1)
|
|
153
|
+
if len(data) != info.file_size:
|
|
154
|
+
raise ArchiveError("posts member size changed while reading")
|
|
155
|
+
return data
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def parse_js_array(data: bytes, member_name: str) -> list[object]:
|
|
159
|
+
try:
|
|
160
|
+
text = data.decode("utf-8-sig")
|
|
161
|
+
except UnicodeDecodeError as exc:
|
|
162
|
+
raise ArchiveError(f"{member_name} is not UTF-8") from exc
|
|
163
|
+
prefix = JS_PREFIX.match(text)
|
|
164
|
+
if prefix is None:
|
|
165
|
+
raise ArchiveError(f"{member_name} has an unsupported JavaScript wrapper")
|
|
166
|
+
payload = text[prefix.end() :].strip()
|
|
167
|
+
if payload.endswith(";"):
|
|
168
|
+
payload = payload[:-1].rstrip()
|
|
169
|
+
try:
|
|
170
|
+
value = json.loads(payload)
|
|
171
|
+
except json.JSONDecodeError as exc:
|
|
172
|
+
raise ArchiveError(f"{member_name} does not contain valid JSON") from exc
|
|
173
|
+
if not isinstance(value, list):
|
|
174
|
+
raise ArchiveError(f"{member_name} must contain an array")
|
|
175
|
+
return value
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def classify_post(tweet: dict[str, object], text: str) -> tuple[str, str]:
|
|
179
|
+
if text.startswith("RT @"):
|
|
180
|
+
return "repost", "mixed"
|
|
181
|
+
if tweet.get("in_reply_to_status_id") or tweet.get("in_reply_to_status_id_str"):
|
|
182
|
+
return "reply", "subject"
|
|
183
|
+
if tweet.get("quoted_status_id") or tweet.get("quoted_status_id_str"):
|
|
184
|
+
return "quote_post", "subject"
|
|
185
|
+
return "post", "subject"
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def bounded_content(text: str) -> dict[str, object]:
|
|
189
|
+
candidate = text[:MAX_TEXT_CHARS]
|
|
190
|
+
truncated = len(candidate) != len(text)
|
|
191
|
+
|
|
192
|
+
def content(prefix: str, was_truncated: bool) -> dict[str, object]:
|
|
193
|
+
return {"text": prefix, "truncated": was_truncated}
|
|
194
|
+
|
|
195
|
+
if len(canonical_bytes(content(candidate, truncated))) <= MAX_RECORD_CONTENT_BYTES:
|
|
196
|
+
return content(candidate, truncated)
|
|
197
|
+
low = 0
|
|
198
|
+
high = len(candidate)
|
|
199
|
+
while low < high:
|
|
200
|
+
middle = (low + high + 1) // 2
|
|
201
|
+
if len(canonical_bytes(content(candidate[:middle], True))) <= MAX_RECORD_CONTENT_BYTES:
|
|
202
|
+
low = middle
|
|
203
|
+
else:
|
|
204
|
+
high = middle - 1
|
|
205
|
+
return content(candidate[:low], True)
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def post_record(raw: object, member_name: str) -> dict[str, object] | None:
|
|
209
|
+
if not isinstance(raw, dict):
|
|
210
|
+
return None
|
|
211
|
+
candidate = raw.get("tweet", raw)
|
|
212
|
+
if not isinstance(candidate, dict):
|
|
213
|
+
return None
|
|
214
|
+
post_id = candidate.get("id_str", candidate.get("id"))
|
|
215
|
+
text = candidate.get("full_text", candidate.get("text"))
|
|
216
|
+
if not isinstance(post_id, (str, int)) or not isinstance(text, str):
|
|
217
|
+
return None
|
|
218
|
+
post_id_text = str(post_id).strip()
|
|
219
|
+
if POST_ID.fullmatch(post_id_text) is None or not text.strip():
|
|
220
|
+
return None
|
|
221
|
+
content = bounded_content(text)
|
|
222
|
+
bounded_text = str(content["text"])
|
|
223
|
+
kind, author_role = classify_post(candidate, bounded_text)
|
|
224
|
+
occurred_at = parse_created_at(candidate.get("created_at"))
|
|
225
|
+
if occurred_at is None:
|
|
226
|
+
return None
|
|
227
|
+
content_hash = sha256_hex(canonical_bytes(content))
|
|
228
|
+
semantic: dict[str, object] = {
|
|
229
|
+
"id": f"x:{post_id_text}",
|
|
230
|
+
"kind": kind,
|
|
231
|
+
"authorRole": author_role,
|
|
232
|
+
"contentRole": "forwarded" if kind == "repost" else "original",
|
|
233
|
+
"authorshipConfidence": "strong",
|
|
234
|
+
"sentStatus": "published",
|
|
235
|
+
"visibility": "public",
|
|
236
|
+
"sourceClass": "polished_self_presentation",
|
|
237
|
+
"content": content,
|
|
238
|
+
"provenance": {
|
|
239
|
+
"provider": "x-archive",
|
|
240
|
+
"operation": "account-authored-public-post-export",
|
|
241
|
+
"sourceId": f"x-post:{post_id_text}",
|
|
242
|
+
"policyVersion": PAYLOAD_SCHEMA,
|
|
243
|
+
"contentSha256": content_hash,
|
|
244
|
+
},
|
|
245
|
+
}
|
|
246
|
+
semantic["occurredAt"] = occurred_at
|
|
247
|
+
semantic["digest"] = "sha256:" + sha256_hex(canonical_bytes(semantic))
|
|
248
|
+
semantic["_member"] = member_name
|
|
249
|
+
return semantic
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def choose_evenly(records: list[dict[str, object]], limit: int) -> list[dict[str, object]]:
|
|
253
|
+
if len(records) <= limit:
|
|
254
|
+
return records
|
|
255
|
+
if limit == 1:
|
|
256
|
+
return [records[-1]]
|
|
257
|
+
indexes = [round(index * (len(records) - 1) / (limit - 1)) for index in range(limit)]
|
|
258
|
+
return [records[index] for index in indexes]
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def build_packet(
|
|
262
|
+
archive_path: Path,
|
|
263
|
+
*,
|
|
264
|
+
limit: int,
|
|
265
|
+
after: dt.datetime | None,
|
|
266
|
+
before: dt.datetime | None,
|
|
267
|
+
) -> tuple[dict[str, object], dict[str, object]]:
|
|
268
|
+
selected_hash = hashlib.sha256()
|
|
269
|
+
records: list[dict[str, object]] = []
|
|
270
|
+
malformed = 0
|
|
271
|
+
duplicate_ids = 0
|
|
272
|
+
input_records = 0
|
|
273
|
+
selected_member_count = 0
|
|
274
|
+
with zipfile.ZipFile(archive_path, "r") as archive:
|
|
275
|
+
members = selected_members(archive)
|
|
276
|
+
selected_member_count = len(members)
|
|
277
|
+
for info in members:
|
|
278
|
+
data = read_bounded(archive, info)
|
|
279
|
+
selected_hash.update(info.filename.encode("utf-8"))
|
|
280
|
+
selected_hash.update(b"\x00")
|
|
281
|
+
selected_hash.update(data)
|
|
282
|
+
raw_records = parse_js_array(data, info.filename)
|
|
283
|
+
input_records += len(raw_records)
|
|
284
|
+
for raw in raw_records:
|
|
285
|
+
record = post_record(raw, info.filename)
|
|
286
|
+
if record is None:
|
|
287
|
+
malformed += 1
|
|
288
|
+
continue
|
|
289
|
+
records.append(record)
|
|
290
|
+
|
|
291
|
+
by_id: dict[str, dict[str, object]] = {}
|
|
292
|
+
for record in records:
|
|
293
|
+
record_id = str(record["id"])
|
|
294
|
+
if record_id in by_id:
|
|
295
|
+
previous = dict(by_id[record_id])
|
|
296
|
+
current = dict(record)
|
|
297
|
+
previous.pop("_member", None)
|
|
298
|
+
current.pop("_member", None)
|
|
299
|
+
if previous != current:
|
|
300
|
+
raise ArchiveError("archive contains conflicting records for one post ID")
|
|
301
|
+
duplicate_ids += 1
|
|
302
|
+
continue
|
|
303
|
+
by_id[record_id] = record
|
|
304
|
+
records = list(by_id.values())
|
|
305
|
+
records.sort(key=lambda value: (str(value.get("occurredAt", "")), str(value["id"])))
|
|
306
|
+
|
|
307
|
+
def in_bounds(record: dict[str, object]) -> bool:
|
|
308
|
+
value = record.get("occurredAt")
|
|
309
|
+
if not isinstance(value, str):
|
|
310
|
+
return after is None and before is None
|
|
311
|
+
timestamp = dt.datetime.fromisoformat(value.replace("Z", "+00:00"))
|
|
312
|
+
if after is not None and timestamp < after:
|
|
313
|
+
return False
|
|
314
|
+
if before is not None and timestamp >= before:
|
|
315
|
+
return False
|
|
316
|
+
return True
|
|
317
|
+
|
|
318
|
+
records = [record for record in records if in_bounds(record)]
|
|
319
|
+
eligible_count = len(records)
|
|
320
|
+
records = choose_evenly(records, limit)
|
|
321
|
+
for record in records:
|
|
322
|
+
record.pop("_member", None)
|
|
323
|
+
|
|
324
|
+
content_bytes = sum(len(canonical_bytes(record["content"])) for record in records)
|
|
325
|
+
if content_bytes > MAX_TOTAL_CONTENT_BYTES:
|
|
326
|
+
raise ArchiveError("selected post content exceeds the aggregate safety limit")
|
|
327
|
+
|
|
328
|
+
revision = selected_hash.hexdigest()
|
|
329
|
+
generated_at = dt.datetime.now(dt.timezone.utc).isoformat().replace("+00:00", "Z")
|
|
330
|
+
latest = max((str(record.get("occurredAt")) for record in records if record.get("occurredAt")), default=None)
|
|
331
|
+
if latest is not None and dt.datetime.fromisoformat(latest.replace("Z", "+00:00")) > dt.datetime.fromisoformat(
|
|
332
|
+
generated_at.replace("Z", "+00:00")
|
|
333
|
+
):
|
|
334
|
+
raise ArchiveError("selected posts contain an occurrence time later than packet generation")
|
|
335
|
+
limits: dict[str, object] = {
|
|
336
|
+
"maxRecords": limit,
|
|
337
|
+
"eligibleRecords": eligible_count,
|
|
338
|
+
"selection": "chronological-even-sample",
|
|
339
|
+
"afterInclusive": after.isoformat().replace("+00:00", "Z") if after else None,
|
|
340
|
+
"beforeExclusive": before.isoformat().replace("+00:00", "Z") if before else None,
|
|
341
|
+
"contentBytes": content_bytes,
|
|
342
|
+
"maxContentBytes": MAX_TOTAL_CONTENT_BYTES,
|
|
343
|
+
"selectedMembers": selected_member_count,
|
|
344
|
+
"inputRecords": input_records,
|
|
345
|
+
"malformedRecordsSkipped": malformed,
|
|
346
|
+
"exactDuplicateRecordsSkipped": duplicate_ids,
|
|
347
|
+
}
|
|
348
|
+
packet: dict[str, object] = {
|
|
349
|
+
"schemaVersion": SCHEMA_VERSION,
|
|
350
|
+
"digestCanonicalization": "JCS-RFC8785",
|
|
351
|
+
"packetId": f"ensoul_x_{revision[:24]}",
|
|
352
|
+
"generatedAt": generated_at,
|
|
353
|
+
"subject": {
|
|
354
|
+
"localId": f"x-archive-owner:{revision[:16]}",
|
|
355
|
+
"kind": "owner",
|
|
356
|
+
"identityBasis": "User-authorized official account archive; the user must confirm archive ownership.",
|
|
357
|
+
},
|
|
358
|
+
"scope": {
|
|
359
|
+
"adapter": "x-archive",
|
|
360
|
+
"payloadSchema": PAYLOAD_SCHEMA,
|
|
361
|
+
"completeness": (
|
|
362
|
+
"sampled" if eligible_count > limit
|
|
363
|
+
else "bounded" if malformed > 0 or duplicate_ids > 0
|
|
364
|
+
else "complete"
|
|
365
|
+
),
|
|
366
|
+
"sourceRevision": "sha256:" + revision,
|
|
367
|
+
"limits": limits,
|
|
368
|
+
},
|
|
369
|
+
"records": records,
|
|
370
|
+
"limitations": [
|
|
371
|
+
"Only allowlisted data/tweets*.js members were opened; direct messages, address books, advertising data, media, deleted posts, and community posts were not accessed.",
|
|
372
|
+
"Archive membership supports account authorship but does not prove that every embedded or quoted phrase was written by the subject; reposts are marked mixed.",
|
|
373
|
+
"Public visibility describes the original post context and is not permission to republish content that may since have been deleted or restricted.",
|
|
374
|
+
"Selection is bounded and chronological; counts do not measure importance, motive, or stable personality.",
|
|
375
|
+
],
|
|
376
|
+
}
|
|
377
|
+
if latest:
|
|
378
|
+
packet["scope"]["sourceCutoff"] = latest # type: ignore[index]
|
|
379
|
+
packet["packetDigest"] = "sha256:" + sha256_hex(canonical_bytes(packet))
|
|
380
|
+
receipt = {
|
|
381
|
+
"schemaVersion": SCHEMA_VERSION,
|
|
382
|
+
"packetDigest": packet["packetDigest"],
|
|
383
|
+
"records": len(records),
|
|
384
|
+
"eligibleRecords": eligible_count,
|
|
385
|
+
"malformedRecordsSkipped": malformed,
|
|
386
|
+
"duplicateIdsSkipped": duplicate_ids,
|
|
387
|
+
"contentBytes": content_bytes,
|
|
388
|
+
"maxContentBytes": MAX_TOTAL_CONTENT_BYTES,
|
|
389
|
+
"selectedMembers": selected_member_count,
|
|
390
|
+
"inputRecords": input_records,
|
|
391
|
+
"exactDuplicateRecordsSkipped": duplicate_ids,
|
|
392
|
+
"sourceRevision": packet["scope"]["sourceRevision"], # type: ignore[index]
|
|
393
|
+
}
|
|
394
|
+
return packet, receipt
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
def write_private_atomic(path: Path, data: bytes) -> None:
|
|
398
|
+
if not path.is_absolute():
|
|
399
|
+
raise ArchiveError("--output must be an absolute path")
|
|
400
|
+
parent = path.parent
|
|
401
|
+
resolved_parent = parent.resolve(strict=True)
|
|
402
|
+
if resolved_parent != parent:
|
|
403
|
+
raise ArchiveError("--output parent must not contain symbolic links")
|
|
404
|
+
if path.exists() or path.is_symlink():
|
|
405
|
+
raise ArchiveError("--output already exists")
|
|
406
|
+
temp = parent / f".{path.name}.tmp-{uuid.uuid4().hex}"
|
|
407
|
+
flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL
|
|
408
|
+
if hasattr(os, "O_NOFOLLOW"):
|
|
409
|
+
flags |= os.O_NOFOLLOW
|
|
410
|
+
descriptor = os.open(temp, flags, 0o600)
|
|
411
|
+
try:
|
|
412
|
+
with os.fdopen(descriptor, "wb", closefd=True) as stream:
|
|
413
|
+
stream.write(data)
|
|
414
|
+
stream.flush()
|
|
415
|
+
os.fsync(stream.fileno())
|
|
416
|
+
os.chmod(temp, 0o600)
|
|
417
|
+
os.link(temp, path, follow_symlinks=False)
|
|
418
|
+
directory = os.open(parent, os.O_RDONLY)
|
|
419
|
+
try:
|
|
420
|
+
os.fsync(directory)
|
|
421
|
+
finally:
|
|
422
|
+
os.close(directory)
|
|
423
|
+
finally:
|
|
424
|
+
try:
|
|
425
|
+
temp.unlink()
|
|
426
|
+
except FileNotFoundError:
|
|
427
|
+
pass
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
def parse_args(argv: list[str]) -> argparse.Namespace:
|
|
431
|
+
parser = argparse.ArgumentParser(description=__doc__)
|
|
432
|
+
parser.add_argument("archive", type=Path, help="absolute path to an official X archive ZIP")
|
|
433
|
+
parser.add_argument("--output", required=True, type=Path, help="new absolute private packet path")
|
|
434
|
+
parser.add_argument("--limit", type=int, default=2000, help="maximum evenly sampled posts (default: 2000)")
|
|
435
|
+
parser.add_argument("--after", help="inclusive ISO-8601 timestamp")
|
|
436
|
+
parser.add_argument("--before", help="exclusive ISO-8601 timestamp")
|
|
437
|
+
return parser.parse_args(argv)
|
|
438
|
+
|
|
439
|
+
|
|
440
|
+
def main(argv: list[str] | None = None) -> int:
|
|
441
|
+
args = parse_args(sys.argv[1:] if argv is None else argv)
|
|
442
|
+
try:
|
|
443
|
+
if not args.archive.is_absolute():
|
|
444
|
+
raise ArchiveError("archive path must be absolute")
|
|
445
|
+
if not args.archive.is_file():
|
|
446
|
+
raise ArchiveError("archive path must be an existing file")
|
|
447
|
+
if args.limit < 1 or args.limit > MAX_POSTS:
|
|
448
|
+
raise ArchiveError(f"--limit must be between 1 and {MAX_POSTS}")
|
|
449
|
+
after = parse_bound(args.after, "--after")
|
|
450
|
+
before = parse_bound(args.before, "--before")
|
|
451
|
+
if after and before and after >= before:
|
|
452
|
+
raise ArchiveError("--after must be earlier than --before")
|
|
453
|
+
packet, receipt = build_packet(args.archive, limit=args.limit, after=after, before=before)
|
|
454
|
+
payload = json.dumps(packet, ensure_ascii=False, indent=2).encode("utf-8") + b"\n"
|
|
455
|
+
if len(payload) > MAX_PACKET_BYTES:
|
|
456
|
+
raise ArchiveError("prepared packet exceeds the 128 MiB output safety limit")
|
|
457
|
+
write_private_atomic(args.output, payload)
|
|
458
|
+
receipt["output"] = str(args.output)
|
|
459
|
+
print(json.dumps(receipt, sort_keys=True, separators=(",", ":")))
|
|
460
|
+
return 0
|
|
461
|
+
except (ArchiveError, OSError, zipfile.BadZipFile) as exc:
|
|
462
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
463
|
+
return 2
|
|
464
|
+
|
|
465
|
+
|
|
466
|
+
if __name__ == "__main__":
|
|
467
|
+
raise SystemExit(main())
|