personal-understanding 2.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. personal_understanding/__init__.py +16 -0
  2. personal_understanding/backup_archive.py +357 -0
  3. personal_understanding/capture_attachment.py +90 -0
  4. personal_understanding/capture_user_update.py +95 -0
  5. personal_understanding/catalog_context.py +118 -0
  6. personal_understanding/catalog_utils.py +455 -0
  7. personal_understanding/cli_runtime.py +13 -0
  8. personal_understanding/conversation_starters.py +187 -0
  9. personal_understanding/derivation_ledger.py +142 -0
  10. personal_understanding/finalize_capture.py +34 -0
  11. personal_understanding/followup_check.py +56 -0
  12. personal_understanding/init_archive.py +54 -0
  13. personal_understanding/install_mcp.py +325 -0
  14. personal_understanding/maintenance_check.py +78 -0
  15. personal_understanding/mcp_server.py +275 -0
  16. personal_understanding/open_dashboard.py +368 -0
  17. personal_understanding/preflight_context.py +67 -0
  18. personal_understanding/query_context.py +404 -0
  19. personal_understanding/rebuild_views.py +60 -0
  20. personal_understanding/record_feedback.py +92 -0
  21. personal_understanding/register_important_update.py +61 -0
  22. personal_understanding/retrieve_context.py +330 -0
  23. personal_understanding/retrieve_v2.py +147 -0
  24. personal_understanding/review_context.py +211 -0
  25. personal_understanding/review_skill.py +379 -0
  26. personal_understanding/review_v2.py +74 -0
  27. personal_understanding/run_review_cycle.py +106 -0
  28. personal_understanding/salience_review.py +127 -0
  29. personal_understanding/session_check.py +99 -0
  30. personal_understanding/sitecustomize.py +10 -0
  31. personal_understanding/source_audit.py +92 -0
  32. personal_understanding/storage.py +99 -0
  33. personal_understanding/turn_receipts.py +75 -0
  34. personal_understanding/update_state.py +96 -0
  35. personal_understanding/v2_archive.py +680 -0
  36. personal_understanding/validate_memory.py +112 -0
  37. personal_understanding-2.2.0.dist-info/METADATA +187 -0
  38. personal_understanding-2.2.0.dist-info/RECORD +42 -0
  39. personal_understanding-2.2.0.dist-info/WHEEL +5 -0
  40. personal_understanding-2.2.0.dist-info/entry_points.txt +3 -0
  41. personal_understanding-2.2.0.dist-info/licenses/LICENSE +21 -0
  42. personal_understanding-2.2.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,16 @@
1
+ # Personal Understanding — packaging shim.
2
+ #
3
+ # The skill scripts are written as flat sibling modules (``from cli_runtime
4
+ # import ...``) so they can run directly out of a checked-out scripts/
5
+ # folder. When pip installs them under the ``personal_understanding`` package,
6
+ # that flat import style would break, so we put the package's own directory on
7
+ # sys.path first. This is pure packaging glue: it changes no behavior of the
8
+ # scripts themselves and is never imported when the skill runs from source.
9
+ from __future__ import annotations
10
+
11
+ import os
12
+ import sys
13
+
14
+ _THIS_DIR = os.path.dirname(os.path.abspath(__file__))
15
+ if _THIS_DIR not in sys.path:
16
+ sys.path.insert(0, _THIS_DIR)
@@ -0,0 +1,357 @@
1
+ #!/usr/bin/env python3
2
+ """Personal-understanding backup: working archive (preview) + archived snapshot (stable).
3
+
4
+ Model (as defined by the user on 2026-08-29):
5
+ - Working archive: the living archive, improving all the time;
6
+ - Archived snapshot (fixed filename, overwrites the old zip, no accumulating
7
+ snapshots): the "known-good" rollback point. The criterion is not "content
8
+ unchanged" but "works fine in use": when more than refresh_after_days days
9
+ (default 7) have passed since the last packaging and structural validation
10
+ passes (the certification test), the current working archive is zipped into
11
+ a new snapshot; if the working archive has not changed at all, no re-zipping
12
+ (saves bandwidth).
13
+ - Cloud (a WebDAV cloud drive via rclone): each run incrementally pushes
14
+ "working archive + archived snapshot" (overwrite update), no resident daemon;
15
+ skipped when rclone_remote is not configured;
16
+ - USB mirror is off by default (manual copying); with usb_mirror=true the
17
+ backup also incrementally updates the working archive to a removable drive.
18
+
19
+ This script only writes backups/ itself and never touches archive content such
20
+ as memory/ or sources/.
21
+ """
22
+ from __future__ import annotations
23
+ from cli_runtime import configure_utf8_stdio
24
+ configure_utf8_stdio()
25
+
26
+ import argparse
27
+ import ctypes
28
+ import hashlib
29
+ import json
30
+ import os
31
+ import shutil
32
+ import string
33
+ import subprocess
34
+ import sys
35
+ import zipfile
36
+ from datetime import date, datetime
37
+ from pathlib import Path
38
+
39
+ ROOT = Path(__file__).resolve().parents[1]
40
+ SCRIPTS = ROOT / "scripts"
41
+ BACKUPS = ROOT / "backups"
42
+ CONFIG = ROOT / "memory" / "backup-config.json"
43
+ STATE = ROOT / "memory" / "backup-state.json"
44
+ BACKUP_DUE_DAYS = 7
45
+ STABLE_ZIP = "personal-understanding-stable.zip"
46
+ STABLE_MANIFEST = "personal-understanding-stable.json"
47
+ PREVIOUS_ZIP = "personal-understanding-previous.zip"
48
+ PREVIOUS_MANIFEST = "personal-understanding-previous.json"
49
+ DRIVE_REMOVABLE = 2 # GetDriveTypeW: DRIVE_REMOVABLE
50
+
51
+ INCLUDE_DIRS = ("memory", "sources", "references", "scripts", "migrations", "dashboard", "agents", "tests")
52
+ INCLUDE_FILES = ("SKILL.md", "VERSION", "CHANGELOG.md", "open-dashboard.cmd", "register-mcp.cmd", "README.md")
53
+
54
+
55
+ def load_backup_config() -> dict:
56
+ if not CONFIG.exists():
57
+ return {}
58
+ try:
59
+ value = json.loads(CONFIG.read_text(encoding="utf-8"))
60
+ return value if isinstance(value, dict) else {}
61
+ except (OSError, json.JSONDecodeError):
62
+ return {}
63
+
64
+
65
+ def write_text_atomic(path: Path, content: str) -> None:
66
+ path.parent.mkdir(parents=True, exist_ok=True)
67
+ tmp = path.with_suffix(path.suffix + ".tmp")
68
+ tmp.write_text(content, encoding="utf-8")
69
+ os.replace(tmp, path)
70
+
71
+
72
+ def snapshot_paths(source_root: Path = ROOT) -> list[Path]:
73
+ paths: list[Path] = []
74
+ for name in INCLUDE_DIRS:
75
+ base = source_root / name
76
+ if not base.is_dir():
77
+ continue
78
+ paths.extend(path for path in base.rglob("*") if path.is_file() and "__pycache__" not in path.parts)
79
+ for name in INCLUDE_FILES:
80
+ path = source_root / name
81
+ if path.is_file():
82
+ paths.append(path)
83
+ return sorted(set(paths))
84
+
85
+
86
+ def archive_fingerprint(source_root: Path = ROOT) -> str:
87
+ digest = hashlib.sha256()
88
+ for path in snapshot_paths(source_root):
89
+ digest.update(path.relative_to(source_root).as_posix().encode("utf-8"))
90
+ digest.update(hashlib.sha256(path.read_bytes()).digest())
91
+ return digest.hexdigest()
92
+
93
+
94
+ def load_state() -> dict:
95
+ if not STATE.exists():
96
+ return {}
97
+ try:
98
+ value = json.loads(STATE.read_text(encoding="utf-8"))
99
+ return value if isinstance(value, dict) else {}
100
+ except (OSError, json.JSONDecodeError):
101
+ return {}
102
+
103
+
104
+ def stable_zip_path() -> Path:
105
+ return BACKUPS / STABLE_ZIP
106
+
107
+
108
+ def backup_age_days() -> int | None:
109
+ """Days since the last stable-snapshot certification; None if never certified."""
110
+ stable = stable_zip_path()
111
+ if not stable.exists():
112
+ return None
113
+ newest = datetime.fromtimestamp(stable.stat().st_mtime)
114
+ return (datetime.now() - newest).days
115
+
116
+
117
+ def _days_since(value: str | None, today: date) -> int | None:
118
+ if not value:
119
+ return None
120
+ try:
121
+ return (today - date.fromisoformat(str(value)[:10])).days
122
+ except ValueError:
123
+ return None
124
+
125
+
126
+ def should_promote(state: dict, fingerprint: str, today: date, config: dict) -> tuple[bool, str]:
127
+ """Snapshot refresh decision: promote a new version once the window elapses (passing validation is a hard gate); skip when the working archive is unchanged."""
128
+ refresh_after = int(config.get("refresh_after_days", 7))
129
+ if not stable_zip_path().exists():
130
+ return True, "no-zip-yet"
131
+ if state.get("body_fingerprint") == fingerprint:
132
+ return False, "body-unchanged-zip-already-current"
133
+ promoted_age = _days_since(state.get("promoted_at"), today)
134
+ if promoted_age is None:
135
+ return True, "no-promotion-record"
136
+ if promoted_age >= refresh_after:
137
+ return True, f"zip-{promoted_age}-days-old"
138
+ return False, "waiting-refresh-window"
139
+
140
+
141
+ def validation_gate() -> tuple[bool, str]:
142
+ """Certification test: promoting a stable snapshot requires structural validation to not be failed."""
143
+ proc = subprocess.run([sys.executable, str(SCRIPTS / "validate_memory.py"), "--json"], cwd=ROOT, capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=300)
144
+ try:
145
+ data = json.loads(proc.stdout)
146
+ except json.JSONDecodeError:
147
+ return False, f"validate output not parseable: {(proc.stderr or proc.stdout)[:150]}"
148
+ if data.get("status") == "failed":
149
+ return False, f"validate failed: {len(data.get('errors', []))} errors, not promoting"
150
+ return True, data.get("status", "unknown")
151
+
152
+
153
+ def promote_stable(source_root: Path = ROOT, backups_dir: Path | None = None) -> dict:
154
+ """Certify the current working archive as the new archived snapshot; the previous snapshot is kept for one generation (previous), capped at two files."""
155
+ backups = backups_dir or BACKUPS
156
+ ok, gate = validation_gate()
157
+ if not ok:
158
+ return {"promoted": False, "reason": gate}
159
+ paths = snapshot_paths(source_root)
160
+ backups.mkdir(parents=True, exist_ok=True)
161
+ target = backups / STABLE_ZIP
162
+ previous = backups / PREVIOUS_ZIP
163
+ if target.exists():
164
+ # Keep the previous snapshot for one generation: the copy that is always
165
+ # one version behind the working archive
166
+ shutil.copy2(target, previous)
167
+ if (backups / STABLE_MANIFEST).exists():
168
+ shutil.copy2(backups / STABLE_MANIFEST, backups / PREVIOUS_MANIFEST)
169
+ digests: dict[str, str] = {}
170
+ with zipfile.ZipFile(target, "w", zipfile.ZIP_DEFLATED) as zf:
171
+ for path in paths:
172
+ arcname = path.relative_to(source_root).as_posix()
173
+ data = path.read_bytes()
174
+ digests[arcname] = hashlib.sha256(data).hexdigest()
175
+ zf.writestr(arcname, data)
176
+ manifest = {
177
+ "promoted_at": datetime.now().astimezone().isoformat(timespec="seconds"),
178
+ "files": len(paths),
179
+ "total_bytes": sum(p.stat().st_size for p in paths),
180
+ "sha256": hashlib.sha256(target.read_bytes()).hexdigest(),
181
+ "member_count": len(digests),
182
+ }
183
+ write_text_atomic(backups / STABLE_MANIFEST, json.dumps(manifest, ensure_ascii=False, indent=2) + "\n")
184
+ return {"promoted": True, "reason": "validated-and-promoted", "zip": target.name, "files": len(paths), "sha256": manifest["sha256"], "previous_kept": previous.exists()}
185
+
186
+
187
+ def verify_stable() -> tuple[bool, str]:
188
+ stable = stable_zip_path()
189
+ if not stable.exists():
190
+ return False, "no stable zip"
191
+ manifest_path = BACKUPS / STABLE_MANIFEST
192
+ if not manifest_path.exists():
193
+ return False, "no stable manifest"
194
+ try:
195
+ expected = json.loads(manifest_path.read_text(encoding="utf-8")).get("sha256")
196
+ except (OSError, json.JSONDecodeError):
197
+ return False, "manifest unreadable"
198
+ actual = hashlib.sha256(stable.read_bytes()).hexdigest()
199
+ return (actual == expected), ("sha256-match" if actual == expected else "sha256-mismatch")
200
+
201
+
202
+ def removable_drives() -> list[Path]:
203
+ found = []
204
+ for letter in string.ascii_uppercase:
205
+ root = f"{letter}:\\"
206
+ if not os.path.exists(root):
207
+ continue
208
+ if ctypes.windll.kernel32.GetDriveTypeW(ctypes.c_wchar_p(root)) == DRIVE_REMOVABLE:
209
+ found.append(Path(root))
210
+ return found
211
+
212
+
213
+ def mirror_target(override: str = "", config: dict | None = None) -> Path | None:
214
+ """Optional full-archive mirror location: --also-to > backup-config.json > environment variable > USB drive."""
215
+ config = config if config is not None else load_backup_config()
216
+ value = str(override or "").strip()
217
+ if not value:
218
+ value = str(config.get("mirror_to") or "")
219
+ if not value:
220
+ value = os.environ.get("PERSONAL_BACKUP_MIRROR", "")
221
+ if value.strip():
222
+ return Path(value.strip())
223
+ if config.get("usb_mirror"):
224
+ wanted = str(config.get("usb_volume_label") or "").strip().casefold()
225
+ for drive in removable_drives():
226
+ if not wanted:
227
+ return drive
228
+ volume = ctypes.create_unicode_buffer(64)
229
+ if ctypes.windll.kernel32.GetVolumeInformationW(ctypes.c_wchar_p(str(drive)), volume, 64, None, None, None, None, 0) and volume.value.strip().casefold() == wanted:
230
+ return drive
231
+ return None
232
+
233
+
234
+ def mirror_body(target: Path, source_root: Path = ROOT) -> dict:
235
+ """Incrementally mirror the full archive to target (copy only changed files; delete files that no longer exist in the source)."""
236
+ dest_root = target / "personal-understanding-archive"
237
+ copied = skipped = 0
238
+ for src in snapshot_paths(source_root):
239
+ rel = src.relative_to(source_root)
240
+ dst = dest_root / rel
241
+ try:
242
+ same = dst.exists() and dst.stat().st_size == src.stat().st_size and dst.stat().st_mtime == src.stat().st_mtime
243
+ except OSError:
244
+ same = False
245
+ if same:
246
+ skipped += 1
247
+ continue
248
+ dst.parent.mkdir(parents=True, exist_ok=True)
249
+ shutil.copy2(src, dst)
250
+ copied += 1
251
+ removed_stale = 0
252
+ for dst in list(dest_root.rglob("*")):
253
+ if dst.is_file():
254
+ rel = dst.relative_to(dest_root)
255
+ if not (source_root / rel).exists():
256
+ dst.unlink()
257
+ removed_stale += 1
258
+ return {"target": str(dest_root), "copied": copied, "unchanged": skipped, "removed_stale": removed_stale}
259
+
260
+
261
+ def rclone_executable(config: dict) -> str | None:
262
+ configured = str(config.get("rclone_path") or "").strip()
263
+ if configured and Path(configured).exists():
264
+ return configured
265
+ return shutil.which("rclone")
266
+
267
+
268
+ def rclone_push(remote: str, config: dict) -> str:
269
+ """Push the archived snapshot (stable + previous) to the cloud as an
270
+ overwrite update, no accumulating history.
271
+
272
+ Pushes only the few files under backups/ instead of the whole working
273
+ archive directory: a WebDAV cloud drive (via rclone) rate-limits WebDAV
274
+ request frequency and hundreds of small files would hit that limit; the
275
+ zip itself is a complete copy of the working archive, so recovery =
276
+ download + unzip.
277
+ """
278
+ exe = rclone_executable(config)
279
+ if not exe:
280
+ return "skipped: rclone not found"
281
+ if not stable_zip_path().exists():
282
+ return "skipped: archived snapshot does not exist yet"
283
+ try:
284
+ proc = subprocess.run(
285
+ [exe, "copy", str(BACKUPS), f"{remote}:personal-understanding-archive/backups", "--transfers", "4"],
286
+ capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=900,
287
+ )
288
+ except subprocess.TimeoutExpired:
289
+ return "push timed out (900s); the next backup run will resume automatically once the network recovers"
290
+ if proc.returncode == 0:
291
+ return f"archived snapshot pushed to {remote}:personal-understanding-archive/backups (overwrite update)"
292
+ return f"push failed ({proc.returncode}): {(proc.stderr or proc.stdout).strip()[:200]}"
293
+
294
+
295
+ def main() -> int:
296
+ ap = argparse.ArgumentParser(description=__doc__)
297
+ ap.add_argument("--also-to", default="", help="temporarily override the full-archive mirror directory (takes priority over config)")
298
+ ap.add_argument("--verify", action="store_true", help="verify the SHA256 of the existing stable snapshot (no packaging, no push)")
299
+ ap.add_argument("--force-promote", action="store_true", help="skip the refresh-window decision and certify the current working archive as the new stable snapshot immediately")
300
+ args = ap.parse_args()
301
+
302
+ if args.verify:
303
+ ok, detail = verify_stable()
304
+ print(json.dumps({"verify": ok, "detail": detail}, ensure_ascii=False, indent=2))
305
+ return 0 if ok else 1
306
+
307
+ today = date.today()
308
+ config = load_backup_config()
309
+ fingerprint = archive_fingerprint()
310
+ state = load_state()
311
+
312
+ due, reason = should_promote(state, fingerprint, today, config)
313
+ if args.force_promote:
314
+ due, reason = True, "forced"
315
+ if reason == "body-unchanged-zip-already-current":
316
+ promoted_note = "working archive unchanged; archived snapshot is already current — skipping re-packaging and upload"
317
+ elif due:
318
+ result = promote_stable()
319
+ state["promoted_at"] = today.isoformat()
320
+ state["body_fingerprint"] = fingerprint
321
+ state["last_promotion_reason"] = reason
322
+ state["last_promotion_result"] = result
323
+ if result.get("promoted"):
324
+ promoted_note = f"certification test passed; archived snapshot updated to the current working archive ({result['files']} files)"
325
+ else:
326
+ promoted_note = f"certification test failed; archived snapshot kept at the previous version: {result.get('reason')}"
327
+ else:
328
+ promoted_note = f"archived snapshot still within its refresh window ({reason}); the working archive keeps running as the preview"
329
+ state["checked_at"] = datetime.now().astimezone().isoformat(timespec="seconds")
330
+ write_text_atomic(STATE, json.dumps(state, ensure_ascii=False, indent=2) + "\n")
331
+
332
+ mirror = mirror_target(args.also_to, config)
333
+ mirror_result = None
334
+ mirror_error = None
335
+ if mirror is not None:
336
+ try:
337
+ mirror_result = mirror_body(mirror)
338
+ except OSError as exc:
339
+ mirror_error = str(exc)
340
+
341
+ rclone_remote = str(config.get("rclone_remote") or "").strip()
342
+ rclone_status = rclone_push(rclone_remote, config) if rclone_remote else "rclone_remote not configured"
343
+ stable = stable_zip_path()
344
+ print(json.dumps({
345
+ "status": "ok",
346
+ "stable": promoted_note,
347
+ "stable_age_days": backup_age_days(),
348
+ "verified_stable": verify_stable()[0],
349
+ "mirror": mirror_result,
350
+ "mirror_error": mirror_error,
351
+ "rclone": rclone_status,
352
+ }, ensure_ascii=False, indent=2))
353
+ return 0
354
+
355
+
356
+ if __name__ == "__main__":
357
+ raise SystemExit(main())
@@ -0,0 +1,90 @@
1
+ #!/usr/bin/env python3
2
+ """Capture an attachment immutably, deduplicating exact binary matches."""
3
+ from __future__ import annotations
4
+ from cli_runtime import configure_utf8_stdio
5
+ configure_utf8_stdio()
6
+
7
+ import argparse
8
+ import hashlib
9
+ import json
10
+ import mimetypes
11
+ import shutil
12
+ import subprocess
13
+ import sys
14
+ from datetime import datetime
15
+ from pathlib import Path
16
+
17
+ from derivation_ledger import ID_RE, register_capture
18
+ from storage import atomic_write_text, mutation_lock
19
+ from turn_receipts import mark_captured, read_receipt
20
+
21
+ ROOT = Path(__file__).resolve().parents[1]
22
+ CONVERSATION = ROOT / "sources" / "conversation"
23
+ ATTACHMENTS = ROOT / "sources" / "attachments"
24
+
25
+
26
+ def sha256_file(path: Path) -> str:
27
+ digest = hashlib.sha256()
28
+ with path.open("rb") as handle:
29
+ for chunk in iter(lambda: handle.read(1024 * 1024), b""):
30
+ digest.update(chunk)
31
+ return digest.hexdigest()
32
+
33
+
34
+ def find_duplicate(source: Path, digest: str) -> Path | None:
35
+ for folder in (ROOT / "sources" / "images", ATTACHMENTS):
36
+ if not folder.exists():
37
+ continue
38
+ for candidate in folder.iterdir():
39
+ if not candidate.is_file() or candidate.suffix.lower() == ".json":
40
+ continue
41
+ if candidate.stat().st_size == source.stat().st_size and sha256_file(candidate) == digest:
42
+ return candidate
43
+ return None
44
+
45
+
46
+ def main() -> int:
47
+ ap = argparse.ArgumentParser(description=__doc__)
48
+ ap.add_argument("--file", required=True)
49
+ ap.add_argument("--capture-id", required=True)
50
+ ap.add_argument("--conversation-id", default="")
51
+ ap.add_argument("--captured-at", default="")
52
+ ap.add_argument("--message-kind", default="attachment")
53
+ ap.add_argument("--turn-id", required=True, help="preflight 生成的 personal turn receipt;附件捕获必须绑定它")
54
+ args = ap.parse_args()
55
+ if not ID_RE.fullmatch(args.capture_id):
56
+ raise SystemExit("capture-id 不合法。")
57
+ source = Path(args.file).resolve()
58
+ if not source.is_file():
59
+ raise SystemExit(f"附件不存在:{source}")
60
+ meta_path = CONVERSATION / f"{args.capture_id}.attachment.json"
61
+ digest = sha256_file(source)
62
+ receipt = read_receipt(args.turn_id, ROOT)
63
+ if not receipt or not receipt.get("requires_personal_understanding"):
64
+ raise SystemExit("必须先运行 preflight_context.py;turn receipt 不存在或并非个人材料。")
65
+ mime = mimetypes.guess_type(source.name)[0] or "application/octet-stream"
66
+ with mutation_lock(ROOT):
67
+ if meta_path.exists() or (CONVERSATION / f"{args.capture_id}.txt").exists():
68
+ raise SystemExit(f"拒绝覆盖已有捕获:{args.capture_id}")
69
+ duplicate = find_duplicate(source, digest); stored = duplicate; deduplicated = bool(duplicate)
70
+ created = False
71
+ try:
72
+ if stored is None:
73
+ ATTACHMENTS.mkdir(parents=True, exist_ok=True); stored = ATTACHMENTS / f"{args.capture_id}{source.suffix.lower()}"; shutil.copyfile(source, stored); created = True
74
+ if sha256_file(stored) != digest: raise RuntimeError("附件回读哈希校验失败")
75
+ meta = {"capture_id": args.capture_id, "captured_at": args.captured_at or datetime.now().astimezone().isoformat(timespec="seconds"), "speaker": "user", "message_kind": args.message_kind, "conversation_id": args.conversation_id or None, "content_type": mime, "original_filename": source.name, "byte_length": source.stat().st_size, "sha256": digest, "immutable": True, "source_path": stored.relative_to(ROOT).as_posix(), "deduplicated_exact_binary": deduplicated}
76
+ atomic_write_text(meta_path, json.dumps(meta, ensure_ascii=False, indent=2) + "\n")
77
+ register_capture(args.capture_id, source_path=meta["source_path"], captured_at=meta["captured_at"], message_kind=meta["message_kind"], content_sha256=digest, root=ROOT)
78
+ if args.turn_id: mark_captured(args.turn_id, args.capture_id, ROOT)
79
+ except Exception:
80
+ meta_path.unlink(missing_ok=True)
81
+ if created and stored is not None: stored.unlink(missing_ok=True)
82
+ raise
83
+ proc = subprocess.run([sys.executable, str(ROOT / "scripts" / "rebuild_views.py")], cwd=ROOT, capture_output=True, text=True, encoding="utf-8", errors="replace")
84
+ result = {"status": "captured", "capture_id": args.capture_id, "turn_id": args.turn_id or None, "source_path": meta["source_path"], "sha256": digest, "deduplicated_exact_binary": deduplicated, "derivation_status": "pending", "view_rebuild": proc.stdout.strip(), "next_required_action": "derive records or explicitly finalize as no-derivation-needed"}
85
+ print(json.dumps(result, ensure_ascii=False, indent=2))
86
+ return proc.returncode
87
+
88
+
89
+ if __name__ == "__main__":
90
+ raise SystemExit(main())
@@ -0,0 +1,95 @@
1
+ #!/usr/bin/env python3
2
+ """Capture a user personal-understanding update verbatim before derivation."""
3
+ from __future__ import annotations
4
+ from cli_runtime import configure_utf8_stdio
5
+ configure_utf8_stdio()
6
+
7
+ import argparse, hashlib, json, subprocess, sys
8
+ from datetime import datetime
9
+ from pathlib import Path
10
+ from derivation_ledger import register_capture
11
+ from storage import atomic_write_bytes, atomic_write_text, mutation_lock
12
+ from turn_receipts import mark_captured, read_receipt
13
+
14
+ ROOT = Path(__file__).resolve().parents[1]
15
+ CAPTURES = ROOT / "sources" / "conversation"
16
+ ID_RE = __import__("re").compile(r"^[a-z0-9][a-z0-9._-]+$")
17
+
18
+
19
+ def read_bytes(path: Path) -> bytes:
20
+ return path.read_bytes()
21
+
22
+
23
+ def main() -> int:
24
+ ap = argparse.ArgumentParser(description=__doc__)
25
+ source = ap.add_mutually_exclusive_group(required=True)
26
+ source.add_argument("--text", help="完整用户原话;命令行转义后的 UTF-8 文本")
27
+ source.add_argument("--file", help="包含完整用户原话的 UTF-8 文件")
28
+ source.add_argument("--stdin", action="store_true", help="从标准输入读取完整原话字节;超长消息优先用这个入口,避开命令行长度限制")
29
+ ap.add_argument("--capture-id", required=True)
30
+ ap.add_argument("--conversation-id", default="")
31
+ ap.add_argument("--captured-at", default="")
32
+ ap.add_argument("--message-kind", default="personal-understanding-update")
33
+ ap.add_argument("--turn-id", default="", help="preflight 生成的 personal turn receipt;新 capture 必须绑定它")
34
+ args = ap.parse_args()
35
+ if not ID_RE.fullmatch(args.capture_id):
36
+ raise SystemExit("capture-id 只能使用小写字母、数字、点、下划线和短横线。")
37
+ txt = CAPTURES / f"{args.capture_id}.txt"
38
+ meta_path = CAPTURES / f"{args.capture_id}.json"
39
+ if args.file:
40
+ raw = Path(args.file).read_bytes()
41
+ elif args.stdin:
42
+ raw = sys.stdin.buffer.read()
43
+ if not raw:
44
+ raise SystemExit("stdin 没有读到任何字节;拒绝写入空原话。")
45
+ else:
46
+ raw = str(args.text).encode("utf-8")
47
+ try:
48
+ text = raw.decode("utf-8")
49
+ except UnicodeDecodeError as exc:
50
+ raise SystemExit(f"原话必须是 UTF-8 文本:{exc}")
51
+ digest = hashlib.sha256(raw).hexdigest()
52
+ meta = {
53
+ "capture_id": args.capture_id,
54
+ "captured_at": args.captured_at or datetime.now().astimezone().isoformat(timespec="seconds"),
55
+ "speaker": "user",
56
+ "message_kind": args.message_kind,
57
+ "conversation_id": args.conversation_id or None,
58
+ "byte_length": len(raw),
59
+ "utf8_sha256": digest,
60
+ "codepoint_length": len(text),
61
+ "immutable": True,
62
+ "source_path": txt.relative_to(ROOT).as_posix(),
63
+ }
64
+ with mutation_lock(ROOT):
65
+ if txt.exists() or meta_path.exists():
66
+ raise SystemExit(f"Refusing to overwrite existing verbatim capture: {args.capture_id}")
67
+ receipt = read_receipt(args.turn_id, ROOT) if args.turn_id else None
68
+ if not receipt or not receipt.get("requires_personal_understanding"):
69
+ raise SystemExit("必须先对完整当前消息运行 preflight_context.py,并提供该 turn-id;turn receipt 不存在或并非个人材料。")
70
+ if receipt.get("message_sha256") != digest:
71
+ raise SystemExit("capture 原话与 preflight 的完整当前消息不一致;拒绝截断或替换原话。")
72
+ try:
73
+ atomic_write_bytes(txt, raw)
74
+ if txt.read_bytes() != raw:
75
+ raise RuntimeError("原话回读校验失败")
76
+ atomic_write_text(meta_path, json.dumps(meta, ensure_ascii=False, indent=2) + "\n")
77
+ register_capture(args.capture_id, source_path=meta["source_path"], captured_at=meta["captured_at"], message_kind=meta["message_kind"], content_sha256=digest, root=ROOT)
78
+ if args.turn_id:
79
+ mark_captured(args.turn_id, args.capture_id, ROOT)
80
+ except Exception:
81
+ meta_path.unlink(missing_ok=True)
82
+ txt.unlink(missing_ok=True)
83
+ raise
84
+ proc = subprocess.run([sys.executable, str(ROOT / "scripts" / "rebuild_views.py")], cwd=ROOT, capture_output=True, text=True, encoding="utf-8", errors="replace")
85
+ if proc.returncode:
86
+ print("原话已保存,但派生视图重建失败;原话不会回滚。", file=sys.stderr)
87
+ print(proc.stdout, file=sys.stderr)
88
+ print(proc.stderr, file=sys.stderr)
89
+ return proc.returncode
90
+ print(json.dumps({"status": "captured", "capture_id": args.capture_id, "turn_id": args.turn_id or None, "path": meta["source_path"], "sha256": digest, "derivation_status": "pending", "view_rebuild": proc.stdout.strip(), "next_required_action": "create derived records, then finalize this capture"}, ensure_ascii=False, indent=2))
91
+ return 0
92
+
93
+
94
+ if __name__ == "__main__":
95
+ raise SystemExit(main())