sediment-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1764 @@
1
+ # SPDX-License-Identifier: AGPL-3.0-or-later
2
+ """Sediment transcript extractor — the session-end edit observation client.
3
+
4
+ Runs as the harness's session-end hook (Claude Code and Codex SessionEnd; the
5
+ pi shim invokes it with ``--agent pi`` on session_shutdown): reads the hook
6
+ payload on stdin, parses the session transcript (JSONL at
7
+ ``transcript_path``), and ships bounded OTLP/JSON logs batches of
8
+ ``sediment.edit_observation`` records to the Sediment ingest endpoint
9
+ (``POST <endpoint>/v1/logs``) — one record per *applied* edit tool call,
10
+ carrying wire attributes that the server maps to an ``EditObservation``:
11
+
12
+ - ``applied_text`` is the ``new_string`` or ``content`` that the tool
13
+ applied.
14
+ - ``observed_file_text`` is the file content at session end (``""`` when
15
+ the file is gone).
16
+
17
+ No scoring happens here. Which edit retention metric to use is a server-side
18
+ derivation decision (``sediment_derive.survival``); the client ships
19
+ the pair, not a number, so the metric stays re-derivable over all history.
20
+
21
+ A second entry point, ``sediment transcript snapshot``, runs as a
22
+ PreToolUse hook and records line hashes around each edit call so the
23
+ session-end pass can report how many lines something *other* than the agent
24
+ changed (``external_lines_added``/``external_lines_removed``). See
25
+ the external-delta section below for why the transcript cannot supply that
26
+ on its own. Whether an external change was a human, a formatter, or a
27
+ rebase is a server-side judgment; the client ships counts, not intent.
28
+
29
+ Privacy contract (normative — tested): the payload contains ONLY model output
30
+ and the state of files the agent itself edited. Concretely: the AI-authored
31
+ text of applied edits (also captured as the completion at the gateway), the
32
+ session-end state of those files (the near-commit state the mirror captures), line
33
+ counts for changes the agent did not make, and the AI-authored text of edits
34
+ the developer refused. Retry linkage adds call identifiers, the file path, and
35
+ the edit-tool name. No prompts, correction text, conversation, Read/tool
36
+ results, environment, duplicate attempt text, or raw transcript leaves the
37
+ machine.
38
+
39
+ One of those is not a duplicate of another seam: a refused proposal on a
40
+ harness that does not route through the gateway is captured here and nowhere
41
+ else. It is still model output, never the developer's own work.
42
+
43
+ Best-effort: always exits 0. A missed session only means the signal degrades
44
+ to the attribution backstop (notes/jaccard) — it must never break the agent.
45
+
46
+ Harness seam: per-agent parsers (``_PARSERS``) extract edits from
47
+ each harness's session-file shape; everything downstream — pair-building,
48
+ the size cap, the privacy contract, the POST — is shared. Each record
49
+ carries an ``agent`` attribute (an ``AgentHarness`` enum value) so the
50
+ server attributes the pair without per-agent event names.
51
+
52
+ Environment:
53
+
54
+ SEDIMENT_OTLP_ENDPOINT ingest base URL — required; deliberately NO
55
+ OTEL_EXPORTER_OTLP_ENDPOINT fallback. Any standard
56
+ collector accepts POST /v1/logs, so inheriting the
57
+ machine's generic telemetry config would silently
58
+ ship edit text pairs to whatever third-party
59
+ collector it happens to export to.
60
+ SEDIMENT_INGEST_TOKEN bearer token (falls back to an Authorization header
61
+ in OTEL_EXPORTER_OTLP_HEADERS — safe, because it is
62
+ only ever sent to the explicit endpoint above)
63
+ OTEL_RESOURCE_ATTRIBUTES ``user.id`` is forwarded when present (parity with
64
+ the agents' own telemetry identity)
65
+
66
+ Unset endpoint = not opted in: the hook exits 0 without reading anything.
67
+ """
68
+
69
+ from __future__ import annotations
70
+
71
+ import difflib
72
+ import hashlib
73
+ import importlib.util
74
+ import json
75
+ import os
76
+ import re
77
+ import shutil
78
+ import stat
79
+ import sys
80
+ import time
81
+ from collections import Counter
82
+ from datetime import datetime
83
+ from pathlib import Path
84
+
85
+ EVENT_NAME = "sediment.edit_observation"
86
+ # The refused-edit record. Claude Code only: the pi session format
87
+ # reports isError without distinguishing a refusal from a tool failure, so
88
+ # the pi parser emits none rather than guessing.
89
+ REJECTED_EVENT_NAME = "sediment.rejected_edit"
90
+ RETRY_LINKAGE_EVENT_NAME = "sediment.retry_linkage"
91
+ # Edit tools whose input carries the full AI-authored text. NotebookEdit and
92
+ # the legacy MultiEdit are skipped (ponytail: cell-JSON and multi-part edits
93
+ # need their own pairing rules; those sessions fall back to the attribution
94
+ # backstop until those pairing rules exist).
95
+ _EDIT_TOOLS = {"Edit": "new_string", "Write": "content"}
96
+ # ponytail: fixed per-side cap keeps a pathological file from turning the
97
+ # hook into a multi-MB POST; raise if real source files exceed it.
98
+ MAX_TEXT_BYTES = 256 * 1024
99
+ _DELIVERY_MODULE = None
100
+
101
+
102
+ def _trail(msg: str) -> None:
103
+ print(f"sediment-transcript: {msg}", file=sys.stderr)
104
+
105
+
106
+ def _iter_jsonl(path: Path):
107
+ """Yield parsed JSONL entries, tolerating junk lines."""
108
+ try:
109
+ # JSONL is LF-framed. Unicode separators inside JSON strings are content.
110
+ lines = path.read_text(encoding="utf-8").split("\n")
111
+ except (OSError, UnicodeError):
112
+ _trail("transcript_unreadable; snapshots retained")
113
+ # The hook's fail-soft handler must skip cache cleanup on extraction
114
+ # failure. An unreadable source is not a readable session with no edits.
115
+ raise RuntimeError("transcript_unreadable") from None
116
+ for line in lines:
117
+ line = line.strip()
118
+ if not line:
119
+ continue
120
+ try:
121
+ entry = json.loads(line)
122
+ except ValueError:
123
+ continue
124
+ if isinstance(entry, dict):
125
+ yield entry
126
+
127
+
128
+ def _nanos(timestamp: object) -> int | None:
129
+ if not isinstance(timestamp, str):
130
+ return None
131
+ if timestamp.endswith("Z"): # fromisoformat only accepts Z on 3.11+
132
+ timestamp = timestamp[:-1] + "+00:00"
133
+ try:
134
+ return int(datetime.fromisoformat(timestamp).timestamp() * 1_000_000_000)
135
+ except ValueError:
136
+ return None
137
+
138
+
139
+ def _ms_nanos(timestamp: object) -> int | None:
140
+ """pi message timestamps are Unix milliseconds (numbers), not ISO."""
141
+ if isinstance(timestamp, bool) or not isinstance(timestamp, (int, float)):
142
+ return None
143
+ if timestamp <= 0:
144
+ return None
145
+ return int(timestamp * 1_000_000)
146
+
147
+
148
+ # How a Claude Code transcript marks a call the developer refused, as
149
+ # opposed to one the tool failed on. Measured over ~150 real transcripts:
150
+ # the entry-level ``toolUseResult`` is exactly this string on every reject,
151
+ # and the model-visible ``content`` opens with the phrase below on the same
152
+ # 8/8 — the two never disagreed.
153
+ #
154
+ # Both are matched **exactly**, never as a substring. "rejected" appears in
155
+ # ordinary tool output (`git push` prints "! [rejected]", source files
156
+ # contain the word), and a substring test turned 8 real rejects into 23
157
+ # during the survey. The Edit/Write filter already excludes those, but the
158
+ # exactness is the part that must not rot.
159
+ _REJECT_RESULT = "User rejected tool use"
160
+ _REJECT_PHRASE = "The user doesn't want to proceed with this tool use."
161
+
162
+
163
+ def _result_text(block: dict) -> str:
164
+ """The model-visible text of a tool_result block, however it is shaped."""
165
+ content = block.get("content")
166
+ if isinstance(content, str):
167
+ return content
168
+ if isinstance(content, list):
169
+ return " ".join(
170
+ part.get("text", "")
171
+ for part in content
172
+ if isinstance(part, dict) and isinstance(part.get("text"), str)
173
+ )
174
+ return ""
175
+
176
+
177
+ def _is_user_reject(entry: dict, block: dict) -> bool:
178
+ """True when the developer refused this call, not when the tool failed.
179
+
180
+ Two independent markers, either sufficient: ``toolUseResult`` is a Claude
181
+ Code internal with no compatibility promise, while the result content is
182
+ the model-facing shape. A genuine tool error ("String to replace not
183
+ found", "File has not been read yet") matches neither.
184
+ """
185
+ if entry.get("toolUseResult") == _REJECT_RESULT:
186
+ return True
187
+ return _result_text(block).startswith(_REJECT_PHRASE)
188
+
189
+
190
+ def _claude_calls(entries, session_id: str | None = None) -> tuple[list[dict], dict]:
191
+ """Every Edit/Write call in the transcript, plus how each one ended.
192
+
193
+ Returns ``(calls, outcome_by_id)`` where the outcome is ``"ok"``,
194
+ ``"reject"`` (the developer refused), or ``"error"`` (the tool failed).
195
+ A call with no result at all — an interrupted session — appears in
196
+ ``calls`` and is absent from the mapping, which is neither an
197
+ application nor a refusal.
198
+
199
+ Lines from other sessions are skipped: a resumed session's transcript
200
+ embeds the prior session's history, each line stamped with its own
201
+ ``sessionId``. Those calls were already observed at their own session's
202
+ end — re-emitting them here would duplicate them under the wrong session
203
+ and drag that session's bounds backwards (ADR 0002).
204
+ """
205
+ calls: list[dict] = []
206
+ outcome: dict[str, str] = {}
207
+ for entry in entries:
208
+ line_sid = entry.get("sessionId")
209
+ if (
210
+ session_id is not None
211
+ and isinstance(line_sid, str)
212
+ and line_sid != session_id
213
+ ):
214
+ continue
215
+ message = entry.get("message")
216
+ if not isinstance(message, dict):
217
+ continue
218
+ content = message.get("content")
219
+ if not isinstance(content, list):
220
+ continue
221
+ for block in content:
222
+ if not isinstance(block, dict):
223
+ continue
224
+ if entry.get("type") == "assistant" and block.get("type") == "tool_use":
225
+ name = block.get("name")
226
+ field = _EDIT_TOOLS.get(name)
227
+ if field is None:
228
+ continue
229
+ tool_input = block.get("input")
230
+ if not isinstance(tool_input, dict):
231
+ continue
232
+ file_path = tool_input.get("file_path")
233
+ text = tool_input.get(field)
234
+ tool_use_id = block.get("id")
235
+ nanos = _nanos(entry.get("timestamp"))
236
+ if not (
237
+ isinstance(file_path, str)
238
+ and file_path
239
+ and isinstance(text, str)
240
+ and isinstance(tool_use_id, str)
241
+ and tool_use_id
242
+ and nanos
243
+ ):
244
+ continue
245
+ calls.append(
246
+ {
247
+ "tool_use_id": tool_use_id,
248
+ "tool_name": name,
249
+ "file_path": file_path,
250
+ "applied_text": text,
251
+ "time_unix_nano": nanos,
252
+ }
253
+ )
254
+ elif entry.get("type") == "user" and block.get("type") == "tool_result":
255
+ tool_use_id = block.get("tool_use_id")
256
+ if not isinstance(tool_use_id, str):
257
+ continue
258
+ if not block.get("is_error"):
259
+ outcome[tool_use_id] = "ok"
260
+ elif _is_user_reject(entry, block):
261
+ outcome[tool_use_id] = "reject"
262
+ else:
263
+ outcome[tool_use_id] = "error"
264
+ return calls, outcome
265
+
266
+
267
+ def extract_edits(entries, session_id: str | None = None) -> list[dict]:
268
+ """Applied Edit/Write tool calls from transcript entries.
269
+
270
+ An edit counts only when its ``tool_result`` arrived without ``is_error``
271
+ — a rejected or failed call never touched the file, and pairing its text
272
+ against the session-end state would fabricate a zero edit-retention signal
273
+ on top of the reject decision the OTLP wire already carries.
274
+ """
275
+ calls, outcome = _claude_calls(entries, session_id)
276
+ return [c for c in calls if outcome.get(c["tool_use_id"]) == "ok"]
277
+
278
+
279
+ def extract_rejected_edits(entries, session_id: str | None = None) -> list[dict]:
280
+ """Edit/Write calls the developer refused, with the text they refused.
281
+
282
+ The rejected side of a DPO pair needs the model's proposed text, and for
283
+ a non-gateway harness it exists nowhere else: the decision fact carries
284
+ ``file_path=""`` and no content, and the gateway stores an empty
285
+ completion for tool-call turns. The transcript still holds the call, so
286
+ the proposal is recoverable (ADR 0007 — model output only, no
287
+ prompts, no conversation, no file state).
288
+
289
+ Only refusals. A call the tool failed on emits nothing: the model's text
290
+ was never judged by anyone, so it is not a preference signal.
291
+ """
292
+ calls, outcome = _claude_calls(entries, session_id)
293
+ rejected = []
294
+ for call in calls:
295
+ if outcome.get(call["tool_use_id"]) != "reject":
296
+ continue
297
+ proposed = call["applied_text"]
298
+ if len(proposed.encode("utf-8", "replace")) > MAX_TEXT_BYTES:
299
+ _trail(
300
+ f"rejected edit over {MAX_TEXT_BYTES}B for {call['file_path']}; dropped"
301
+ )
302
+ continue
303
+ rejected.append({**call, "proposed": proposed})
304
+ return rejected
305
+
306
+
307
+ def _has_user_text(entry: dict) -> bool:
308
+ """Whether a user transcript entry carries a developer text turn."""
309
+ if entry.get("type") != "user":
310
+ return False
311
+ message = entry.get("message")
312
+ if not isinstance(message, dict):
313
+ return False
314
+ content = message.get("content")
315
+ if isinstance(content, str):
316
+ return bool(content.strip())
317
+ if not isinstance(content, list):
318
+ return False
319
+ return any(
320
+ isinstance(block, dict)
321
+ and block.get("type") == "text"
322
+ and isinstance(block.get("text"), str)
323
+ and bool(block["text"].strip())
324
+ for block in content
325
+ )
326
+
327
+
328
+ def _retry_entry_order(entry: dict) -> tuple[int, int, str]:
329
+ """Return deterministic transcript order for retry extraction."""
330
+ timestamp = _nanos(entry.get("timestamp")) or 2**63 - 1
331
+ type_order = 0 if entry.get("type") == "assistant" else 1
332
+ return timestamp, type_order, json.dumps(entry, sort_keys=True, default=str)
333
+
334
+
335
+ def extract_retry_linkages(entries, session_id: str | None = None) -> list[dict]:
336
+ """Link a human-rejected edit call to its accepted correction retry.
337
+
338
+ The rule requires the same session, Edit/Write tool, and file; a real
339
+ refusal for A; at least one developer text entry after A's rejection and
340
+ before B; and a successful result for B. Tool failures, agent-only
341
+ rewrites, and pure regenerate sequences emit nothing.
342
+ """
343
+ ordered = sorted(list(entries), key=_retry_entry_order)
344
+ if session_id is not None and (
345
+ not isinstance(session_id, str) or not session_id.strip()
346
+ ):
347
+ return []
348
+ ordered = [
349
+ entry
350
+ for entry in ordered
351
+ if isinstance(entry.get("sessionId"), str)
352
+ and bool(entry["sessionId"].strip())
353
+ and (session_id is None or entry["sessionId"] == session_id)
354
+ ]
355
+ calls, outcomes = _claude_calls(ordered, session_id)
356
+ calls_by_id = {call["tool_use_id"]: call for call in calls}
357
+ call_positions: dict[str, int] = {}
358
+ call_sessions: dict[str, str | None] = {}
359
+ outcome_positions: dict[str, int] = {}
360
+ outcome_sessions: dict[str, str | None] = {}
361
+ user_text_positions: dict[int, str | None] = {}
362
+
363
+ for index, entry in enumerate(ordered):
364
+ line_sid = entry.get("sessionId")
365
+ effective_session = line_sid
366
+ if _has_user_text(entry):
367
+ user_text_positions[index] = effective_session
368
+ message = entry.get("message")
369
+ content = message.get("content") if isinstance(message, dict) else None
370
+ if not isinstance(content, list):
371
+ continue
372
+ for block in content:
373
+ if not isinstance(block, dict):
374
+ continue
375
+ if entry.get("type") == "assistant" and block.get("type") == "tool_use":
376
+ call_id = block.get("id")
377
+ if call_id in calls_by_id:
378
+ call_positions[call_id] = index
379
+ call_sessions[call_id] = effective_session
380
+ elif entry.get("type") == "user" and block.get("type") == "tool_result":
381
+ call_id = block.get("tool_use_id")
382
+ if call_id in calls_by_id:
383
+ outcome_positions[call_id] = index
384
+ outcome_sessions[call_id] = effective_session
385
+
386
+ def position(call: dict) -> int:
387
+ return call_positions[call["tool_use_id"]]
388
+
389
+ rejected = sorted(
390
+ (
391
+ call
392
+ for call in calls
393
+ if outcomes.get(call["tool_use_id"]) == "reject"
394
+ and call["tool_use_id"] in outcome_positions
395
+ and call_sessions.get(call["tool_use_id"])
396
+ == outcome_sessions.get(call["tool_use_id"])
397
+ ),
398
+ key=position,
399
+ )
400
+ accepted = sorted(
401
+ (
402
+ call
403
+ for call in calls
404
+ if outcomes.get(call["tool_use_id"]) == "ok"
405
+ and call["tool_use_id"] in outcome_positions
406
+ and call_sessions.get(call["tool_use_id"])
407
+ == outcome_sessions.get(call["tool_use_id"])
408
+ ),
409
+ key=position,
410
+ )
411
+ linkages = []
412
+ for first in rejected:
413
+ rejected_id = first["tool_use_id"]
414
+ rejection_position = outcome_positions[rejected_id]
415
+ retry_session = call_sessions[rejected_id]
416
+ for retry in accepted:
417
+ accepted_id = retry["tool_use_id"]
418
+ retry_position = call_positions[accepted_id]
419
+ if retry_position <= rejection_position:
420
+ continue
421
+ if (
422
+ retry["tool_name"] != first["tool_name"]
423
+ or retry["file_path"] != first["file_path"]
424
+ or call_sessions[accepted_id] != retry_session
425
+ ):
426
+ continue
427
+ if not any(
428
+ rejection_position < user_position < retry_position
429
+ and user_session == retry_session
430
+ for user_position, user_session in user_text_positions.items()
431
+ ):
432
+ continue
433
+ linkages.append(
434
+ {
435
+ "rejected_call_id": rejected_id,
436
+ "accepted_call_id": accepted_id,
437
+ "tool_name": retry["tool_name"],
438
+ "file_path": retry["file_path"],
439
+ "time_unix_nano": retry["time_unix_nano"],
440
+ }
441
+ )
442
+ break
443
+ return linkages
444
+
445
+
446
+ # pi session files are JSONL too, but entry-shaped: a {"type": "session"}
447
+ # header then {"type": "message"} entries whose message holds the content
448
+ # (the upstream pi session format). Edit tools are "edit" ({path, edits:
449
+ # [{oldText, newText}]}) and "write" ({path, content}); results are
450
+ # role="toolResult" messages keyed by toolCallId.
451
+ _PI_EDIT_TOOLS = {"edit", "write"}
452
+ _PI_PARENT_MAX_BYTES = 64 * 1024 * 1024
453
+
454
+
455
+ def _pi_entry_index(entries: list[dict]) -> dict[str, str]:
456
+ """Validate a complete native source before deciding fork ownership."""
457
+ if not entries or entries[0].get("type") != "session":
458
+ raise ValueError
459
+ header = entries[0]
460
+ if (
461
+ header.get("version") != 3
462
+ or not isinstance(header.get("id"), str)
463
+ or not header["id"].strip()
464
+ ):
465
+ raise ValueError
466
+ index = {}
467
+ for entry in entries[1:]:
468
+ identity = entry.get("id")
469
+ if (
470
+ entry.get("type") == "session"
471
+ or not isinstance(entry.get("type"), str)
472
+ or not isinstance(identity, str)
473
+ or not identity.strip()
474
+ or identity in index
475
+ or (
476
+ entry.get("type") == "message"
477
+ and not isinstance(entry.get("message"), dict)
478
+ )
479
+ ):
480
+ raise ValueError
481
+ # pi re-chains parentId while copying the otherwise immutable entry.
482
+ index[identity] = json.dumps(
483
+ {k: v for k, v in entry.items() if k != "parentId"},
484
+ sort_keys=True,
485
+ allow_nan=False,
486
+ separators=(",", ":"),
487
+ )
488
+ return index
489
+
490
+
491
+ def _pi_parent_entries(reference: object, source_path: Path | None) -> list[dict]:
492
+ """Read one bounded regular parent leaf inside the child source directory."""
493
+ if not isinstance(reference, str) or not reference or source_path is None:
494
+ raise ValueError
495
+ if "\0" in reference or any(p in {".", ".."} for p in reference.split("/")):
496
+ raise ValueError
497
+ source = source_path.resolve(strict=True)
498
+ parent = Path(reference)
499
+ if not parent.is_absolute():
500
+ if parent.name != reference:
501
+ raise ValueError
502
+ parent = source.parent / parent
503
+ if (
504
+ parent.suffix != ".jsonl"
505
+ or parent.parent.resolve(strict=True) != source.parent
506
+ or parent.name == source.name
507
+ ):
508
+ raise ValueError
509
+ directory_fd = os.open(source.parent, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW)
510
+ try:
511
+ # NONBLOCK ensures an untrusted FIFO cannot block before the fstat gate.
512
+ fd = os.open(
513
+ parent.name,
514
+ os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK,
515
+ dir_fd=directory_fd,
516
+ )
517
+ with os.fdopen(fd, "rb") as stream:
518
+ before = os.fstat(stream.fileno())
519
+ if (
520
+ not stat.S_ISREG(before.st_mode)
521
+ or before.st_size > _PI_PARENT_MAX_BYTES
522
+ ):
523
+ raise ValueError
524
+ content = stream.read(_PI_PARENT_MAX_BYTES + 1)
525
+ after = os.fstat(stream.fileno())
526
+ if (
527
+ len(content) != before.st_size
528
+ or after.st_size != before.st_size
529
+ or after.st_mtime_ns != before.st_mtime_ns
530
+ ):
531
+ raise ValueError
532
+ finally:
533
+ os.close(directory_fd)
534
+ # Unlike ordinary malformed siblings, a damaged parent makes ownership
535
+ # unprovable. Never classify omitted parent entries as child-owned work.
536
+ entries = [
537
+ json.loads(line) for line in content.decode("utf-8").split("\n") if line.strip()
538
+ ]
539
+ if not all(isinstance(entry, dict) for entry in entries):
540
+ raise ValueError
541
+ return entries
542
+
543
+
544
+ def _pi_owned_entries(entries: list[dict], source_path: Path | None) -> list[dict]:
545
+ headers = [entry for entry in entries if entry.get("type") == "session"]
546
+ if not any("parentSession" in header for header in headers):
547
+ return entries
548
+ try:
549
+ child = _pi_entry_index(entries)
550
+ header = entries[0]
551
+ parent_entries = _pi_parent_entries(header["parentSession"], source_path)
552
+ parent = _pi_entry_index(parent_entries)
553
+ if parent_entries[0]["id"] == header["id"]:
554
+ raise ValueError
555
+ if any(child[key] != parent[key] for key in child.keys() & parent.keys()):
556
+ raise ValueError
557
+ except (OSError, ValueError, RuntimeError):
558
+ count = sum(entry.get("type") == "message" for entry in entries)
559
+ _trail(
560
+ f"pi_transcript_declined reason=parent_source_unverified skipped_entries={count}"
561
+ )
562
+ # Preserve snapshots on a source failure, just like an unreadable child.
563
+ raise RuntimeError("parent_source_unverified") from None
564
+ owned = [
565
+ entry
566
+ for entry in entries
567
+ if entry.get("type") != "message" or entry.get("id") not in parent
568
+ ]
569
+ count = len(entries) - len(owned)
570
+ if count:
571
+ _trail(f"pi_transcript_skipped reason=inherited_entry skipped_entries={count}")
572
+ return owned
573
+
574
+
575
+ def _pi_edit_path(file_path: str, cwd: object) -> str | None:
576
+ # These native expansion forms need their own source-path contract. The
577
+ # ordinary filesystem-path contract never consults the extractor's HOME.
578
+ if (
579
+ "\0" in file_path
580
+ or file_path.startswith(("~", "@", "file://"))
581
+ or re.search(r"[\u00a0\u2000-\u200a\u202f\u205f\u3000]", file_path)
582
+ ):
583
+ _trail("pi_transcript_declined reason=unsupported_path skipped_edits=1")
584
+ return None
585
+ if Path(file_path).is_absolute():
586
+ return os.path.normpath(file_path)
587
+ if not isinstance(cwd, str) or "\0" in cwd or not Path(cwd).is_absolute():
588
+ _trail(
589
+ "pi_transcript_declined reason=execution_directory_invalid skipped_edits=1"
590
+ )
591
+ return None
592
+ return os.path.normpath(os.path.join(cwd, file_path))
593
+
594
+
595
+ def _pi_applied_text(name: str, arguments: dict) -> str | None:
596
+ """The AI-authored text of a pi edit/write call, or None if unobservable."""
597
+ if name == "write":
598
+ content = arguments.get("content")
599
+ return content if isinstance(content, str) else None
600
+ # edit: the authored text is the replacement spans. ponytail: multi-span
601
+ # calls join newTexts with newlines — span-vs-file pairing is the same
602
+ # posture as Claude's new_string; the upgrade path, if precision floors
603
+ # show it matters, is reconstructing the post-edit file from the
604
+ # toolResult's details.patch (a standard unified diff).
605
+ edits = arguments.get("edits")
606
+ if not isinstance(edits, list) or not edits:
607
+ return None
608
+ texts = [e.get("newText") for e in edits if isinstance(e, dict)]
609
+ if len(texts) != len(edits) or not all(isinstance(t, str) for t in texts):
610
+ return None
611
+ return "\n".join(texts)
612
+
613
+
614
+ def extract_edits_pi(
615
+ entries, session_id: str | None = None, *, source_path: Path | None = None
616
+ ) -> list[dict]:
617
+ """Applied edit/write tool calls from pi session-file entries.
618
+
619
+ The header supplies execution cwd; fork ownership comes from an authorized
620
+ immediate parent source. A contradicted Session never supplies observations.
621
+ """
622
+ entries = list(entries)
623
+ for entry in entries:
624
+ if entry.get("type") == "session" and (
625
+ session_id is not None and entry.get("id") != session_id
626
+ ):
627
+ count = sum(item.get("type") == "message" for item in entries)
628
+ _trail(
629
+ f"pi_transcript_declined reason=session_mismatch skipped_entries={count}"
630
+ )
631
+ return []
632
+ entries = _pi_owned_entries(entries, source_path)
633
+ headers = [entry for entry in entries if entry.get("type") == "session"]
634
+ cwd = headers[0].get("cwd") if len(headers) == 1 else None
635
+ edits: list[dict] = []
636
+ ok: set[str] = set()
637
+ for entry in entries:
638
+ entry_type = entry.get("type")
639
+ if entry_type == "session":
640
+ continue
641
+ if entry_type != "message":
642
+ continue
643
+ message = entry.get("message")
644
+ if not isinstance(message, dict):
645
+ continue
646
+ role = message.get("role")
647
+ if role == "assistant":
648
+ content = message.get("content")
649
+ if not isinstance(content, list):
650
+ continue
651
+ nanos = _ms_nanos(message.get("timestamp"))
652
+ for block in content:
653
+ if not isinstance(block, dict) or block.get("type") != "toolCall":
654
+ continue
655
+ name = block.get("name")
656
+ if name not in _PI_EDIT_TOOLS:
657
+ continue
658
+ arguments = block.get("arguments")
659
+ if not isinstance(arguments, dict):
660
+ continue
661
+ file_path = arguments.get("path")
662
+ text = _pi_applied_text(name, arguments)
663
+ tool_use_id = block.get("id")
664
+ if not (
665
+ isinstance(file_path, str)
666
+ and file_path
667
+ and isinstance(text, str)
668
+ and isinstance(tool_use_id, str)
669
+ and tool_use_id
670
+ and nanos
671
+ ):
672
+ continue
673
+ file_path = _pi_edit_path(file_path, cwd)
674
+ if file_path is None:
675
+ continue
676
+ edits.append(
677
+ {
678
+ "tool_use_id": tool_use_id,
679
+ "tool_name": name,
680
+ "file_path": file_path,
681
+ "applied_text": text,
682
+ "time_unix_nano": nanos,
683
+ }
684
+ )
685
+ elif role == "toolResult":
686
+ if not message.get("isError"):
687
+ tool_use_id = message.get("toolCallId")
688
+ if isinstance(tool_use_id, str):
689
+ ok.add(tool_use_id)
690
+ return [e for e in edits if e["tool_use_id"] in ok]
691
+
692
+
693
+ def _codex_added_text(unified_diff: str) -> str | None:
694
+ """Read additions inside complete native unified-diff hunks."""
695
+ additions = []
696
+ remaining_old = remaining_new = 0
697
+ saw_hunk = False
698
+ lines = unified_diff.split("\n")
699
+ if lines[-1:] == [""]:
700
+ lines.pop()
701
+ for line in lines:
702
+ header = re.fullmatch(r"@@ -\d+(?:,(\d+))? \+\d+(?:,(\d+))? @@(?: .*)?", line)
703
+ if header:
704
+ if remaining_old or remaining_new:
705
+ return None
706
+ # A hunk cannot consume more lines than the supplied diff. Bound
707
+ # digit parsing too, so malformed counts cannot sink sibling edits.
708
+ counts = (header[1] or "1", header[2] or "1")
709
+ if any(len(count.lstrip("0")) > len(str(len(lines))) for count in counts):
710
+ return None
711
+ remaining_old, remaining_new = (
712
+ int(count.lstrip("0") or "0") for count in counts
713
+ )
714
+ if max(remaining_old, remaining_new) > len(lines):
715
+ return None
716
+ saw_hunk = True
717
+ continue
718
+ if not saw_hunk and line.startswith(("diff --git ", "index ", "--- ", "+++ ")):
719
+ continue
720
+ if line == "\" and saw_hunk:
721
+ continue
722
+ if not saw_hunk or not line:
723
+ return None
724
+ if line[0] == "+" and remaining_new:
725
+ remaining_new -= 1
726
+ additions.append(line[1:])
727
+ elif line[0] == "-" and remaining_old:
728
+ remaining_old -= 1
729
+ elif line[0] == " " and remaining_old and remaining_new:
730
+ remaining_old -= 1
731
+ remaining_new -= 1
732
+ else:
733
+ return None
734
+ return (
735
+ "\n".join(additions)
736
+ if saw_hunk and not (remaining_old or remaining_new)
737
+ else None
738
+ )
739
+
740
+
741
+ def _codex_shell_patch(command: str) -> tuple[str, str] | None:
742
+ """Read one apply_patch heredoc; never execute or infer shell context."""
743
+ lines = command.lstrip().split("\n")
744
+ while lines[-1:] == [""]:
745
+ lines.pop()
746
+ if not lines:
747
+ return None
748
+ header = re.fullmatch(
749
+ r"apply_patch <<\s*(?:'([^']+)'|\"([^\"]+)\"|(\w+))", lines[0]
750
+ )
751
+ if header is None:
752
+ if "apply_patch" in command:
753
+ _trail("execution_directory_unknown: unsupported shell command")
754
+ return None
755
+ delimiter = next(value for value in header.groups() if value is not None)
756
+ if lines[-1] != delimiter:
757
+ _trail("execution_directory_unknown: command extends beyond patch heredoc")
758
+ return None
759
+ patch = lines[1:-1]
760
+ if header[3] is not None and any(char in "\n".join(patch) for char in "$`\\"):
761
+ _trail("malformed_patch: unquoted heredoc can change authored text")
762
+ return None
763
+ if len(patch) < 3 or patch[0] != "*** Begin Patch" or patch[-1] != "*** End Patch":
764
+ _trail("malformed_patch: incomplete patch envelope")
765
+ return None
766
+ file_headers = [
767
+ line for line in patch if re.match(r"\*\*\* (Add|Update|Delete) File:", line)
768
+ ]
769
+ file_header = re.fullmatch(r"\*\*\* (Add|Update|Delete) File: (.+)", patch[1])
770
+ if len(file_headers) != 1 or file_header is None:
771
+ _trail("unsupported_patch_kind: expected one file")
772
+ return None
773
+ kind, path = file_header.groups()
774
+ if kind == "Delete":
775
+ _trail("unsupported_patch_kind: deletion has no authored proposal")
776
+ return None
777
+ body = patch[2:-1]
778
+ if kind == "Update" and body and body[0].startswith("*** Move to: "):
779
+ path = body.pop(0).removeprefix("*** Move to: ")
780
+ additions = []
781
+ # The first Update hunk may start with change lines without an @@ header.
782
+ in_hunk = kind == "Add" or bool(body)
783
+ for index, line in enumerate(body):
784
+ if kind == "Update" and (line == "@@" or line.startswith("@@ ")):
785
+ in_hunk = True
786
+ elif in_hunk and line.startswith("+"):
787
+ additions.append(line[1:])
788
+ elif (
789
+ kind == "Update" and in_hunk and (line.startswith((" ", "-")) or line == "")
790
+ ):
791
+ continue
792
+ elif (
793
+ kind == "Update"
794
+ and in_hunk
795
+ and index > 0
796
+ and line == "*** End of File"
797
+ and index == len(body) - 1
798
+ ):
799
+ continue
800
+ else:
801
+ _trail("malformed_patch: unsupported patch body")
802
+ return None
803
+ if not in_hunk or not path:
804
+ _trail("malformed_patch: missing patch body or path")
805
+ return None
806
+ return path, "\n".join(additions)
807
+
808
+
809
+ def _codex_status_reason(outputs: list[object]) -> str | None:
810
+ """Accept status metadata before the output delimiter, never stdout text."""
811
+ statuses = []
812
+ unknown = not outputs
813
+ for output in outputs:
814
+ if not isinstance(output, str):
815
+ unknown = True
816
+ continue
817
+ header, delimiter, _stdout = output.partition("\nOutput:")
818
+ wall_times = 0
819
+ codes = []
820
+ for line in header.splitlines():
821
+ status = re.fullmatch(
822
+ r"(?:Exit code: |Process exited with code )(-?\d{1,10})", line
823
+ )
824
+ if status:
825
+ codes.append(int(status[1]))
826
+ elif re.fullmatch(r"Wall time: \d+(?:\.\d+)? seconds", line):
827
+ wall_times += 1
828
+ elif not re.fullmatch(r"(?:Chunk ID: \S+|Original token count: \d+)", line):
829
+ unknown = True
830
+ statuses.extend(codes)
831
+ if not delimiter or wall_times != 1 or len(codes) != 1:
832
+ unknown = True
833
+ if len(set(statuses)) > 1:
834
+ return "execution_status_conflict"
835
+ if unknown or not statuses:
836
+ return "execution_status_unknown"
837
+ return "execution_failed" if statuses[0] != 0 else None
838
+
839
+
840
+ def _codex_edit_path(
841
+ file_path: str, cwd: object, arguments: dict | None = None
842
+ ) -> str | None:
843
+ """Resolve an edit against its declared execution directory."""
844
+ explicit = arguments is not None and "workdir" in arguments
845
+ directory = arguments["workdir"] if explicit else cwd
846
+ try:
847
+ path = Path(file_path)
848
+ if directory is None and not explicit:
849
+ if path.is_absolute():
850
+ return str(path)
851
+ _trail("execution_directory_unknown: relative path has no directory")
852
+ return None
853
+ if not isinstance(directory, str) or not directory:
854
+ raise ValueError("directory must be a nonempty path")
855
+ base = Path(directory)
856
+ if not base.is_absolute():
857
+ if not explicit or not isinstance(cwd, str) or not Path(cwd).is_absolute():
858
+ raise ValueError("directory cannot be resolved")
859
+ base = Path(cwd) / base
860
+ base = base.resolve(strict=True)
861
+ if not base.is_dir():
862
+ raise ValueError("directory is not a directory")
863
+ if "\x00" in file_path:
864
+ raise ValueError("invalid file path")
865
+ return str(path if path.is_absolute() else base / path)
866
+ except (OSError, ValueError, UnicodeError):
867
+ _trail("execution_directory_invalid: unusable execution directory or path")
868
+ return None
869
+
870
+
871
+ def extract_edits_codex(entries, session_id: str | None = None) -> list[dict]:
872
+ """Extract successful single-file native changes and supported shell patches.
873
+
874
+ # ponytail: multi-file calls need a Fact identity that can join each file;
875
+ # skip them until that contract exists, without inventing child call ids.
876
+ """
877
+ material = list(entries)
878
+ cwd: object = None
879
+ transcript_session: object = None
880
+ found_session = False
881
+ for entry in material:
882
+ if entry.get("type") != "session_meta":
883
+ continue
884
+ payload = entry.get("payload")
885
+ if not isinstance(payload, dict):
886
+ continue
887
+ transcript_session = payload.get("id") or payload.get("session_id")
888
+ if (
889
+ session_id is not None
890
+ and isinstance(transcript_session, str)
891
+ and transcript_session != session_id
892
+ ):
893
+ _trail(
894
+ f"session file header id {transcript_session} != {session_id}; skipping session"
895
+ )
896
+ return []
897
+ found_session = isinstance(transcript_session, str) and bool(transcript_session)
898
+ cwd = payload.get("cwd")
899
+ break
900
+ if session_id is not None and not found_session:
901
+ return []
902
+
903
+ outputs: dict[str, list[object]] = {}
904
+ for entry in material:
905
+ payload = entry.get("payload")
906
+ if (
907
+ entry.get("type") == "response_item"
908
+ and isinstance(payload, dict)
909
+ and payload.get("type") == "function_call_output"
910
+ and isinstance(payload.get("call_id"), str)
911
+ ):
912
+ outputs.setdefault(payload["call_id"], []).append(payload.get("output"))
913
+ edits = []
914
+ for entry in material:
915
+ payload = entry.get("payload")
916
+ if not isinstance(payload, dict):
917
+ continue
918
+ call_id = payload.get("call_id")
919
+ nanos = _nanos(entry.get("timestamp"))
920
+ item = payload.get("item")
921
+ completed_change = (
922
+ entry.get("type") == "event_msg"
923
+ and payload.get("type") == "item_completed"
924
+ and isinstance(item, dict)
925
+ and item.get("type") == "FileChange"
926
+ )
927
+ if completed_change:
928
+ thread_id = payload.get("thread_id")
929
+ if (
930
+ not isinstance(thread_id, str)
931
+ or not thread_id.strip()
932
+ or (transcript_session is not None and thread_id != transcript_session)
933
+ ):
934
+ _trail(
935
+ "malformed_patch: FileChange Session identity is absent or mismatched"
936
+ )
937
+ continue
938
+ # Codex 0.153.4 puts the real tool-call id and result inside item;
939
+ # the enclosing turn or event id cannot join a Developer decision.
940
+ payload = item
941
+ call_id = payload.get("id")
942
+ if not isinstance(call_id, str) or not call_id.strip() or not nanos:
943
+ _trail(
944
+ "malformed_patch: FileChange call identity or timestamp is absent"
945
+ )
946
+ continue
947
+ status = payload.get("status")
948
+ if status != "completed":
949
+ reason = (
950
+ "execution_failed"
951
+ if status in ("failed", "declined")
952
+ else "execution_status_unknown"
953
+ )
954
+ _trail(f"{reason}: patch {call_id}")
955
+ continue
956
+ if not isinstance(call_id, str) or not call_id or not nanos:
957
+ continue
958
+ arguments = None
959
+ if (
960
+ entry.get("type") == "response_item"
961
+ and payload.get("type") == "function_call"
962
+ and payload.get("name") == "exec_command"
963
+ ):
964
+ try:
965
+ arguments = (
966
+ json.loads(payload["arguments"])
967
+ if isinstance(payload.get("arguments"), str)
968
+ else None
969
+ )
970
+ except ValueError:
971
+ arguments = None
972
+ command = arguments.get("cmd") if isinstance(arguments, dict) else None
973
+ if not isinstance(command, str):
974
+ continue
975
+ patch = _codex_shell_patch(command)
976
+ if patch is None:
977
+ continue
978
+ reason = _codex_status_reason(outputs.get(call_id, []))
979
+ if reason:
980
+ _trail(f"{reason}: patch {call_id}")
981
+ continue
982
+ file_path, applied_text = patch
983
+ tool_name = "exec_command"
984
+ elif entry.get("type") == "event_msg" and (
985
+ completed_change
986
+ or (
987
+ payload.get("type") == "patch_apply_end"
988
+ and payload.get("success") is True
989
+ )
990
+ ):
991
+ changes = payload.get("changes")
992
+ if not isinstance(changes, dict) or not changes:
993
+ _trail(f"malformed_patch: patch {call_id} changes are absent")
994
+ continue
995
+ if len(changes) != 1:
996
+ _trail(
997
+ f"unsupported_patch_kind: multi-file patch {call_id} cannot join one EditObservation"
998
+ )
999
+ continue
1000
+ file_path, change = next(iter(changes.items()))
1001
+ if (
1002
+ not isinstance(file_path, str)
1003
+ or not file_path
1004
+ or not isinstance(change, dict)
1005
+ ):
1006
+ _trail(f"malformed_patch: patch {call_id}")
1007
+ continue
1008
+ kind = change.get("type")
1009
+ if kind == "add":
1010
+ applied_text = change.get("content")
1011
+ elif kind == "update":
1012
+ move_path = change.get("move_path")
1013
+ if move_path is not None:
1014
+ if not isinstance(move_path, str) or not move_path:
1015
+ _trail(f"malformed_patch: patch {call_id} move path")
1016
+ continue
1017
+ file_path = move_path
1018
+ diff = change.get("unified_diff")
1019
+ applied_text = (
1020
+ _codex_added_text(diff) if isinstance(diff, str) else None
1021
+ )
1022
+ else:
1023
+ _trail(f"unsupported_patch_kind: patch {call_id}")
1024
+ continue
1025
+ if not isinstance(applied_text, str):
1026
+ _trail(f"malformed_patch: patch {call_id}")
1027
+ continue
1028
+ tool_name = "apply_patch"
1029
+ else:
1030
+ continue
1031
+ path = _codex_edit_path(file_path, cwd, arguments)
1032
+ if path is not None:
1033
+ edits.append(
1034
+ {
1035
+ "tool_use_id": call_id,
1036
+ "tool_name": tool_name,
1037
+ "file_path": path,
1038
+ "applied_text": applied_text,
1039
+ "time_unix_nano": nanos,
1040
+ }
1041
+ )
1042
+ return edits
1043
+
1044
+
1045
+ # The harness seam: per-agent parsers, one shipper. A new harness
1046
+ # plugs in as one parser + tests; pair-building, caps, the privacy contract,
1047
+ # and the POST never change.
1048
+ _PARSERS = {
1049
+ "claude-code": extract_edits,
1050
+ "codex": extract_edits_codex,
1051
+ "pi": extract_edits_pi,
1052
+ }
1053
+
1054
+
1055
+ # A low edit retention score is ambiguous: the agent revising its own edit and a
1056
+ # human correcting it look identical. Telling them apart needs the file's
1057
+ # state *between* two agent edits, and the transcript cannot supply it —
1058
+ # Claude Code records ``toolUseResult.originalFile`` only for files under
1059
+ # ~10 KiB (measured over 1,056 samples: none above 9,969 chars, null for
1060
+ # every larger file). A transcript-only baseline would therefore exist for
1061
+ # small files and vanish for large ones, and a gap that tracks file size
1062
+ # biases every label derived from it.
1063
+ #
1064
+ # So the pre-edit state is read from disk at the one moment it is still
1065
+ # there: a PreToolUse hook (``sediment transcript snapshot``). Per edit
1066
+ # call it stores two line-hash lists — the file before the edit, and the
1067
+ # file as the edit will leave it, computed by performing the tool's own
1068
+ # substitution. The window between edit i and edit i+1 on one file is then
1069
+ # ``post_i`` vs ``pre_{i+1}``: whatever changed in there, this agent did
1070
+ # not do. The last edit's window closes against the session-end content
1071
+ # the extractor already reads.
1072
+ #
1073
+ # Hashes, never lines: the cache is local-only and never shipped, and the
1074
+ # wire carries counts alone. A window whose baseline is unobservable is
1075
+ # omitted, never reported as zero (AGENTS.md, absent-never-guessed).
1076
+
1077
+
1078
+ # Filename-safe, and never an all-dots name. The charset alone still admits
1079
+ # "." and "..", which are not children of anything — and this id becomes a
1080
+ # path that `_clear_cache` deletes, so a payload carrying ".." would walk out
1081
+ # of the cache root and take a sibling session's snapshots with it.
1082
+ _SAFE_ID = re.compile(r"^(?!\.+$)[A-Za-z0-9._-]{1,128}$")
1083
+ # ponytail: a crashed session leaves its dir behind; SessionEnd sweeps
1084
+ # siblings older than this. Raise it if sessions legitimately outlive it.
1085
+ _CACHE_TTL_S = 7 * 24 * 3600
1086
+
1087
+
1088
+ def _cache_root() -> Path | None:
1089
+ """The directory holding every session's snapshots, or None when no
1090
+ cache base is resolvable.
1091
+
1092
+ The ``sediment/deltas`` segments are appended under *any* base, the
1093
+ override included, so this path is always one this hook created. That is
1094
+ what makes the stale-session sweep in :func:`_clear_cache` safe: it
1095
+ deletes children of this directory, and pointing the override at a home
1096
+ directory must never turn that into deleting the home directory's
1097
+ contents.
1098
+
1099
+ A missing home (``HOME`` unset and no passwd entry — a minimal
1100
+ container) must not gate the emit: the cache no-ops and the session
1101
+ ships without deltas.
1102
+ """
1103
+ base = os.environ.get("SEDIMENT_DELTA_CACHE") or os.environ.get("XDG_CACHE_HOME")
1104
+ if not base:
1105
+ try:
1106
+ base = str(Path.home() / ".cache")
1107
+ except RuntimeError:
1108
+ return None
1109
+ return Path(base) / "sediment" / "deltas"
1110
+
1111
+
1112
+ def _cache_dir(session_id: str) -> Path | None:
1113
+ """This session's snapshot dir, or None when there is nothing to cache.
1114
+
1115
+ None when the id is not filename-safe, or when no cache base is
1116
+ resolvable — the cache is an optimization and never gates the emit.
1117
+ """
1118
+ if not _SAFE_ID.match(session_id):
1119
+ return None
1120
+ root = _cache_root()
1121
+ if root is None:
1122
+ return None
1123
+ return root / session_id
1124
+
1125
+
1126
+ def _line_hashes(text: str) -> list[str]:
1127
+ """Per-line digests — enough to diff, never enough to reconstruct."""
1128
+ return [
1129
+ hashlib.blake2b(line.encode("utf-8", "replace"), digest_size=8).hexdigest()
1130
+ for line in text.splitlines()
1131
+ ]
1132
+
1133
+
1134
+ def _post_edit_text(pre: str, tool_name: str, tool_input: dict) -> str | None:
1135
+ """The file as this edit will leave it, or None if that is unpredictable.
1136
+
1137
+ Performs the tool's own substitution against the on-disk state. An
1138
+ ``old_string`` that does not occur means the call will not apply as
1139
+ described (a stale read, a racing writer) — unpredictable, so the window
1140
+ it would have opened is omitted rather than guessed.
1141
+ """
1142
+ if tool_name == "Write":
1143
+ content = tool_input.get("content")
1144
+ return content if isinstance(content, str) else None
1145
+ old = tool_input.get("old_string")
1146
+ new = tool_input.get("new_string")
1147
+ if not isinstance(old, str) or not isinstance(new, str) or old not in pre:
1148
+ return None
1149
+ if tool_input.get("replace_all"):
1150
+ return pre.replace(old, new)
1151
+ return pre.replace(old, new, 1)
1152
+
1153
+
1154
+ def cmd_snapshot(hook: dict) -> None:
1155
+ """The PreToolUse hook: record one edit call's before/after line hashes.
1156
+
1157
+ One file per call rather than appends to a shared log — parallel edit
1158
+ calls would interleave a shared file, and a per-call name needs no lock.
1159
+ """
1160
+ session_id = hook.get("session_id")
1161
+ call_id = hook.get("tool_use_id")
1162
+ tool_name = hook.get("tool_name")
1163
+ tool_input = hook.get("tool_input")
1164
+ if not (
1165
+ isinstance(session_id, str)
1166
+ and isinstance(call_id, str)
1167
+ and _SAFE_ID.match(call_id)
1168
+ and tool_name in _EDIT_TOOLS
1169
+ and isinstance(tool_input, dict)
1170
+ ):
1171
+ return
1172
+ file_path = tool_input.get("file_path")
1173
+ directory = _cache_dir(session_id)
1174
+ if directory is None or not (isinstance(file_path, str) and file_path):
1175
+ return
1176
+ try:
1177
+ pre = Path(file_path).read_text(encoding="utf-8")
1178
+ except FileNotFoundError:
1179
+ pre = "" # a Write creating the file: no prior lines, not a gap
1180
+ except (OSError, UnicodeDecodeError) as exc:
1181
+ _trail(f"pre-edit state unreadable for {file_path} ({exc}); window skipped")
1182
+ return
1183
+ post = _post_edit_text(pre, tool_name, tool_input)
1184
+ directory.mkdir(parents=True, exist_ok=True)
1185
+ (directory / f"{call_id}.json").write_text(
1186
+ json.dumps(
1187
+ {
1188
+ "call_id": call_id,
1189
+ "file_path": file_path,
1190
+ # PreToolUse fires in edit order, so this orders the windows.
1191
+ "ts": time.time_ns(),
1192
+ "pre": _line_hashes(pre),
1193
+ "post": None if post is None else _line_hashes(post),
1194
+ }
1195
+ ),
1196
+ encoding="utf-8",
1197
+ )
1198
+
1199
+
1200
+ def _load_snapshots(session_id: str) -> list[dict]:
1201
+ directory = _cache_dir(session_id)
1202
+ if directory is None:
1203
+ return []
1204
+ try:
1205
+ entries = sorted(directory.iterdir())
1206
+ except OSError:
1207
+ return []
1208
+ out: list[dict] = []
1209
+ for entry in entries:
1210
+ try:
1211
+ record = json.loads(entry.read_text(encoding="utf-8"))
1212
+ except (OSError, ValueError):
1213
+ continue
1214
+ if (
1215
+ isinstance(record, dict)
1216
+ and isinstance(record.get("call_id"), str)
1217
+ and isinstance(record.get("file_path"), str)
1218
+ and isinstance(record.get("ts"), int)
1219
+ and isinstance(record.get("pre"), list)
1220
+ and isinstance(record.get("post"), (list, type(None)))
1221
+ ):
1222
+ out.append(record)
1223
+ return out
1224
+
1225
+
1226
+ def _clear_cache(session_id: str) -> None:
1227
+ """Drop this session's snapshots, and any left by a session that died.
1228
+
1229
+ Only ever deletes children of :func:`_cache_root`, and only ones whose
1230
+ name is a session id this hook could have written — a stray file or
1231
+ directory sharing the cache root is left alone.
1232
+ """
1233
+ directory = _cache_dir(session_id)
1234
+ if directory is None:
1235
+ return
1236
+ shutil.rmtree(directory, ignore_errors=True)
1237
+ root = _cache_root()
1238
+ if root is None:
1239
+ return
1240
+ cutoff = time.time() - _CACHE_TTL_S
1241
+ try:
1242
+ siblings = list(root.iterdir())
1243
+ except OSError:
1244
+ return
1245
+ for sibling in siblings:
1246
+ if not (sibling.is_dir() and _SAFE_ID.match(sibling.name)):
1247
+ continue
1248
+ try:
1249
+ if sibling.stat().st_mtime < cutoff:
1250
+ shutil.rmtree(sibling, ignore_errors=True)
1251
+ except OSError:
1252
+ continue
1253
+
1254
+
1255
+ def _line_delta(before: list[str], after: list[str]) -> tuple[int, int]:
1256
+ """(lines_added, lines_removed) between two line-hash sequences.
1257
+
1258
+ ``autojunk=False``: the default heuristic treats any line recurring in
1259
+ more than 1% of a long file as junk, which would quietly distort counts
1260
+ on exactly the large files this path exists to cover.
1261
+ """
1262
+ matcher = difflib.SequenceMatcher(None, before, after, autojunk=False)
1263
+ added = removed = 0
1264
+ for tag, i1, i2, j1, j2 in matcher.get_opcodes():
1265
+ if tag in ("replace", "delete"):
1266
+ removed += i2 - i1
1267
+ if tag in ("replace", "insert"):
1268
+ added += j2 - j1
1269
+ return added, removed
1270
+
1271
+
1272
+ def external_deltas(
1273
+ snapshots: list[dict],
1274
+ observed_file_states: dict[str, str | None],
1275
+ applied: set[str],
1276
+ ) -> dict[str, tuple[int, int]]:
1277
+ """Per applied edit call, the lines something other than this agent changed.
1278
+
1279
+ Windows are delimited by **applied** edits only. A call the transcript
1280
+ shows as rejected or failed never touched the file, so it is transparent
1281
+ here: letting one close the preceding window would end that window early
1282
+ and strand everything after it on a call that ships no pair. The result
1283
+ would be a confident ``(0, 0)`` for the very pattern this measurement
1284
+ exists to catch — the agent edits, the developer rejects its next
1285
+ proposal, then fixes the file by hand.
1286
+
1287
+ Only ``post`` opens a window, and only ``pre`` closes one, so an edit
1288
+ whose post state was unpredictable still closes its predecessor's window.
1289
+ """
1290
+ by_file: dict[str, list[dict]] = {}
1291
+ for snap in sorted(snapshots, key=lambda s: (s["ts"], s["call_id"])):
1292
+ if snap["call_id"] in applied:
1293
+ by_file.setdefault(snap["file_path"], []).append(snap)
1294
+ out: dict[str, tuple[int, int]] = {}
1295
+ for file_path, snaps in by_file.items():
1296
+ for index, snap in enumerate(snaps):
1297
+ opened = snap["post"]
1298
+ if opened is None: # unpredictable edit — no window to open
1299
+ continue
1300
+ if index + 1 < len(snaps):
1301
+ closed = snaps[index + 1]["pre"]
1302
+ else:
1303
+ observed_file_text = observed_file_states.get(file_path)
1304
+ if observed_file_text is None: # unreadable — absent, not zero
1305
+ continue
1306
+ closed = _line_hashes(observed_file_text)
1307
+ out[snap["call_id"]] = _line_delta(opened, closed)
1308
+ return out
1309
+
1310
+
1311
+ def _observed_file_states(edits: list[dict]) -> dict[str, str | None]:
1312
+ """Read each edited file's session-end content from disk, once per path.
1313
+
1314
+ A missing file reads as ``""`` (the edit fully discarded by deletion — a
1315
+ legitimate zero-survival observation). Any other failure reads as ``None``
1316
+ and drops that file's pairs: an unobservable session-end state is a capture
1317
+ gap, not a zero.
1318
+ """
1319
+ observed_file_states: dict[str, str | None] = {}
1320
+ for path in {e["file_path"] for e in edits}:
1321
+ try:
1322
+ observed_file_states[path] = Path(path).read_text(encoding="utf-8")
1323
+ except FileNotFoundError:
1324
+ target = Path(path)
1325
+ if target.is_symlink() or not target.parent.is_dir():
1326
+ _trail(
1327
+ f"observation_unreadable: unresolved file path {path}; pairs dropped"
1328
+ )
1329
+ observed_file_states[path] = None
1330
+ else:
1331
+ observed_file_states[path] = ""
1332
+ except (OSError, UnicodeError, ValueError) as exc:
1333
+ _trail(
1334
+ f"observation_unreadable: observed file state unreadable for {path} ({exc}); pairs dropped"
1335
+ )
1336
+ observed_file_states[path] = None
1337
+ return observed_file_states
1338
+
1339
+
1340
+ def build_pairs(
1341
+ entries,
1342
+ session_id: str | None = None,
1343
+ *,
1344
+ agent: str,
1345
+ source_path: Path | None = None,
1346
+ ) -> list[dict]:
1347
+ """Extract applied edits, attach observed file states, enforce the size cap.
1348
+
1349
+ Each pair also carries its external-delta counts when the
1350
+ PreToolUse snapshots cover that call; a call with no usable window
1351
+ simply omits them.
1352
+
1353
+ Counts ship per file under one rule: **every** applied edit on that file
1354
+ must both have a window and ship a pair. The server sums a file's
1355
+ windows forward from each edit, so the shipped windows have to tile the
1356
+ file's whole timeline. Any hole and the remaining counts understate by
1357
+ an unknown amount, or — worse — swallow the agent's own next edit and
1358
+ report it as somebody else's work.
1359
+
1360
+ A hole is anything that costs one applied edit its window or its pair: a
1361
+ missing or unreadable PreToolUse snapshot, an edit whose post state was
1362
+ unpredictable, an unreadable observed file state, an oversized side. Each of
1363
+ those is invisible to the server, so the honest move is to drop the
1364
+ file's counts here. Other files in the session keep theirs.
1365
+ """
1366
+ edits = (
1367
+ extract_edits_pi(entries, session_id, source_path=source_path)
1368
+ if agent == "pi"
1369
+ else _PARSERS[agent](entries, session_id)
1370
+ )
1371
+ observed_file_states = _observed_file_states(edits)
1372
+ deltas = (
1373
+ external_deltas(
1374
+ _load_snapshots(session_id),
1375
+ observed_file_states,
1376
+ {e["tool_use_id"] for e in edits},
1377
+ )
1378
+ if session_id
1379
+ else {}
1380
+ )
1381
+ shippable, holed = [], set()
1382
+ for edit in edits:
1383
+ observed_file_text = observed_file_states[edit["file_path"]]
1384
+ if observed_file_text is None:
1385
+ holed.add(edit["file_path"])
1386
+ continue
1387
+ if (
1388
+ len(edit["applied_text"].encode("utf-8", "replace")) > MAX_TEXT_BYTES
1389
+ or len(observed_file_text.encode("utf-8", "replace")) > MAX_TEXT_BYTES
1390
+ ):
1391
+ _trail(f"pair over {MAX_TEXT_BYTES}B for {edit['file_path']}; dropped")
1392
+ holed.add(edit["file_path"])
1393
+ continue
1394
+ shippable.append((edit, observed_file_text))
1395
+ # The whole-timeline check: an applied edit missing from `deltas` had no
1396
+ # usable window, which leaves a gap its neighbours would silently absorb.
1397
+ for edit in edits:
1398
+ if edit["tool_use_id"] not in deltas:
1399
+ holed.add(edit["file_path"])
1400
+ pairs = []
1401
+ for edit, observed_file_text in shippable:
1402
+ pair = {**edit, "observed_file_text": observed_file_text}
1403
+ if edit["file_path"] not in holed:
1404
+ pair["external_lines_added"], pair["external_lines_removed"] = deltas[
1405
+ edit["tool_use_id"]
1406
+ ]
1407
+ pairs.append(pair)
1408
+ return pairs
1409
+
1410
+
1411
+ def _resource_user_id() -> str | None:
1412
+ for part in os.environ.get("OTEL_RESOURCE_ATTRIBUTES", "").split(","):
1413
+ key, sep, value = part.partition("=")
1414
+ if sep and key.strip() == "user.id" and value.strip():
1415
+ return value.strip()
1416
+ return None
1417
+
1418
+
1419
+ def build_payload(
1420
+ session_id: str,
1421
+ pairs: list[dict],
1422
+ *,
1423
+ agent: str,
1424
+ rejected: list[dict] | None = None,
1425
+ linkages: list[dict] | None = None,
1426
+ ) -> dict:
1427
+ """The OTLP/JSON resource and ordered records for the session.
1428
+
1429
+ Applied edits ship as ``sediment.edit_observation``; refused ones ship as
1430
+ ``sediment.rejected_edit`` records. Publication packs these independent
1431
+ records into bounded requests; the server self-filters each record by body.
1432
+ """
1433
+
1434
+ def attrs(mapping: dict[str, str | int]) -> list[dict]:
1435
+ return [
1436
+ {
1437
+ "key": k,
1438
+ # OTLP/JSON encodes int64 as a string; the counts are the only
1439
+ # non-string attribute this client emits.
1440
+ "value": {"intValue": str(v)}
1441
+ if isinstance(v, int)
1442
+ else {"stringValue": v},
1443
+ }
1444
+ for k, v in mapping.items()
1445
+ ]
1446
+
1447
+ records = [
1448
+ {
1449
+ "body": {"stringValue": EVENT_NAME},
1450
+ "timeUnixNano": str(p["time_unix_nano"]),
1451
+ "attributes": attrs(
1452
+ {
1453
+ "session.id": session_id,
1454
+ "tool_use_id": p["tool_use_id"],
1455
+ "tool_name": p["tool_name"],
1456
+ "file_path": p["file_path"],
1457
+ "applied_text": p["applied_text"],
1458
+ "observed_file_text": p["observed_file_text"],
1459
+ "agent": agent,
1460
+ # Omitted, never zeroed, when no window covered the call.
1461
+ **{
1462
+ k: p[k]
1463
+ for k in ("external_lines_added", "external_lines_removed")
1464
+ if k in p
1465
+ },
1466
+ }
1467
+ ),
1468
+ }
1469
+ for p in pairs
1470
+ ]
1471
+ records += [
1472
+ {
1473
+ "body": {"stringValue": REJECTED_EVENT_NAME},
1474
+ "timeUnixNano": str(r["time_unix_nano"]),
1475
+ "attributes": attrs(
1476
+ {
1477
+ "session.id": session_id,
1478
+ "tool_use_id": r["tool_use_id"],
1479
+ "tool_name": r["tool_name"],
1480
+ "file_path": r["file_path"],
1481
+ # A refused edit never reached the file, so there is no
1482
+ # observed file state to ship.
1483
+ "proposed": r["proposed"],
1484
+ "agent": agent,
1485
+ }
1486
+ ),
1487
+ }
1488
+ for r in rejected or []
1489
+ ]
1490
+ records += [
1491
+ {
1492
+ "body": {"stringValue": RETRY_LINKAGE_EVENT_NAME},
1493
+ "timeUnixNano": str(linkage["time_unix_nano"]),
1494
+ "attributes": attrs(
1495
+ {
1496
+ "session.id": session_id,
1497
+ "rejected_call_id": linkage["rejected_call_id"],
1498
+ "accepted_call_id": linkage["accepted_call_id"],
1499
+ "tool_name": linkage["tool_name"],
1500
+ "file_path": linkage["file_path"],
1501
+ "agent": agent,
1502
+ }
1503
+ ),
1504
+ }
1505
+ for linkage in linkages or []
1506
+ ]
1507
+ user_id = _resource_user_id()
1508
+ resource = {"attributes": attrs({"user.id": user_id})} if user_id else {}
1509
+ return {
1510
+ "resourceLogs": [{"resource": resource, "scopeLogs": [{"logRecords": records}]}]
1511
+ }
1512
+
1513
+
1514
+ def _validated_endpoint(configured: str | None) -> str | None:
1515
+ """Validate an explicit capture endpoint without including its value in errors."""
1516
+ if not configured:
1517
+ return None
1518
+ return _delivery_client().configured_destination(
1519
+ "otlp", {"SEDIMENT_OTLP_ENDPOINT": configured}
1520
+ )
1521
+
1522
+
1523
+ def _endpoint() -> str | None:
1524
+ # Deliberately NOT OTEL_EXPORTER_OTLP_ENDPOINT: a generic collector must
1525
+ # never receive edit text because another tool enabled shared telemetry.
1526
+ try:
1527
+ return _validated_endpoint(os.environ.get("SEDIMENT_OTLP_ENDPOINT"))
1528
+ except ValueError:
1529
+ _trail(
1530
+ "configured ingest endpoint rejected; remote endpoints require HTTPS "
1531
+ "and HTTP is limited to literal loopback hosts"
1532
+ )
1533
+ return None
1534
+
1535
+
1536
+ def _delivery_client():
1537
+ """Load the same stdlib owner from a package or adjacent standalone copy."""
1538
+ global _DELIVERY_MODULE
1539
+ if _DELIVERY_MODULE is None:
1540
+ if __package__:
1541
+ from . import delivery
1542
+
1543
+ _DELIVERY_MODULE = delivery
1544
+ else:
1545
+ directory = Path(__file__).resolve().parent
1546
+ path = directory / "delivery.py"
1547
+ if not path.is_file():
1548
+ path = directory / "sediment_delivery.py"
1549
+ spec = importlib.util.spec_from_file_location("_sediment_delivery", path)
1550
+ if spec is None or spec.loader is None:
1551
+ raise RuntimeError("helper_unavailable")
1552
+ module = importlib.util.module_from_spec(spec)
1553
+ sys.modules[spec.name] = module
1554
+ spec.loader.exec_module(module)
1555
+ _DELIVERY_MODULE = module
1556
+ return _DELIVERY_MODULE
1557
+
1558
+
1559
+ def _post(url: str, payload: dict, *, buffered: bool = False) -> None:
1560
+ """Accept one prepared payload; Cursor's existing caller remains direct."""
1561
+ delivery = _delivery_client()
1562
+ request = delivery.prepare_request(
1563
+ "otlp", url, json.dumps(payload, allow_nan=False).encode("utf-8")
1564
+ )
1565
+ if buffered:
1566
+ result = delivery.deliver(request)
1567
+ else:
1568
+ result = delivery.send_once(request)
1569
+ if result.status not in {"queued", "acknowledged"}:
1570
+ raise RuntimeError(result.reason)
1571
+
1572
+
1573
+ def _prepare_requests(url: str, payload: dict):
1574
+ """Pack this client's independent records, encoding each record once.
1575
+
1576
+ This envelope belongs to build_payload, not arbitrary native OTLP batches:
1577
+ native translators may join decision and result records within a request.
1578
+ Preserve the same JSON encoding as _post, including envelope and separators.
1579
+ """
1580
+ delivery = _delivery_client()
1581
+ resource = payload["resourceLogs"][0]
1582
+ prefix = (
1583
+ b'{"resourceLogs": [{"resource": '
1584
+ + json.dumps(resource["resource"], allow_nan=False).encode("utf-8")
1585
+ + b', "scopeLogs": [{"logRecords": ['
1586
+ )
1587
+ suffix = b"]}]}]}"
1588
+ envelope_bytes = len(prefix) + len(suffix)
1589
+ prepared = []
1590
+ chunk = []
1591
+ chunk_bytes = envelope_bytes
1592
+ oversized = 0
1593
+
1594
+ def finish():
1595
+ prepared.append(
1596
+ (
1597
+ delivery.prepare_request(
1598
+ "otlp", url, prefix + b", ".join(chunk) + suffix
1599
+ ),
1600
+ len(chunk),
1601
+ )
1602
+ )
1603
+
1604
+ for record in resource["scopeLogs"][0]["logRecords"]:
1605
+ encoded = json.dumps(record, allow_nan=False).encode("utf-8")
1606
+ if envelope_bytes + len(encoded) > delivery.MAX_ENTRY_BYTES:
1607
+ oversized += 1
1608
+ continue
1609
+ if chunk and chunk_bytes + 2 + len(encoded) > delivery.MAX_ENTRY_BYTES:
1610
+ finish()
1611
+ chunk = []
1612
+ chunk_bytes = envelope_bytes
1613
+ chunk_bytes += len(encoded) + (2 if chunk else 0)
1614
+ chunk.append(encoded)
1615
+ if chunk:
1616
+ finish()
1617
+ return prepared, oversized
1618
+
1619
+
1620
+ def _publish(url: str, payload: dict) -> bool:
1621
+ """Attempt each prepared request once and report publication by unit."""
1622
+ delivery = _delivery_client()
1623
+ candidates = Counter(
1624
+ record["body"]["stringValue"]
1625
+ for record in payload["resourceLogs"][0]["scopeLogs"][0]["logRecords"]
1626
+ )
1627
+ prepared, oversized = _prepare_requests(url, payload)
1628
+ summary = {
1629
+ "candidate_records": {
1630
+ "edit_observations": candidates[EVENT_NAME],
1631
+ "rejected_edits": candidates[REJECTED_EVENT_NAME],
1632
+ "retry_linkages": candidates[RETRY_LINKAGE_EVENT_NAME],
1633
+ },
1634
+ "prepared_requests": len(prepared),
1635
+ "prepared_records": sum(count for _, count in prepared),
1636
+ "queued_requests": 0,
1637
+ "queued_records": 0,
1638
+ "acknowledged_requests": 0,
1639
+ "acknowledged_records": 0,
1640
+ "unsuccessful_requests": {},
1641
+ "unsuccessful_records": {},
1642
+ "record_too_large": oversized,
1643
+ "unsubmitted_requests": 0,
1644
+ "unsubmitted_records": 0,
1645
+ "delivery_mode": "buffered"
1646
+ if os.environ.get("SEDIMENT_DELIVERY_DIR")
1647
+ else "best_effort",
1648
+ }
1649
+ for index, (request, records) in enumerate(prepared):
1650
+ interrupted = False
1651
+ try:
1652
+ result = delivery.deliver(request)
1653
+ status = result.status
1654
+ if status not in {
1655
+ "queued",
1656
+ "acknowledged",
1657
+ "pending",
1658
+ "blocked",
1659
+ "declined",
1660
+ }:
1661
+ status = "invalid_disposition"
1662
+ except Exception as exc:
1663
+ # Exception text can contain payloads or credentials. These reasons
1664
+ # describe the boundary only; the attempted request isn't a suffix.
1665
+ status = "io_error" if isinstance(exc, OSError) else "publication_error"
1666
+ interrupted = True
1667
+ if status in {"queued", "acknowledged"}:
1668
+ summary[f"{status}_requests"] += 1
1669
+ summary[f"{status}_records"] += records
1670
+ else:
1671
+ for unit, count in (("requests", 1), ("records", records)):
1672
+ failures = summary[f"unsuccessful_{unit}"]
1673
+ failures[status] = failures.get(status, 0) + count
1674
+ if interrupted:
1675
+ suffix = prepared[index + 1 :]
1676
+ summary["unsubmitted_requests"] = len(suffix)
1677
+ summary["unsubmitted_records"] = sum(count for _, count in suffix)
1678
+ break
1679
+ complete = not (oversized or summary["unsuccessful_requests"])
1680
+ summary["outcome"] = "complete" if complete else "partial"
1681
+ _trail("emission_summary " + json.dumps(summary, sort_keys=True))
1682
+ if not complete:
1683
+ _trail(
1684
+ "partial_publication; snapshots retained; queued requests remain replayable; "
1685
+ "observations not durably enqueued can be lost and snapshots cannot reconstruct them"
1686
+ )
1687
+ return complete
1688
+
1689
+
1690
+ def main(argv: list[str] | None = None) -> int:
1691
+ """The hook entry point — SessionEnd, or ``snapshot`` for PreToolUse.
1692
+
1693
+ Every path exits 0.
1694
+ """
1695
+ try:
1696
+ url = _endpoint()
1697
+ if url is None:
1698
+ return 0 # not opted in — and nothing is cached either
1699
+ args = sys.argv[1:] if argv is None else argv
1700
+ if "--agent" not in args:
1701
+ _trail("missing --agent; skipping")
1702
+ return 0
1703
+ index = args.index("--agent")
1704
+ if index + 1 >= len(args):
1705
+ _trail("missing --agent value; skipping")
1706
+ return 0
1707
+ agent = args[index + 1]
1708
+ if agent not in _PARSERS:
1709
+ _trail(f"unknown --agent {agent!r} (known: {sorted(_PARSERS)}); skipping")
1710
+ return 0
1711
+ hook = json.loads(sys.stdin.read() or "{}")
1712
+ if not isinstance(hook, dict):
1713
+ return 0
1714
+ if args and args[0] == "snapshot": # positional: the PreToolUse hook
1715
+ cmd_snapshot(hook)
1716
+ return 0
1717
+ session_id = hook.get("session_id")
1718
+ transcript_path = hook.get("transcript_path")
1719
+ if not (
1720
+ isinstance(session_id, str)
1721
+ and session_id
1722
+ and isinstance(transcript_path, str)
1723
+ and transcript_path
1724
+ ):
1725
+ return 0
1726
+ entries = list(_iter_jsonl(Path(transcript_path)))
1727
+ pairs = build_pairs(
1728
+ entries, session_id, agent=agent, source_path=Path(transcript_path)
1729
+ )
1730
+ # Refusals are Claude Code-only for now; the pi parser cannot tell a
1731
+ # refusal from a tool failure, so it contributes nothing here.
1732
+ rejected = (
1733
+ extract_rejected_edits(entries, session_id)
1734
+ if agent == "claude-code"
1735
+ else []
1736
+ )
1737
+ linkages = (
1738
+ extract_retry_linkages(entries, session_id)
1739
+ if agent == "claude-code"
1740
+ else []
1741
+ )
1742
+ if pairs or rejected or linkages:
1743
+ complete = _publish(
1744
+ url,
1745
+ build_payload(
1746
+ session_id,
1747
+ pairs,
1748
+ agent=agent,
1749
+ rejected=rejected,
1750
+ linkages=linkages,
1751
+ ),
1752
+ )
1753
+ if not complete:
1754
+ return 0
1755
+ # Durable enqueue or acknowledged direct delivery owns the prepared
1756
+ # bytes. A decline preserves snapshots; replay never re-extracts them.
1757
+ _clear_cache(session_id)
1758
+ except Exception: # fail-soft: never fail the session end
1759
+ _trail("emit failed; Edit observation coverage is incomplete")
1760
+ return 0
1761
+
1762
+
1763
+ if __name__ == "__main__":
1764
+ sys.exit(main())