cctally 1.91.0 → 1.92.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/CHANGELOG.md +51 -0
  2. package/README.md +4 -2
  3. package/bin/_cctally_cache.py +903 -74
  4. package/bin/_cctally_config.py +57 -0
  5. package/bin/_cctally_core.py +94 -14
  6. package/bin/_cctally_dashboard.py +217 -19
  7. package/bin/_cctally_dashboard_conversation.py +170 -20
  8. package/bin/_cctally_dashboard_envelope.py +2 -0
  9. package/bin/_cctally_db.py +481 -19
  10. package/bin/_cctally_doctor.py +18 -1
  11. package/bin/_cctally_journal.py +1156 -21
  12. package/bin/_cctally_journal_repair.py +6 -0
  13. package/bin/_cctally_parser.py +26 -0
  14. package/bin/_cctally_quota.py +171 -55
  15. package/bin/_cctally_record.py +13 -1
  16. package/bin/_cctally_rederive.py +4 -0
  17. package/bin/_cctally_statusline.py +6 -6
  18. package/bin/_cctally_store.py +1061 -40
  19. package/bin/_cctally_transcript.py +32 -2
  20. package/bin/_cctally_tui.py +54 -6
  21. package/bin/_lib_cache_report.py +8 -3
  22. package/bin/_lib_codex_conversation.py +851 -81
  23. package/bin/_lib_codex_conversation_query.py +2031 -96
  24. package/bin/_lib_codex_find_projection.py +517 -0
  25. package/bin/_lib_codex_harness_preamble.py +176 -0
  26. package/bin/_lib_codex_hooks.py +5 -3
  27. package/bin/_lib_codex_js_scan.py +254 -0
  28. package/bin/_lib_codex_landmarks.py +309 -0
  29. package/bin/_lib_codex_title_clean.py +116 -0
  30. package/bin/_lib_conversation_dispatch.py +168 -22
  31. package/bin/_lib_conversation_query.py +62 -2
  32. package/bin/_lib_conversation_watch.py +4 -2
  33. package/bin/_lib_doctor.py +64 -0
  34. package/bin/_lib_quota_alert_axes.py +31 -34
  35. package/bin/_lib_stats_damage.py +523 -0
  36. package/bin/_lib_stats_publish.py +243 -0
  37. package/bin/cctally +17 -3
  38. package/dashboard/static/assets/index-Dat-mza6.js +97 -0
  39. package/dashboard/static/assets/{index-Dwirao3Y.css → index-DnWdv8um.css} +1 -1
  40. package/dashboard/static/dashboard.html +2 -2
  41. package/package.json +8 -1
  42. package/dashboard/static/assets/index-CILAoEja.js +0 -90
@@ -0,0 +1,517 @@
1
+ """Canonical visible-text projection and matching for Codex conversation find.
2
+
3
+ The module is deliberately stdlib-only. Its Markdown scanner models the
4
+ visible text-node boundaries used by the dashboard's ReactMarkdown/remark-gfm
5
+ surface; it is not an HTML renderer and never interprets raw HTML.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ from dataclasses import dataclass
10
+ import html
11
+ import re
12
+ from typing import Iterable, Iterator, Sequence
13
+
14
+
15
+ CODEX_FIND_PROJECTION_VERSION = 2
16
+
17
+
18
+ @dataclass(frozen=True)
19
+ class RenderLeaf:
20
+ key: str
21
+ text: str
22
+
23
+
24
+ @dataclass(frozen=True)
25
+ class ProjectedLeaf:
26
+ key: str
27
+ start: int
28
+ end: int
29
+
30
+
31
+ @dataclass(frozen=True)
32
+ class FindRange:
33
+ start: int
34
+ end: int
35
+
36
+
37
+ @dataclass(frozen=True)
38
+ class LeafFragment:
39
+ leaf_key: str
40
+ start: int
41
+ end: int
42
+
43
+
44
+ class _ProjectionBuilder:
45
+ def __init__(self) -> None:
46
+ self.parts: list[str] = []
47
+ self.leaves: list[dict[str, int | str]] = []
48
+ self.length = 0
49
+ self._open_leaf: int | None = None
50
+
51
+ def boundary(self) -> None:
52
+ self._open_leaf = None
53
+
54
+ def separator(self, value: str) -> None:
55
+ if not value:
56
+ return
57
+ self.boundary()
58
+ self.parts.append(value)
59
+ self.length += len(value)
60
+
61
+ def emit(self, value: str, *, boundary: bool = False, key: str | None = None) -> None:
62
+ if not value:
63
+ return
64
+ if boundary:
65
+ self.boundary()
66
+ start = self.length
67
+ self.parts.append(value)
68
+ self.length += len(value)
69
+ if key is not None:
70
+ self.leaves.append({"key": key, "start": start, "end": self.length})
71
+ self._open_leaf = None
72
+ return
73
+ if self._open_leaf is None:
74
+ self._open_leaf = len(self.leaves)
75
+ self.leaves.append({
76
+ "key": f"t{self._open_leaf}",
77
+ "start": start,
78
+ "end": self.length,
79
+ })
80
+ else:
81
+ self.leaves[self._open_leaf]["end"] = self.length
82
+
83
+ def value(self) -> tuple[str, tuple[ProjectedLeaf, ...]]:
84
+ return (
85
+ "".join(self.parts),
86
+ tuple(ProjectedLeaf(**leaf) for leaf in self.leaves),
87
+ )
88
+
89
+
90
+ _TABLE_DELIMITER_RE = re.compile(
91
+ r"^\s*\|?\s*:?-{3,}:?\s*(?:\|\s*:?-{3,}:?\s*)+\|?\s*$"
92
+ )
93
+ _BLOCK_PREFIX_RE = re.compile(
94
+ r"^\s*(?:(?:#{1,6})\s+|>\s?|(?:[-+*]|\d+[.)])\s+)"
95
+ )
96
+ _TASK_MARKER_RE = re.compile(r"^\[[ xX]\]\s+")
97
+ _AUTOLINK_RE = re.compile(r"<((?:https?://|mailto:)[^ <>]+|[^ <>@]+@[^ <>@]+)>")
98
+
99
+
100
+ def _table_cells(line: str) -> list[str]:
101
+ value = line.strip()
102
+ if value.startswith("|"):
103
+ value = value[1:]
104
+ if value.endswith("|"):
105
+ value = value[:-1]
106
+ cells: list[str] = []
107
+ current: list[str] = []
108
+ escaped = False
109
+ for char in value:
110
+ if escaped:
111
+ current.append(char)
112
+ escaped = False
113
+ elif char == "\\":
114
+ current.append(char)
115
+ escaped = True
116
+ elif char == "|":
117
+ cells.append("".join(current).strip())
118
+ current = []
119
+ else:
120
+ current.append(char)
121
+ cells.append("".join(current).strip())
122
+ return cells
123
+
124
+
125
+ def _find_closing(source: str, token: str, start: int) -> int:
126
+ cursor = start
127
+ while True:
128
+ found = source.find(token, cursor)
129
+ if found < 0:
130
+ return -1
131
+ backslashes = 0
132
+ probe = found - 1
133
+ while probe >= 0 and source[probe] == "\\":
134
+ backslashes += 1
135
+ probe -= 1
136
+ if backslashes % 2 == 0:
137
+ return found
138
+ cursor = found + len(token)
139
+
140
+
141
+ def _project_inline(source: str, builder: _ProjectionBuilder) -> None:
142
+ plain: list[str] = []
143
+
144
+ def flush() -> None:
145
+ if plain:
146
+ builder.emit(html.unescape("".join(plain)))
147
+ plain.clear()
148
+
149
+ cursor = 0
150
+ while cursor < len(source):
151
+ if source[cursor] == "\\" and cursor + 1 < len(source):
152
+ plain.append(source[cursor + 1])
153
+ cursor += 2
154
+ continue
155
+
156
+ if source[cursor] == "`":
157
+ run = 1
158
+ while cursor + run < len(source) and source[cursor + run] == "`":
159
+ run += 1
160
+ token = "`" * run
161
+ close = _find_closing(source, token, cursor + run)
162
+ if close >= 0:
163
+ flush()
164
+ builder.emit(source[cursor + run:close].strip(" "), boundary=True)
165
+ builder.boundary()
166
+ cursor = close + run
167
+ continue
168
+
169
+ if source.startswith("![", cursor) or source[cursor] == "[":
170
+ image = source.startswith("![", cursor)
171
+ label_start = cursor + (2 if image else 1)
172
+ label_end = source.find("](", label_start)
173
+ if label_end >= 0:
174
+ destination_end = source.find(")", label_end + 2)
175
+ if destination_end >= 0:
176
+ flush()
177
+ builder.boundary()
178
+ if not image:
179
+ _project_inline(source[label_start:label_end], builder)
180
+ builder.boundary()
181
+ cursor = destination_end + 1
182
+ continue
183
+
184
+ if source[cursor] == "<":
185
+ autolink = _AUTOLINK_RE.match(source, cursor)
186
+ if autolink is not None:
187
+ flush()
188
+ builder.boundary()
189
+ label = autolink.group(1)
190
+ builder.emit(label[7:] if label.startswith("mailto:") else label)
191
+ builder.boundary()
192
+ cursor = autolink.end()
193
+ continue
194
+
195
+ matched_delimiter = False
196
+ for token in ("**", "__", "~~", "*", "_"):
197
+ if not source.startswith(token, cursor):
198
+ continue
199
+ close = _find_closing(source, token, cursor + len(token))
200
+ if close < 0 or close == cursor + len(token):
201
+ continue
202
+ flush()
203
+ builder.boundary()
204
+ _project_inline(source[cursor + len(token):close], builder)
205
+ builder.boundary()
206
+ cursor = close + len(token)
207
+ matched_delimiter = True
208
+ break
209
+ if matched_delimiter:
210
+ continue
211
+
212
+ plain.append(source[cursor])
213
+ cursor += 1
214
+ flush()
215
+
216
+
217
+ def _strip_block_prefix(line: str) -> str:
218
+ value = _BLOCK_PREFIX_RE.sub("", line, count=1)
219
+ if _TASK_MARKER_RE.match(value):
220
+ value = _TASK_MARKER_RE.sub(" ", value, count=1)
221
+ if value.endswith(" "):
222
+ value = value[:-2]
223
+ elif value.endswith("\\"):
224
+ value = value[:-1]
225
+ return value
226
+
227
+
228
+ def project_markdown(source: str) -> tuple[str, tuple[ProjectedLeaf, ...]]:
229
+ builder = _ProjectionBuilder()
230
+ lines = source.replace("\r\n", "\n").replace("\r", "\n").split("\n")
231
+ blocks: list[tuple[str, object]] = []
232
+ cursor = 0
233
+ while cursor < len(lines):
234
+ line = lines[cursor]
235
+ if not line.strip():
236
+ cursor += 1
237
+ continue
238
+
239
+ fence = re.match(r"^\s*(`{3,}|~{3,})(?:[^`]*)$", line)
240
+ if fence:
241
+ token = fence.group(1)
242
+ body: list[str] = []
243
+ cursor += 1
244
+ while cursor < len(lines) and not re.match(
245
+ rf"^\s*{re.escape(token[0])}{{{len(token)},}}\s*$", lines[cursor]
246
+ ):
247
+ body.append(lines[cursor])
248
+ cursor += 1
249
+ closed = cursor < len(lines)
250
+ if closed:
251
+ cursor += 1
252
+ code = "\n".join(body)
253
+ if body:
254
+ code += "\n"
255
+ blocks.append(("code", code))
256
+ continue
257
+
258
+ if cursor + 1 < len(lines) and "|" in line and _TABLE_DELIMITER_RE.match(lines[cursor + 1]):
259
+ rows = [_table_cells(line)]
260
+ cursor += 2
261
+ while cursor < len(lines) and lines[cursor].strip() and "|" in lines[cursor]:
262
+ rows.append(_table_cells(lines[cursor]))
263
+ cursor += 1
264
+ blocks.append(("table", rows))
265
+ continue
266
+
267
+ paragraph = [_strip_block_prefix(line)]
268
+ cursor += 1
269
+ while cursor < len(lines) and lines[cursor].strip():
270
+ if re.match(r"^\s*(`{3,}|~{3,})", lines[cursor]):
271
+ break
272
+ paragraph.append(_strip_block_prefix(lines[cursor]))
273
+ cursor += 1
274
+ blocks.append(("paragraph", paragraph))
275
+
276
+ for block_index, (kind, value) in enumerate(blocks):
277
+ if block_index:
278
+ builder.separator("\n")
279
+ if kind == "code":
280
+ builder.emit(str(value), boundary=True)
281
+ builder.boundary()
282
+ elif kind == "table":
283
+ for row_index, row in enumerate(value):
284
+ if row_index:
285
+ builder.separator("\n")
286
+ for cell_index, cell in enumerate(row):
287
+ if cell_index:
288
+ builder.separator("\t")
289
+ builder.boundary()
290
+ _project_inline(cell, builder)
291
+ builder.boundary()
292
+ else:
293
+ for line_index, line in enumerate(value):
294
+ if line_index:
295
+ builder.separator("\n")
296
+ _project_inline(line, builder)
297
+ return builder.value()
298
+
299
+
300
+ def project_plain(leaves: Sequence[RenderLeaf]) -> tuple[str, tuple[ProjectedLeaf, ...]]:
301
+ builder = _ProjectionBuilder()
302
+ for leaf in leaves:
303
+ builder.emit(leaf.text, key=leaf.key)
304
+ return builder.value()
305
+
306
+
307
+ _CONTEXT_DIFF_GIT_RE = re.compile(r"diff --git a/\S+ b/\S+")
308
+ _CONTEXT_HUNK_RE = re.compile(r"^@@ -(\d+)(?:,\d+)? \+(\d+)(?:,\d+)? @@")
309
+ _CONTEXT_EXTENDED_HEADER_PREFIXES = (
310
+ "old mode ", "new mode ", "new file mode ", "deleted file mode ",
311
+ "rename from ", "rename to ", "copy from ", "copy to ",
312
+ "similarity index ", "dissimilarity index ", "index ",
313
+ )
314
+
315
+
316
+ def _context_is_diff_line(line: str) -> bool:
317
+ if _CONTEXT_DIFF_GIT_RE.search(line):
318
+ return True
319
+ if line.startswith(("--- ", "+++ ", "@@")):
320
+ return True
321
+ if line.startswith(_CONTEXT_EXTENDED_HEADER_PREFIXES):
322
+ return True
323
+ return line == "" or line[0] in "+- \\"
324
+
325
+
326
+ def _segment_context_body(text: str) -> list[tuple[str, str]]:
327
+ """Mirror ``contextDiff.ts::segmentContextBody`` without rendering HTML."""
328
+ lines = text.split("\n")
329
+ if lines and lines[-1] == "":
330
+ lines.pop()
331
+ segments: list[tuple[str, str]] = []
332
+ prose: list[str] = []
333
+ diff: list[str] = []
334
+ in_diff = False
335
+
336
+ def flush(kind: str, values: list[str]) -> None:
337
+ if values:
338
+ segments.append((kind, "\n".join(values)))
339
+ values.clear()
340
+
341
+ for line in lines:
342
+ if not in_diff:
343
+ match = _CONTEXT_DIFF_GIT_RE.search(line)
344
+ if match is None:
345
+ prose.append(line)
346
+ continue
347
+ before = line[:match.start()].rstrip()
348
+ if before:
349
+ prose.append(before)
350
+ flush("prose", prose)
351
+ in_diff = True
352
+ diff.append(line[match.start():])
353
+ elif _context_is_diff_line(line):
354
+ diff.append(line)
355
+ else:
356
+ flush("diff", diff)
357
+ in_diff = False
358
+ prose.append(line)
359
+ flush("prose", prose)
360
+ flush("diff", diff)
361
+ return segments
362
+
363
+
364
+ def _context_diff_rows(text: str) -> list[tuple[int, int, int, str]]:
365
+ """Mirror the visible row walk in ``contextDiff.ts::parseUnifiedDiff``."""
366
+ rows: list[tuple[int, int, int, str]] = []
367
+ file_index = -1
368
+ hunk_index = -1
369
+ row_index = 0
370
+ in_hunk = False
371
+ for line in text.split("\n"):
372
+ if _CONTEXT_DIFF_GIT_RE.search(line):
373
+ file_index += 1
374
+ hunk_index = -1
375
+ row_index = 0
376
+ in_hunk = False
377
+ continue
378
+ if _CONTEXT_HUNK_RE.match(line):
379
+ hunk_index += 1
380
+ row_index = 0
381
+ in_hunk = True
382
+ continue
383
+ if not in_hunk:
384
+ continue
385
+ if line.startswith(("--- ", "+++ ")) or line.startswith(
386
+ _CONTEXT_EXTENDED_HEADER_PREFIXES
387
+ ):
388
+ continue
389
+ if line == "" or line.startswith("\\"):
390
+ continue
391
+ rows.append((file_index, hunk_index, row_index, line[1:]))
392
+ row_index += 1
393
+ return rows
394
+
395
+
396
+ def _append_projected(
397
+ builder: _ProjectionBuilder,
398
+ projected: tuple[str, tuple[ProjectedLeaf, ...]],
399
+ *,
400
+ prefix: str,
401
+ ) -> None:
402
+ text, leaves = projected
403
+ if not text:
404
+ return
405
+ start = builder.length
406
+ builder.parts.append(text)
407
+ builder.length += len(text)
408
+ builder.boundary()
409
+ builder.leaves.extend({
410
+ "key": f"{prefix}/{leaf.key}",
411
+ "start": start + leaf.start,
412
+ "end": start + leaf.end,
413
+ } for leaf in leaves)
414
+
415
+
416
+ def project_context(source: str) -> tuple[str, tuple[ProjectedLeaf, ...]]:
417
+ """Project visible prose and diff-row leaves from one context body.
418
+
419
+ File headers and +/- statistics are derived card chrome, matching #482's
420
+ rule that only provider-authored render leaves enter the search surface.
421
+ """
422
+ builder = _ProjectionBuilder()
423
+ for segment_index, (kind, text) in enumerate(_segment_context_body(source)):
424
+ if kind == "prose":
425
+ projected = project_markdown(text)
426
+ if not projected[0]:
427
+ continue
428
+ if builder.parts:
429
+ builder.separator("\n")
430
+ _append_projected(
431
+ builder, projected, prefix=f"segments.{segment_index}.prose"
432
+ )
433
+ continue
434
+ for file_index, hunk_index, row_index, row_text in _context_diff_rows(text):
435
+ if not row_text:
436
+ continue
437
+ if builder.parts:
438
+ builder.separator("\n")
439
+ builder.emit(
440
+ row_text,
441
+ key=(
442
+ f"segments.{segment_index}.files.{file_index}."
443
+ f"hunks.{hunk_index}.rows.{row_index}"
444
+ ),
445
+ )
446
+ return builder.value()
447
+
448
+
449
+ def _single_scalar_lower(value: str) -> str:
450
+ return "".join((lowered if len(lowered := scalar.lower()) == 1 else scalar) for scalar in value)
451
+
452
+
453
+ def iter_literal_ranges(
454
+ text: str, query: str, *, case_sensitive: bool,
455
+ ) -> Iterator[FindRange]:
456
+ if not query:
457
+ return
458
+ haystack = text if case_sensitive else _single_scalar_lower(text)
459
+ needle = query if case_sensitive else _single_scalar_lower(query)
460
+ if not needle:
461
+ return
462
+ cursor = 0
463
+ while cursor <= len(haystack) - len(needle):
464
+ found = haystack.find(needle, cursor)
465
+ if found < 0:
466
+ break
467
+ yield FindRange(found, found + len(needle))
468
+ cursor = found + len(needle)
469
+
470
+
471
+ def literal_ranges(text: str, query: str, *, case_sensitive: bool) -> tuple[FindRange, ...]:
472
+ return tuple(iter_literal_ranges(text, query, case_sensitive=case_sensitive))
473
+
474
+
475
+ def iter_regex_ranges(text: str, pattern: re.Pattern[str]) -> Iterator[FindRange]:
476
+ for match in pattern.finditer(text):
477
+ if match.end() > match.start():
478
+ yield FindRange(match.start(), match.end())
479
+
480
+
481
+ def regex_ranges(text: str, pattern: re.Pattern[str]) -> tuple[FindRange, ...]:
482
+ return tuple(iter_regex_ranges(text, pattern))
483
+
484
+
485
+ def slice_range_to_leaves(
486
+ match: FindRange,
487
+ leaves: Sequence[ProjectedLeaf],
488
+ ) -> tuple[LeafFragment, ...]:
489
+ fragments: list[LeafFragment] = []
490
+ for leaf in leaves:
491
+ start = max(match.start, leaf.start)
492
+ end = min(match.end, leaf.end)
493
+ if end <= start:
494
+ continue
495
+ fragments.append(LeafFragment(
496
+ leaf_key=leaf.key,
497
+ start=start - leaf.start,
498
+ end=end - leaf.start,
499
+ ))
500
+ return tuple(fragments)
501
+
502
+
503
+ __all__ = [
504
+ "CODEX_FIND_PROJECTION_VERSION",
505
+ "FindRange",
506
+ "LeafFragment",
507
+ "ProjectedLeaf",
508
+ "RenderLeaf",
509
+ "literal_ranges",
510
+ "iter_literal_ranges",
511
+ "iter_regex_ranges",
512
+ "project_markdown",
513
+ "project_context",
514
+ "project_plain",
515
+ "regex_ranges",
516
+ "slice_range_to_leaves",
517
+ ]
@@ -0,0 +1,176 @@
1
+ """#463 S3 — the closed line reader for Codex tool-output harness preambles.
2
+
3
+ Pure kernel: no I/O, no DB, no config, and nothing imported beyond ``re``.
4
+
5
+ Replaces ``_HARNESS_STATUS_RE``, which was one regex written against one assumed
6
+ shape. Executed against every retained tool output on a read-only copy of the
7
+ production store it examined 60,862 outputs, of which 39,942 carry the exact
8
+ ``Script completed`` / ``Script failed`` preamble it targets and **0 match**: it
9
+ required a blank line before ``Output:`` and end-of-string after it, while the
10
+ harness writes a single newline and a trailing newline. Five distinct preamble
11
+ grammars exist; that regex targeted one of them.
12
+
13
+ The replacement is a line reader over a CLOSED vocabulary rather than one regex
14
+ per grammar, deliberately (spec section 4.1). The defect being fixed is a
15
+ pattern written against an assumed shape, and the grammar set was mis-described
16
+ twice while the design was written. A closed vocabulary degrades to ``unknown``
17
+ on an arrangement it has not seen, instead of silently matching nothing.
18
+
19
+ Two rules bound what may be consumed:
20
+
21
+ * **Anchored, never searched.** Matching starts at position zero, because the
22
+ preamble is positionally guaranteed and a search would match a user's own
23
+ output somewhere in the middle of a real result.
24
+ * **Terminated by an ``Output:`` line.** All five observed grammars end that
25
+ way. Without the terminator a result that legitimately begins with
26
+ ``Exit code: 0`` would lose its first line; with it, the 10,177 outputs that
27
+ carry no preamble are untouched by construction.
28
+
29
+ Any malformed or over-limit field makes the WHOLE preamble unrecognized rather
30
+ than partially parsed, so a near-match degrades to today's behaviour instead of
31
+ stripping a line it did not understand.
32
+ """
33
+ from __future__ import annotations
34
+
35
+ import re
36
+
37
+ # At most this many lines and this many characters are examined before an
38
+ # `Output:` line. Reaching either limit first leaves the text untouched.
39
+ _MAX_PREAMBLE_LINES = 8
40
+ _MAX_PREAMBLE_CHARS = 512
41
+
42
+ _TERMINATOR = "Output:"
43
+
44
+ # Value syntaxes (spec section 4.1). A token is 1-64 characters from
45
+ # `[A-Za-z0-9_-]`; an integer is 1-10 digits with an optional leading `-`; a wall
46
+ # time is a decimal number with a unit from a closed set.
47
+ _TOKEN = r"[A-Za-z0-9_-]{1,64}"
48
+ _INT = r"-?\d{1,10}"
49
+ _UNSIGNED_INT = r"\d{1,10}"
50
+ _NUMBER = r"\d+(?:\.\d+)?"
51
+ _UNITS = r"seconds|second|ms"
52
+
53
+ _CHUNK_ID_RE = re.compile(rf"Chunk ID: ({_TOKEN})")
54
+ _EXIT_CODE_RE = re.compile(rf"Exit code: ({_INT})")
55
+ _WALL_TIME_RE = re.compile(rf"Wall time:? ({_NUMBER}) ({_UNITS})")
56
+ _PROCESS_EXITED_RE = re.compile(rf"Process exited with code ({_INT})")
57
+ _PROCESS_RUNNING_RE = re.compile(rf"Process running with session ID ({_TOKEN})")
58
+ _TOKEN_COUNT_RE = re.compile(rf"Original token count: ({_UNSIGNED_INT})")
59
+
60
+ _MILLISECOND_UNITS = frozenset({"ms"})
61
+
62
+
63
+ class _Reject(Exception):
64
+ """The line vocabulary refused a line; the whole preamble is unrecognized."""
65
+
66
+
67
+ def _wall_time_seconds(number: str, unit: str) -> float:
68
+ value = float(number)
69
+ return value / 1000.0 if unit in _MILLISECOND_UNITS else value
70
+
71
+
72
+ def _read_line(line: str, observed: dict) -> bool:
73
+ """Record one recognized preamble line, or raise ``_Reject``.
74
+
75
+ Returns True when the line carried a FIELD (so a bare terminator with nothing
76
+ before it can be refused) and False for a blank separator.
77
+ """
78
+ if line == "":
79
+ # A blank separator carries no field, so tolerating it cannot mis-parse
80
+ # one, and the terminator plus the two caps still bound what is consumed.
81
+ # Zero production outputs carry it, but the shipped `session-b-card-wire`
82
+ # fixture does, and a reader that refused it would silently regress that
83
+ # fixture's resolved status to `unknown`.
84
+ return False
85
+ if line in ("Script completed", "Script failed"):
86
+ observed["script"] = "completed" if line.endswith("completed") else "failed"
87
+ return True
88
+ match = _CHUNK_ID_RE.fullmatch(line)
89
+ if match is not None:
90
+ return True # deliberately not published (section 4.3)
91
+ match = _EXIT_CODE_RE.fullmatch(line)
92
+ if match is not None:
93
+ observed["exit_code"] = int(match.group(1))
94
+ return True
95
+ match = _WALL_TIME_RE.fullmatch(line)
96
+ if match is not None:
97
+ observed["wall_time_seconds"] = _wall_time_seconds(
98
+ match.group(1), match.group(2))
99
+ return True
100
+ match = _PROCESS_EXITED_RE.fullmatch(line)
101
+ if match is not None:
102
+ observed["exit_code"] = int(match.group(1))
103
+ return True
104
+ match = _PROCESS_RUNNING_RE.fullmatch(line)
105
+ if match is not None:
106
+ observed["session_announcement"] = match.group(1)
107
+ observed["running"] = True
108
+ return True
109
+ match = _TOKEN_COUNT_RE.fullmatch(line)
110
+ if match is not None:
111
+ return True
112
+ raise _Reject(line)
113
+
114
+
115
+ def _resolve_status(observed: dict) -> str:
116
+ """Spec section 4.2, in priority order.
117
+
118
+ ``running`` is explicitly NOT an error: 4,585 outputs are open sessions, and
119
+ treating them as failures would be worse than the present silence.
120
+ """
121
+ if observed.get("script") == "failed":
122
+ return "failed"
123
+ if "exit_code" in observed:
124
+ return "completed" if observed["exit_code"] == 0 else "failed"
125
+ if observed.get("running"):
126
+ return "running"
127
+ if observed.get("script") == "completed":
128
+ return "completed"
129
+ return "unknown"
130
+
131
+
132
+ def parse_harness_preamble(text: str) -> tuple[dict, str] | None:
133
+ """``(fields, remainder)`` for a recognized preamble, else ``None``.
134
+
135
+ ``fields`` carries ``status`` (one of ``completed``, ``failed``, ``running``,
136
+ ``unknown``), ``exit_code``, ``wall_time_seconds`` and
137
+ ``session_announcement``. ``remainder`` is ``text`` with the consumed run,
138
+ terminator included, removed.
139
+
140
+ ``None`` means the caller leaves the text exactly as it found it.
141
+
142
+ ``session_announcement`` carries the provider's raw session id for the
143
+ conversation-level session index ONLY. It is never published on a card:
144
+ the reader sees a conversation-local ordinal instead (spec section 4.3).
145
+ """
146
+ if not isinstance(text, str) or not text:
147
+ return None
148
+ observed: dict = {}
149
+ consumed = 0
150
+ field_lines = 0
151
+ for index, raw in enumerate(text.split("\n")):
152
+ line = raw[:-1] if raw.endswith("\r") else raw
153
+ consumed += len(raw) + 1
154
+ if index >= _MAX_PREAMBLE_LINES or consumed > _MAX_PREAMBLE_CHARS:
155
+ return None
156
+ if line == _TERMINATOR:
157
+ # A bare terminator with no field before it is not a preamble — it is
158
+ # a tool's own first line, and consuming it would delete real output.
159
+ if field_lines == 0:
160
+ return None
161
+ fields = {
162
+ "status": _resolve_status(observed),
163
+ "exit_code": observed.get("exit_code"),
164
+ "wall_time_seconds": observed.get("wall_time_seconds"),
165
+ "session_announcement": observed.get("session_announcement"),
166
+ }
167
+ return fields, text[consumed:]
168
+ try:
169
+ if _read_line(line, observed):
170
+ field_lines += 1
171
+ except _Reject:
172
+ return None
173
+ return None # ran out of text before a terminator
174
+
175
+
176
+ __all__ = ["parse_harness_preamble"]