cctally 1.91.0 → 1.92.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/CHANGELOG.md +39 -0
  2. package/bin/_cctally_cache.py +863 -74
  3. package/bin/_cctally_config.py +57 -0
  4. package/bin/_cctally_core.py +39 -8
  5. package/bin/_cctally_dashboard.py +146 -5
  6. package/bin/_cctally_dashboard_conversation.py +164 -18
  7. package/bin/_cctally_dashboard_envelope.py +2 -0
  8. package/bin/_cctally_db.py +372 -10
  9. package/bin/_cctally_doctor.py +18 -1
  10. package/bin/_cctally_journal.py +535 -13
  11. package/bin/_cctally_journal_repair.py +6 -0
  12. package/bin/_cctally_parser.py +6 -0
  13. package/bin/_cctally_quota.py +171 -55
  14. package/bin/_cctally_record.py +13 -1
  15. package/bin/_cctally_rederive.py +4 -0
  16. package/bin/_cctally_store.py +311 -6
  17. package/bin/_cctally_transcript.py +32 -2
  18. package/bin/_lib_cache_report.py +8 -3
  19. package/bin/_lib_codex_conversation.py +851 -81
  20. package/bin/_lib_codex_conversation_query.py +2005 -95
  21. package/bin/_lib_codex_find_projection.py +370 -0
  22. package/bin/_lib_codex_harness_preamble.py +176 -0
  23. package/bin/_lib_codex_hooks.py +5 -3
  24. package/bin/_lib_codex_js_scan.py +254 -0
  25. package/bin/_lib_codex_landmarks.py +309 -0
  26. package/bin/_lib_codex_title_clean.py +116 -0
  27. package/bin/_lib_conversation_dispatch.py +153 -21
  28. package/bin/_lib_conversation_watch.py +4 -2
  29. package/bin/_lib_doctor.py +64 -0
  30. package/bin/_lib_quota_alert_axes.py +31 -34
  31. package/bin/_lib_stats_damage.py +523 -0
  32. package/bin/cctally +5 -0
  33. package/dashboard/static/assets/index-BEzzJtUd.js +97 -0
  34. package/dashboard/static/assets/{index-Dwirao3Y.css → index-DnWdv8um.css} +1 -1
  35. package/dashboard/static/dashboard.html +2 -2
  36. package/package.json +7 -1
  37. package/dashboard/static/assets/index-CILAoEja.js +0 -90
@@ -0,0 +1,254 @@
1
+ """#463 S3 — a bounded lexical scanner for `tools.<name>(` in Codex programs.
2
+
3
+ Pure kernel: no I/O, nothing imported beyond ``re``.
4
+
5
+ Why this is a tokenizer and not a search. 17,777 uncarded `exec` calls, 51% of
6
+ the uncarded set, are JavaScript programs that declare constants, filter
7
+ `ALL_TOOLS`, await `Promise.all` and invoke `tools.exec_command`,
8
+ `tools.write_stdin` or other tools. The obvious way to card them is a textual
9
+ scan for `tools.<name>(`, and it is not acceptable:
10
+ `tests/test_codex_conversation_normalization.py` asserts that a call written
11
+ inside a string literal, inside a `//` comment and inside a regex literal each
12
+ decode to `None`, and a textual scanner would decode all three — fabricating
13
+ tool activity from a comment. A `complete: false` flag communicates omission; it
14
+ cannot make a false positive truthful.
15
+
16
+ So the source is walked ONCE through a small state machine that skips string
17
+ literals, template literals (including `${}` substitutions, which return to code
18
+ state), regex literals and both comment forms, and an invocation is recognized
19
+ only at a genuine token position whose member path is exactly `tools.<name>`.
20
+
21
+ It stays closed, bounded and non-executing. Lexing is not evaluation: nothing
22
+ here runs, resolves an identifier or follows a reference, so an aliased
23
+ `alias.exec_command(…)` and a computed `tools[key](…)` are both invisible by
24
+ construction, which is correct — they are not provably `tools.<name>`.
25
+
26
+ Anything that cannot be lexed cleanly returns ``None``, and the caller then
27
+ produces no card at all rather than a partial one.
28
+ """
29
+ from __future__ import annotations
30
+
31
+ import re
32
+
33
+ # The same ceiling `decode_tool_call_card` already applies to a harness body.
34
+ # Duplicated as a literal rather than imported, because importing the decoder
35
+ # module here would be circular — that module imports this one.
36
+ _PARSE_CAP = 1_000_000
37
+
38
+ _IDENT_START = re.compile(r"[A-Za-z_$]")
39
+ _IDENT_RE = re.compile(r"[A-Za-z_$][A-Za-z0-9_$]*")
40
+
41
+ # Anchored at a genuine `tools` token. Whitespace is tolerated around the dot and
42
+ # before the argument list because real programs wrap long chains across lines.
43
+ _MEMBER_CALL_RE = re.compile(
44
+ r"tools\s*\.\s*([A-Za-z_$][A-Za-z0-9_$]*)\s*\(")
45
+
46
+ # A `/` opens a regex literal only when the previous significant token is one of
47
+ # these, or when it is the first token in the source. After a value — an
48
+ # identifier, a number, a string, `)` or `]` — a `/` is division. Getting this
49
+ # wrong in the permissive direction would swallow the rest of the program into a
50
+ # regex and silently lose every invocation after it.
51
+ _REGEX_ALLOWED_PUNCTUATION = frozenset("(,=:[!&|?{};+-*%<>~^")
52
+ _REGEX_ALLOWED_KEYWORDS = frozenset({
53
+ "return", "typeof", "instanceof", "in", "of", "new", "delete", "void",
54
+ "case", "do", "else", "yield", "await", "throw",
55
+ })
56
+
57
+ # Sentinel `prev` value meaning "a completed value", after which `/` is division.
58
+ _VALUE = "\0value"
59
+
60
+ _DEFAULT = 0
61
+ _SINGLE = 1
62
+ _DOUBLE = 2
63
+ _TEMPLATE = 3
64
+ _LINE_COMMENT = 4
65
+ _BLOCK_COMMENT = 5
66
+ _REGEX = 6
67
+
68
+
69
+ def _regex_allowed(prev: str | None) -> bool:
70
+ if prev is None:
71
+ return True
72
+ if prev == _VALUE:
73
+ return False
74
+ if len(prev) == 1:
75
+ return prev in _REGEX_ALLOWED_PUNCTUATION
76
+ return prev in _REGEX_ALLOWED_KEYWORDS
77
+
78
+
79
+ def scan_tool_invocations(
80
+ source: str, *, limit: int,
81
+ ) -> list[tuple[str, int]] | None:
82
+ """``[(tool_name, arg_open_paren_index), …]``, or ``None``.
83
+
84
+ ``None`` means the source could not be lexed cleanly — an unterminated
85
+ string, template, regex or block comment, or a source past the parse cap —
86
+ and the caller must then produce no card.
87
+
88
+ An empty list means the source lexed but contains no `tools.<name>(` at a
89
+ genuine token position. A call inside a literal or a comment is therefore
90
+ indistinguishable from no call at all, which is the point.
91
+
92
+ ``limit`` bounds how many invocations are COLLECTED, not how far the scan
93
+ runs: the walk always continues to the end of the source, because an
94
+ unterminated literal after the limit still means the source could not be
95
+ lexed and no card may be built from it.
96
+ """
97
+ if not isinstance(source, str) or len(source) > _PARSE_CAP:
98
+ return None
99
+ found: list[tuple[str, int]] = []
100
+ state = _DEFAULT
101
+ prev: str | None = None
102
+ brace_depth = 0
103
+ template_returns: list[int] = []
104
+ in_char_class = False
105
+ index = 0
106
+ size = len(source)
107
+ while index < size:
108
+ char = source[index]
109
+ if state == _DEFAULT:
110
+ if char in " \t\r\n":
111
+ index += 1
112
+ continue
113
+ if char == "/" and index + 1 < size and source[index + 1] == "/":
114
+ state = _LINE_COMMENT
115
+ index += 2
116
+ continue
117
+ if char == "/" and index + 1 < size and source[index + 1] == "*":
118
+ state = _BLOCK_COMMENT
119
+ index += 2
120
+ continue
121
+ if char == "/":
122
+ if _regex_allowed(prev):
123
+ state = _REGEX
124
+ in_char_class = False
125
+ index += 1
126
+ continue
127
+ prev = "/"
128
+ index += 1
129
+ continue
130
+ if char == "'":
131
+ state = _SINGLE
132
+ index += 1
133
+ continue
134
+ if char == '"':
135
+ state = _DOUBLE
136
+ index += 1
137
+ continue
138
+ if char == "`":
139
+ state = _TEMPLATE
140
+ index += 1
141
+ continue
142
+ if char == "{":
143
+ brace_depth += 1
144
+ prev = "{"
145
+ index += 1
146
+ continue
147
+ if char == "}":
148
+ if template_returns and brace_depth == template_returns[-1]:
149
+ template_returns.pop()
150
+ state = _TEMPLATE
151
+ prev = _VALUE
152
+ index += 1
153
+ continue
154
+ brace_depth -= 1
155
+ prev = "}"
156
+ index += 1
157
+ continue
158
+ if _IDENT_START.match(char):
159
+ # A `tools` token is only a member path when nothing binds to it
160
+ # on the left. The test is the previous significant TOKEN, not
161
+ # the character immediately before `tools`: JavaScript permits
162
+ # whitespace and newlines around `.`, so `evil . tools.x` and
163
+ # `evil.\n tools.x` are member accesses on `evil` even though
164
+ # the character before `tools` is a space. `prev` is unchanged by
165
+ # the whitespace branch above, so it still holds the `.`.
166
+ if prev != ".":
167
+ match = _MEMBER_CALL_RE.match(source, index)
168
+ if match is not None:
169
+ if len(found) < limit:
170
+ found.append((match.group(1), match.end() - 1))
171
+ prev = "("
172
+ index = match.end()
173
+ continue
174
+ word = _IDENT_RE.match(source, index)
175
+ prev = word.group(0)
176
+ index = word.end()
177
+ continue
178
+ if char.isdigit():
179
+ index += 1
180
+ prev = _VALUE
181
+ continue
182
+ prev = char
183
+ index += 1
184
+ continue
185
+ if state in (_SINGLE, _DOUBLE):
186
+ if char == "\\":
187
+ index += 2
188
+ continue
189
+ if (state == _SINGLE and char == "'") or (state == _DOUBLE and char == '"'):
190
+ state = _DEFAULT
191
+ prev = _VALUE
192
+ index += 1
193
+ continue
194
+ if char == "\n":
195
+ return None # an unterminated ordinary string literal
196
+ index += 1
197
+ continue
198
+ if state == _TEMPLATE:
199
+ if char == "\\":
200
+ index += 2
201
+ continue
202
+ if char == "`":
203
+ state = _DEFAULT
204
+ prev = _VALUE
205
+ index += 1
206
+ continue
207
+ if char == "$" and index + 1 < size and source[index + 1] == "{":
208
+ template_returns.append(brace_depth)
209
+ state = _DEFAULT
210
+ prev = "{"
211
+ index += 2
212
+ continue
213
+ index += 1
214
+ continue
215
+ if state == _LINE_COMMENT:
216
+ if char == "\n":
217
+ state = _DEFAULT
218
+ index += 1
219
+ continue
220
+ if state == _BLOCK_COMMENT:
221
+ if char == "*" and index + 1 < size and source[index + 1] == "/":
222
+ state = _DEFAULT
223
+ index += 2
224
+ continue
225
+ index += 1
226
+ continue
227
+ # _REGEX
228
+ if char == "\\":
229
+ index += 2
230
+ continue
231
+ if char == "[":
232
+ in_char_class = True
233
+ index += 1
234
+ continue
235
+ if char == "]":
236
+ in_char_class = False
237
+ index += 1
238
+ continue
239
+ if char == "/" and not in_char_class:
240
+ state = _DEFAULT
241
+ prev = _VALUE
242
+ index += 1
243
+ continue
244
+ if char == "\n":
245
+ return None # an unterminated regex literal
246
+ index += 1
247
+ if state != _DEFAULT and state != _LINE_COMMENT:
248
+ return None # unterminated literal or block comment
249
+ if template_returns:
250
+ return None # unterminated `${` substitution
251
+ return found
252
+
253
+
254
+ __all__ = ["scan_tool_invocations"]
@@ -0,0 +1,309 @@
1
+ """Read-time landmark derivation for the Codex conversation outline (#463 S4).
2
+
3
+ Pure kernel: no SQLite, no I/O, and nothing from the query layer. It consumes
4
+ the S3 card decoders in ``_lib_codex_conversation`` and the S2 heading
5
+ decomposition in ``_lib_codex_reasoning_headings``, and returns small derived
6
+ facts — a failure verdict per outcome-bearing row, the authored reasoning
7
+ headings of a reasoning row, and the per-file touches of a patch event.
8
+
9
+ **Everything here is read-time only.** Nothing it produces may reach
10
+ ``codex_conversation_messages.detail_json``: those bytes feed ``_row_source_bytes``,
11
+ which drives ``PAGE_SOURCE_BYTE_BUDGET`` page boundaries, so persisting a derived
12
+ field would make two conversations with identical content paginate differently
13
+ according to which binary ingested them. The standing guard is
14
+ ``tests/test_codex_conversation_normalization.py::test_s3_no_read_time_enrichment_is_ever_persisted``.
15
+ """
16
+ from __future__ import annotations
17
+
18
+ import dataclasses
19
+ import re
20
+ from typing import Any, Iterable
21
+
22
+ from _lib_codex_reasoning_headings import decompose_reasoning_headings
23
+
24
+ # A physical row address — ``(source_path, line_offset)``, the key every
25
+ # conversation table is unique on and the one the payload pass joins by.
26
+ Position = tuple[str, int]
27
+
28
+ # The two status strings that mean the call failed. `decode_tool_output_card`
29
+ # already sets `is_error` from exactly this set, and `classify_tool_failure`
30
+ # below applies it to the patch and completion cards too so one definition
31
+ # covers every family.
32
+ #
33
+ # `"error"` is deliberately in this set. The CLIENT's `OUTCOME_STATUSES`
34
+ # excludes it, so a card carrying `status: "error"` on a call whose call-side
35
+ # card is not `terminal` collapses to `unknown` and is not flagged there. Spec
36
+ # §6.4 takes correctness over bug-compatibility: the server states one
37
+ # enumerated classification and Task 9 brings the client to it, because
38
+ # asserting "the server matches the client exactly" would have frozen a defect.
39
+ #
40
+ # `"running"` and `"unknown"` are NOT failures. `unknown` is a real state
41
+ # covering 17.6% of outputs — 4,585 of them are open sessions, measured — rather
42
+ # than an absence, and reporting it as a failure would invent errors that the
43
+ # reader could then not find.
44
+ _FAILED_STATUSES = frozenset({"failed", "error"})
45
+
46
+
47
+ def classify_tool_failure(cards: Any) -> bool:
48
+ """Did this tool call fail? One enumerated definition, spec §6.4.
49
+
50
+ ``cards`` maps a card family to the card the decoders produced for this
51
+ call — ``terminal_output``, ``patch``, and ``web``/``mcp`` whose value holds
52
+ the folded ``completion``. An absent family contributes nothing; an
53
+ unrecognized one is ignored rather than guessed at.
54
+
55
+ Each disjunct is named with the card it comes from:
56
+ """
57
+ if not isinstance(cards, dict):
58
+ return False
59
+
60
+ # 1. The result side of a terminal call — `decode_tool_output_card`, whose
61
+ # own `is_error` is the resolved five-grammar verdict.
62
+ output = cards.get("terminal_output")
63
+ if isinstance(output, dict):
64
+ if output.get("is_error") is True:
65
+ return True
66
+ # 2. The same card's resolved status, which covers a card built by a
67
+ # caller that did not carry `is_error` forward.
68
+ if output.get("status") in _FAILED_STATUSES:
69
+ return True
70
+
71
+ # 3. A patch, from either side — `decode_patch_event_card`'s `success`, and
72
+ # the standalone patch event's own `status`, which the client treats as
73
+ # an error independently of `success`.
74
+ patch = cards.get("patch")
75
+ if isinstance(patch, dict):
76
+ if patch.get("success") is False:
77
+ return True
78
+ if patch.get("status") in _FAILED_STATUSES:
79
+ return True
80
+
81
+ # 4/5. The web-search and MCP completions — `decode_secondary_event_card`,
82
+ # reached through the call-side card's folded `completion`.
83
+ for family in ("web", "mcp"):
84
+ holder = cards.get(family)
85
+ if not isinstance(holder, dict):
86
+ continue
87
+ completion = holder.get("completion")
88
+ if isinstance(completion, dict) and completion.get("status") in _FAILED_STATUSES:
89
+ return True
90
+
91
+ return False
92
+
93
+
94
+ def reasoning_heading_texts(payload: Any) -> list[str] | None:
95
+ """The authored headings of one reasoning payload's ``summary``, or ``None``.
96
+
97
+ Lifted verbatim out of ``_lib_codex_conversation_query._reasoning_headings``
98
+ so the outline and the reader decompose by ONE rule. S2's §4.6 precedent is
99
+ that the wire publishes every heading and the render layer dedupes; a second
100
+ copy of this parse is exactly how the two would drift.
101
+
102
+ Headings come from ``payload["summary"]`` ONLY. ``payload["content"]`` is the
103
+ body, which stays disclosure content and is never decomposed.
104
+
105
+ All-or-nothing. When the payload is absent, unreadable or malformed this
106
+ returns ``None`` and the caller omits the field entirely, so the client falls
107
+ back to the stored ``title``/``summary`` rendering. Decomposition never fails
108
+ the request and never partially populates.
109
+ """
110
+ if not isinstance(payload, dict):
111
+ return None
112
+ summary = payload.get("summary")
113
+ if not isinstance(summary, list) or not summary:
114
+ return None
115
+ entries: list[str] = []
116
+ for entry in summary:
117
+ if not isinstance(entry, dict):
118
+ return None
119
+ text = entry.get("text")
120
+ # Mirror `_join_content_texts`, which is what produced the stored
121
+ # summary: it keeps non-empty string `text` leaves and ignores the rest.
122
+ if text is None:
123
+ continue
124
+ if not isinstance(text, str):
125
+ return None
126
+ if text:
127
+ entries.append(text)
128
+ headings = decompose_reasoning_headings(entries)
129
+ return headings or None
130
+
131
+
132
+ # `@@ -<old start>[,<old count>] +<new start>[,<new count>] @@`. An omitted count
133
+ # means 1, which is what a single-line hunk emits.
134
+ _HUNK_HEADER_RE = re.compile(r"^@@ -\d+(?:,(\d+))? \+\d+(?:,(\d+))? @@")
135
+
136
+
137
+ def _diff_counts(diff: str) -> tuple[int, int] | None:
138
+ """Added and removed line counts of one unified diff, or ``None``.
139
+
140
+ Counted INSIDE the ``@@`` hunks, never by prefix over the whole text. A
141
+ prefix filter cannot separate a file header from content that looks like
142
+ one: removing the SQL comment ``-- legacy`` renders ``--- legacy`` and
143
+ adding ``++i;`` renders ``+++i;``, so excluding every line that starts with
144
+ ``---`` or ``+++`` drops real changed lines and silently understates the
145
+ diff — the same undercount §4.5 exists to prevent, reached by a different
146
+ route. A header can only appear before a hunk opens, and the hunk header
147
+ states how many old-side and new-side lines follow, so inside a hunk nothing
148
+ has to be recognised by its prefix alone.
149
+
150
+ ``None`` when the text carries no hunk at all. Such a string is not a
151
+ unified diff, and counting its ``+``/``-`` prefixed lines would be the same
152
+ guess; the caller reports an undetermined count rather than a wrong one.
153
+ """
154
+ added = removed = 0
155
+ old_left = new_left = 0
156
+ in_hunk = saw_hunk = False
157
+ for line in diff.split("\n"):
158
+ if not in_hunk:
159
+ header = _HUNK_HEADER_RE.match(line)
160
+ if header is None:
161
+ continue
162
+ saw_hunk = True
163
+ old_left = int(header.group(1)) if header.group(1) is not None else 1
164
+ new_left = int(header.group(2)) if header.group(2) is not None else 1
165
+ in_hunk = old_left > 0 or new_left > 0
166
+ continue
167
+ if line.startswith("\\"):
168
+ # "" belongs to neither side's line count.
169
+ continue
170
+ if line.startswith("+"):
171
+ added += 1
172
+ new_left -= 1
173
+ elif line.startswith("-"):
174
+ removed += 1
175
+ old_left -= 1
176
+ else:
177
+ old_left -= 1
178
+ new_left -= 1
179
+ if old_left <= 0 and new_left <= 0:
180
+ in_hunk = False
181
+ return (added, removed) if saw_hunk else None
182
+
183
+
184
+ def patch_file_touches(payload: Any) -> list[dict]:
185
+ """Per-file touches of one ``patch_apply_end`` payload, in provider order.
186
+
187
+ Each entry is ``{"path", "op", "added", "removed"}``; ``added``/``removed``
188
+ are ``None`` when the count cannot be determined, never 0, because 0 is a
189
+ claim and an undetermined count is not.
190
+
191
+ **Counted from the UNBOUNDED raw ``changes`` entry, before card allocation**
192
+ (spec §4.5). ``decode_patch_event_card`` shares one 16,000-character budget
193
+ across stdout, stderr and every file and caps the file list at 128, and S3
194
+ measured 85 real change entries over that budget — so counting `+`/`-` lines
195
+ out of a served ``unified_diff`` silently undercounts, and a wholly clipped
196
+ diff would read as no change at all despite complete evidence sitting in the
197
+ payload. Presentation cards and whole-session statistics have different
198
+ bounds and must not share one.
199
+
200
+ Only the DICT-shaped ``changes`` is a real production source. S3 measured
201
+ all 4,829 patch events arriving as a dict keyed by file path; #489 taught
202
+ the stored file-search projection to consume that shape too. The legacy
203
+ list shape stays supported here and at ingest so a provider change cannot
204
+ silently produce an empty file list.
205
+ """
206
+ if not isinstance(payload, dict) or payload.get("type") != "patch_apply_end":
207
+ return []
208
+ changes = payload.get("changes")
209
+ touches: list[dict] = []
210
+ if isinstance(changes, dict):
211
+ items: Iterable[tuple[Any, Any]] = changes.items()
212
+ elif isinstance(changes, list):
213
+ # The path lives on the entry in this shape, and the kind is `status`
214
+ # rather than `type` — the two shapes disagree on both.
215
+ items = (
216
+ (change.get("path"), change) for change in changes
217
+ if isinstance(change, dict)
218
+ )
219
+ else:
220
+ return []
221
+ for path, change in items:
222
+ if not isinstance(path, str) or not isinstance(change, dict):
223
+ continue
224
+ kind = change.get("type")
225
+ if not isinstance(kind, str):
226
+ kind = change.get("status")
227
+ added = removed = None
228
+ diff = change.get("unified_diff")
229
+ if isinstance(diff, str):
230
+ counted = _diff_counts(diff)
231
+ if counted is not None:
232
+ added, removed = counted
233
+ elif kind in {"add", "delete"} and isinstance(change.get("content"), str):
234
+ lines = change["content"].split("\n")
235
+ if lines and lines[-1] == "":
236
+ lines.pop()
237
+ added, removed = ((len(lines), 0) if kind == "add"
238
+ else (0, len(lines)))
239
+ touches.append({
240
+ "path": path,
241
+ "op": kind if isinstance(kind, str) else None,
242
+ "added": added,
243
+ "removed": removed,
244
+ })
245
+ return touches
246
+
247
+
248
+ def fold_owner_by_position(
249
+ groups: Iterable[Iterable[Position]],
250
+ ) -> dict[Position, Position]:
251
+ """Map every row position to the position of its fold group's first row.
252
+
253
+ That first row is the ``tool_call`` a failing ``tool_output`` or completion
254
+ event belongs to. ``_fold_groups_for_item`` computes this membership
255
+ payload-free and ``_build_segment_index`` discarded it, so the outline had no
256
+ way to say WHICH call a failure belongs to — only that the segment contained
257
+ one.
258
+
259
+ Grouping is a conservative SUPERSET of what the block builder actually folds,
260
+ which is the safe direction here: attributing a failure to the call it was
261
+ grouped with can at worst name a call that did not fold, and never splits a
262
+ call from an outcome that did.
263
+ """
264
+ owners: dict[Position, Position] = {}
265
+ for group in groups:
266
+ head: Position | None = None
267
+ for position in group:
268
+ if head is None:
269
+ head = position
270
+ owners[position] = head
271
+ return owners
272
+
273
+
274
+ @dataclasses.dataclass
275
+ class EventDerivation:
276
+ """Everything the outline derives from one scoped pass over the raw payloads.
277
+
278
+ Keyed by physical position throughout, because that is the only identity the
279
+ message table and the event table share, and because the item and segment
280
+ keys a landmark ultimately carries are minted later from the rows these
281
+ positions name.
282
+
283
+ **Mutable, and filled in by the pass as it streams.** The maps are complete
284
+ only once ``_derive_outline_events`` has returned; a partly-filled instance
285
+ is never handed to a consumer. The dataclass is deliberately not frozen:
286
+ freezing rebinding of the three attributes while the dicts they name are
287
+ mutated in place would advertise an immutability this object does not have.
288
+
289
+ ``errors_by_position`` answers "did this call fail" for the outcome-bearing
290
+ rows the pass classified — a ``tool_output``, or a ``patch_apply_end`` /
291
+ ``web_search_end`` / ``mcp_tool_call_end`` event. **Absence is a third state,
292
+ not ``False``.** A position is absent when the scope did not select it, and
293
+ equally when it was selected and could not be classified — its event row is
294
+ gone, its payload does not parse, or the decoder declined the shape. The
295
+ caller cannot tell those apart from this map and must not read absence as a
296
+ claim that the call succeeded: a count derived from it is a count of
297
+ failures FOUND, and a route that needs to distinguish "looked and found
298
+ none" from "could not look" compares these keys against the in-scope set it
299
+ supplied.
300
+ """
301
+
302
+ errors_by_position: dict[Position, bool] = dataclasses.field(default_factory=dict)
303
+ patch_files_by_position: dict[Position, list[dict]] = dataclasses.field(
304
+ default_factory=dict)
305
+ headings_by_position: dict[Position, list[str]] = dataclasses.field(
306
+ default_factory=dict)
307
+
308
+ def failing_positions(self) -> set[Position]:
309
+ return {pos for pos, failed in self.errors_by_position.items() if failed}
@@ -0,0 +1,116 @@
1
+ """Read-time cleaning of harness markup out of a Codex title (#463 S4 §5).
2
+
3
+ Pure kernel: one entry point, ``clean_codex_title(text) -> str``, and a CLOSED
4
+ allowlist of the grammars a census of the real store actually found.
5
+
6
+ **Read time, not ingest (D5).** The title is stored —
7
+ ``codex_conversation_rollups.title`` is written at ingest and ``_rollup_fields``
8
+ returns it on its fast path — so repairing ``derive_title`` would heal nothing
9
+ for the conversations that are already wrong. Cleaning on the read path heals
10
+ all history with no migration and no reingest flag.
11
+
12
+ **The allowlist is a measurement, not a guess (§5.4).** Over the 438 stored
13
+ Codex rollup titles in the production store on 2026-08-04:
14
+
15
+ =========================================== ===== ===========
16
+ grammar count disposition
17
+ =========================================== ===== ===========
18
+ ``[$name](<abs path>/SKILL.md) <rest>`` 165 unwrap
19
+ ``<command-name>…</command-name> …`` 40 see below
20
+ ``<recommended_plugins> …`` 6 strip
21
+ ``<command-message>…</command-message> …`` 1 see below
22
+ =========================================== ===== ===========
23
+
24
+ Nothing else occurred. No title carried a tag anywhere but at its head (0 of
25
+ 438), and the two ``CODEX_TITLE_SKIP_PREFIXES`` wrappers
26
+ (``<environment_context>``, ``<user_instructions>``) appeared 0 times, because
27
+ they are skipped at ingest by a different mechanism that stays where it is
28
+ (§5.3).
29
+
30
+ Within the command wrapper the three tags are dispositioned separately, from
31
+ what their content actually looks like: ``command-name`` is the slash command
32
+ and is STRIPPED, while ``command-message`` and ``command-args`` carry the human
33
+ text and are UNWRAPPED. On the corpus that turns
34
+ ``<command-name>/model</command-name> <command-message>model</command-message>
35
+ <command-args>fable</command-args>`` into ``model fable``.
36
+
37
+ ``recommended_plugins`` never closes in the data: titles are capped at 120
38
+ characters, so the stored value is the head of a plugin catalogue. Stripping it
39
+ leaves nothing, and ``_display_chain`` falls through to the project label and
40
+ then to a short native thread id, which the chain already does.
41
+
42
+ **Closed, and deliberately so.** A general tag stripper would eat user-authored
43
+ angle brackets in a title. An unrecognized construct passes through BYTE for
44
+ byte — the function returns its input unchanged when no rule fires, so the 226
45
+ titles the census found clean, and every prose label this is applied to, cannot
46
+ move.
47
+ """
48
+ from __future__ import annotations
49
+
50
+ import re
51
+
52
+ # `[$skill-name](/abs/path/to/SKILL.md)` — the Codex skill invocation. The
53
+ # prompt text after it is real, and the link target is a private filesystem path
54
+ # that leaks into every title surface. Byte-identical to the client's
55
+ # `cleanQualifiedTitle` regex, so the two agree on the same input and applying
56
+ # both is a no-op.
57
+ #
58
+ # NO trailing lookahead. The first version required whitespace or end of string
59
+ # after the closing paren, so `…/SKILL.md)Task B of issue #450.` — prompt text
60
+ # written straight against the paren — did not match and the whole link,
61
+ # absolute path included, reached the reader header and the outline rail. Two of
62
+ # 300 served titles in the test store carry that form. Nothing is lost by
63
+ # dropping the lookahead: the pattern is head-anchored (`pattern.match`) and its
64
+ # target is the literal `/SKILL.md)`, so it cannot start matching mid-title or
65
+ # consume any other Markdown link.
66
+ _SKILL_LINK_RE = re.compile(
67
+ r"\[((?:\$)[^\]\r\n]+)\]\([^)\r\n]*/SKILL\.md\)")
68
+
69
+ _STRIP, _UNWRAP = "strip", "unwrap"
70
+
71
+ # Head-anchored, in match order. Each entry is (pattern, disposition, group) —
72
+ # `group` names the capture an `unwrap` keeps.
73
+ _GRAMMARS: tuple[tuple[re.Pattern, str, int], ...] = (
74
+ (_SKILL_LINK_RE, _UNWRAP, 1),
75
+ (re.compile(r"<command-name>(.*?)</command-name>", re.S), _STRIP, 1),
76
+ (re.compile(r"<command-message>(.*?)</command-message>", re.S), _UNWRAP, 1),
77
+ (re.compile(r"<command-args>(.*?)</command-args>", re.S), _UNWRAP, 1),
78
+ # The closing form first: alternation is ordered, and the open-ended arm
79
+ # would otherwise swallow a closed construct's tail.
80
+ (re.compile(r"<recommended_plugins>.*?</recommended_plugins>"
81
+ r"|<recommended_plugins>.*", re.S), _STRIP, 0),
82
+ )
83
+
84
+
85
+ def clean_codex_title(text) -> str:
86
+ """The title with recognized leading harness markup removed or unwrapped.
87
+
88
+ Returns the input unchanged when no grammar in the allowlist matches its
89
+ head, including for a non-string or empty input, which keeps every
90
+ untouched title and every prose label byte-stable.
91
+
92
+ A construct that strips to nothing yields ``""``, and the caller's fallback
93
+ chain takes over (§5.3).
94
+ """
95
+ if not isinstance(text, str) or not text:
96
+ return text if isinstance(text, str) else ""
97
+ kept: list[str] = []
98
+ rest = text
99
+ matched = False
100
+ while rest:
101
+ head = rest.lstrip()
102
+ for pattern, disposition, group in _GRAMMARS:
103
+ found = pattern.match(head)
104
+ if found is None:
105
+ continue
106
+ matched = True
107
+ if disposition == _UNWRAP:
108
+ kept.append(found.group(group))
109
+ rest = head[found.end():]
110
+ break
111
+ else:
112
+ kept.append(head)
113
+ break
114
+ if not matched:
115
+ return text
116
+ return " ".join(" ".join(part.split()) for part in kept if part.strip()).strip()