cctally 1.91.0 → 1.92.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +39 -0
- package/bin/_cctally_cache.py +863 -74
- package/bin/_cctally_config.py +57 -0
- package/bin/_cctally_core.py +39 -8
- package/bin/_cctally_dashboard.py +146 -5
- package/bin/_cctally_dashboard_conversation.py +164 -18
- package/bin/_cctally_dashboard_envelope.py +2 -0
- package/bin/_cctally_db.py +372 -10
- package/bin/_cctally_doctor.py +18 -1
- package/bin/_cctally_journal.py +535 -13
- package/bin/_cctally_journal_repair.py +6 -0
- package/bin/_cctally_parser.py +6 -0
- package/bin/_cctally_quota.py +171 -55
- package/bin/_cctally_record.py +13 -1
- package/bin/_cctally_rederive.py +4 -0
- package/bin/_cctally_store.py +311 -6
- package/bin/_cctally_transcript.py +32 -2
- package/bin/_lib_cache_report.py +8 -3
- package/bin/_lib_codex_conversation.py +851 -81
- package/bin/_lib_codex_conversation_query.py +2005 -95
- package/bin/_lib_codex_find_projection.py +370 -0
- package/bin/_lib_codex_harness_preamble.py +176 -0
- package/bin/_lib_codex_hooks.py +5 -3
- package/bin/_lib_codex_js_scan.py +254 -0
- package/bin/_lib_codex_landmarks.py +309 -0
- package/bin/_lib_codex_title_clean.py +116 -0
- package/bin/_lib_conversation_dispatch.py +153 -21
- package/bin/_lib_conversation_watch.py +4 -2
- package/bin/_lib_doctor.py +64 -0
- package/bin/_lib_quota_alert_axes.py +31 -34
- package/bin/_lib_stats_damage.py +523 -0
- package/bin/cctally +5 -0
- package/dashboard/static/assets/index-BEzzJtUd.js +97 -0
- package/dashboard/static/assets/{index-Dwirao3Y.css → index-DnWdv8um.css} +1 -1
- package/dashboard/static/dashboard.html +2 -2
- package/package.json +7 -1
- package/dashboard/static/assets/index-CILAoEja.js +0 -90
|
@@ -0,0 +1,254 @@
|
|
|
1
|
+
"""#463 S3 — a bounded lexical scanner for `tools.<name>(` in Codex programs.
|
|
2
|
+
|
|
3
|
+
Pure kernel: no I/O, nothing imported beyond ``re``.
|
|
4
|
+
|
|
5
|
+
Why this is a tokenizer and not a search. 17,777 uncarded `exec` calls, 51% of
|
|
6
|
+
the uncarded set, are JavaScript programs that declare constants, filter
|
|
7
|
+
`ALL_TOOLS`, await `Promise.all` and invoke `tools.exec_command`,
|
|
8
|
+
`tools.write_stdin` or other tools. The obvious way to card them is a textual
|
|
9
|
+
scan for `tools.<name>(`, and it is not acceptable:
|
|
10
|
+
`tests/test_codex_conversation_normalization.py` asserts that a call written
|
|
11
|
+
inside a string literal, inside a `//` comment and inside a regex literal each
|
|
12
|
+
decode to `None`, and a textual scanner would decode all three — fabricating
|
|
13
|
+
tool activity from a comment. A `complete: false` flag communicates omission; it
|
|
14
|
+
cannot make a false positive truthful.
|
|
15
|
+
|
|
16
|
+
So the source is walked ONCE through a small state machine that skips string
|
|
17
|
+
literals, template literals (including `${}` substitutions, which return to code
|
|
18
|
+
state), regex literals and both comment forms, and an invocation is recognized
|
|
19
|
+
only at a genuine token position whose member path is exactly `tools.<name>`.
|
|
20
|
+
|
|
21
|
+
It stays closed, bounded and non-executing. Lexing is not evaluation: nothing
|
|
22
|
+
here runs, resolves an identifier or follows a reference, so an aliased
|
|
23
|
+
`alias.exec_command(…)` and a computed `tools[key](…)` are both invisible by
|
|
24
|
+
construction, which is correct — they are not provably `tools.<name>`.
|
|
25
|
+
|
|
26
|
+
Anything that cannot be lexed cleanly returns ``None``, and the caller then
|
|
27
|
+
produces no card at all rather than a partial one.
|
|
28
|
+
"""
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
import re
|
|
32
|
+
|
|
33
|
+
# The same ceiling `decode_tool_call_card` already applies to a harness body.
|
|
34
|
+
# Duplicated as a literal rather than imported, because importing the decoder
|
|
35
|
+
# module here would be circular — that module imports this one.
|
|
36
|
+
_PARSE_CAP = 1_000_000
|
|
37
|
+
|
|
38
|
+
_IDENT_START = re.compile(r"[A-Za-z_$]")
|
|
39
|
+
_IDENT_RE = re.compile(r"[A-Za-z_$][A-Za-z0-9_$]*")
|
|
40
|
+
|
|
41
|
+
# Anchored at a genuine `tools` token. Whitespace is tolerated around the dot and
|
|
42
|
+
# before the argument list because real programs wrap long chains across lines.
|
|
43
|
+
_MEMBER_CALL_RE = re.compile(
|
|
44
|
+
r"tools\s*\.\s*([A-Za-z_$][A-Za-z0-9_$]*)\s*\(")
|
|
45
|
+
|
|
46
|
+
# A `/` opens a regex literal only when the previous significant token is one of
|
|
47
|
+
# these, or when it is the first token in the source. After a value — an
|
|
48
|
+
# identifier, a number, a string, `)` or `]` — a `/` is division. Getting this
|
|
49
|
+
# wrong in the permissive direction would swallow the rest of the program into a
|
|
50
|
+
# regex and silently lose every invocation after it.
|
|
51
|
+
_REGEX_ALLOWED_PUNCTUATION = frozenset("(,=:[!&|?{};+-*%<>~^")
|
|
52
|
+
_REGEX_ALLOWED_KEYWORDS = frozenset({
|
|
53
|
+
"return", "typeof", "instanceof", "in", "of", "new", "delete", "void",
|
|
54
|
+
"case", "do", "else", "yield", "await", "throw",
|
|
55
|
+
})
|
|
56
|
+
|
|
57
|
+
# Sentinel `prev` value meaning "a completed value", after which `/` is division.
|
|
58
|
+
_VALUE = "\0value"
|
|
59
|
+
|
|
60
|
+
_DEFAULT = 0
|
|
61
|
+
_SINGLE = 1
|
|
62
|
+
_DOUBLE = 2
|
|
63
|
+
_TEMPLATE = 3
|
|
64
|
+
_LINE_COMMENT = 4
|
|
65
|
+
_BLOCK_COMMENT = 5
|
|
66
|
+
_REGEX = 6
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _regex_allowed(prev: str | None) -> bool:
|
|
70
|
+
if prev is None:
|
|
71
|
+
return True
|
|
72
|
+
if prev == _VALUE:
|
|
73
|
+
return False
|
|
74
|
+
if len(prev) == 1:
|
|
75
|
+
return prev in _REGEX_ALLOWED_PUNCTUATION
|
|
76
|
+
return prev in _REGEX_ALLOWED_KEYWORDS
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def scan_tool_invocations(
|
|
80
|
+
source: str, *, limit: int,
|
|
81
|
+
) -> list[tuple[str, int]] | None:
|
|
82
|
+
"""``[(tool_name, arg_open_paren_index), …]``, or ``None``.
|
|
83
|
+
|
|
84
|
+
``None`` means the source could not be lexed cleanly — an unterminated
|
|
85
|
+
string, template, regex or block comment, or a source past the parse cap —
|
|
86
|
+
and the caller must then produce no card.
|
|
87
|
+
|
|
88
|
+
An empty list means the source lexed but contains no `tools.<name>(` at a
|
|
89
|
+
genuine token position. A call inside a literal or a comment is therefore
|
|
90
|
+
indistinguishable from no call at all, which is the point.
|
|
91
|
+
|
|
92
|
+
``limit`` bounds how many invocations are COLLECTED, not how far the scan
|
|
93
|
+
runs: the walk always continues to the end of the source, because an
|
|
94
|
+
unterminated literal after the limit still means the source could not be
|
|
95
|
+
lexed and no card may be built from it.
|
|
96
|
+
"""
|
|
97
|
+
if not isinstance(source, str) or len(source) > _PARSE_CAP:
|
|
98
|
+
return None
|
|
99
|
+
found: list[tuple[str, int]] = []
|
|
100
|
+
state = _DEFAULT
|
|
101
|
+
prev: str | None = None
|
|
102
|
+
brace_depth = 0
|
|
103
|
+
template_returns: list[int] = []
|
|
104
|
+
in_char_class = False
|
|
105
|
+
index = 0
|
|
106
|
+
size = len(source)
|
|
107
|
+
while index < size:
|
|
108
|
+
char = source[index]
|
|
109
|
+
if state == _DEFAULT:
|
|
110
|
+
if char in " \t\r\n":
|
|
111
|
+
index += 1
|
|
112
|
+
continue
|
|
113
|
+
if char == "/" and index + 1 < size and source[index + 1] == "/":
|
|
114
|
+
state = _LINE_COMMENT
|
|
115
|
+
index += 2
|
|
116
|
+
continue
|
|
117
|
+
if char == "/" and index + 1 < size and source[index + 1] == "*":
|
|
118
|
+
state = _BLOCK_COMMENT
|
|
119
|
+
index += 2
|
|
120
|
+
continue
|
|
121
|
+
if char == "/":
|
|
122
|
+
if _regex_allowed(prev):
|
|
123
|
+
state = _REGEX
|
|
124
|
+
in_char_class = False
|
|
125
|
+
index += 1
|
|
126
|
+
continue
|
|
127
|
+
prev = "/"
|
|
128
|
+
index += 1
|
|
129
|
+
continue
|
|
130
|
+
if char == "'":
|
|
131
|
+
state = _SINGLE
|
|
132
|
+
index += 1
|
|
133
|
+
continue
|
|
134
|
+
if char == '"':
|
|
135
|
+
state = _DOUBLE
|
|
136
|
+
index += 1
|
|
137
|
+
continue
|
|
138
|
+
if char == "`":
|
|
139
|
+
state = _TEMPLATE
|
|
140
|
+
index += 1
|
|
141
|
+
continue
|
|
142
|
+
if char == "{":
|
|
143
|
+
brace_depth += 1
|
|
144
|
+
prev = "{"
|
|
145
|
+
index += 1
|
|
146
|
+
continue
|
|
147
|
+
if char == "}":
|
|
148
|
+
if template_returns and brace_depth == template_returns[-1]:
|
|
149
|
+
template_returns.pop()
|
|
150
|
+
state = _TEMPLATE
|
|
151
|
+
prev = _VALUE
|
|
152
|
+
index += 1
|
|
153
|
+
continue
|
|
154
|
+
brace_depth -= 1
|
|
155
|
+
prev = "}"
|
|
156
|
+
index += 1
|
|
157
|
+
continue
|
|
158
|
+
if _IDENT_START.match(char):
|
|
159
|
+
# A `tools` token is only a member path when nothing binds to it
|
|
160
|
+
# on the left. The test is the previous significant TOKEN, not
|
|
161
|
+
# the character immediately before `tools`: JavaScript permits
|
|
162
|
+
# whitespace and newlines around `.`, so `evil . tools.x` and
|
|
163
|
+
# `evil.\n tools.x` are member accesses on `evil` even though
|
|
164
|
+
# the character before `tools` is a space. `prev` is unchanged by
|
|
165
|
+
# the whitespace branch above, so it still holds the `.`.
|
|
166
|
+
if prev != ".":
|
|
167
|
+
match = _MEMBER_CALL_RE.match(source, index)
|
|
168
|
+
if match is not None:
|
|
169
|
+
if len(found) < limit:
|
|
170
|
+
found.append((match.group(1), match.end() - 1))
|
|
171
|
+
prev = "("
|
|
172
|
+
index = match.end()
|
|
173
|
+
continue
|
|
174
|
+
word = _IDENT_RE.match(source, index)
|
|
175
|
+
prev = word.group(0)
|
|
176
|
+
index = word.end()
|
|
177
|
+
continue
|
|
178
|
+
if char.isdigit():
|
|
179
|
+
index += 1
|
|
180
|
+
prev = _VALUE
|
|
181
|
+
continue
|
|
182
|
+
prev = char
|
|
183
|
+
index += 1
|
|
184
|
+
continue
|
|
185
|
+
if state in (_SINGLE, _DOUBLE):
|
|
186
|
+
if char == "\\":
|
|
187
|
+
index += 2
|
|
188
|
+
continue
|
|
189
|
+
if (state == _SINGLE and char == "'") or (state == _DOUBLE and char == '"'):
|
|
190
|
+
state = _DEFAULT
|
|
191
|
+
prev = _VALUE
|
|
192
|
+
index += 1
|
|
193
|
+
continue
|
|
194
|
+
if char == "\n":
|
|
195
|
+
return None # an unterminated ordinary string literal
|
|
196
|
+
index += 1
|
|
197
|
+
continue
|
|
198
|
+
if state == _TEMPLATE:
|
|
199
|
+
if char == "\\":
|
|
200
|
+
index += 2
|
|
201
|
+
continue
|
|
202
|
+
if char == "`":
|
|
203
|
+
state = _DEFAULT
|
|
204
|
+
prev = _VALUE
|
|
205
|
+
index += 1
|
|
206
|
+
continue
|
|
207
|
+
if char == "$" and index + 1 < size and source[index + 1] == "{":
|
|
208
|
+
template_returns.append(brace_depth)
|
|
209
|
+
state = _DEFAULT
|
|
210
|
+
prev = "{"
|
|
211
|
+
index += 2
|
|
212
|
+
continue
|
|
213
|
+
index += 1
|
|
214
|
+
continue
|
|
215
|
+
if state == _LINE_COMMENT:
|
|
216
|
+
if char == "\n":
|
|
217
|
+
state = _DEFAULT
|
|
218
|
+
index += 1
|
|
219
|
+
continue
|
|
220
|
+
if state == _BLOCK_COMMENT:
|
|
221
|
+
if char == "*" and index + 1 < size and source[index + 1] == "/":
|
|
222
|
+
state = _DEFAULT
|
|
223
|
+
index += 2
|
|
224
|
+
continue
|
|
225
|
+
index += 1
|
|
226
|
+
continue
|
|
227
|
+
# _REGEX
|
|
228
|
+
if char == "\\":
|
|
229
|
+
index += 2
|
|
230
|
+
continue
|
|
231
|
+
if char == "[":
|
|
232
|
+
in_char_class = True
|
|
233
|
+
index += 1
|
|
234
|
+
continue
|
|
235
|
+
if char == "]":
|
|
236
|
+
in_char_class = False
|
|
237
|
+
index += 1
|
|
238
|
+
continue
|
|
239
|
+
if char == "/" and not in_char_class:
|
|
240
|
+
state = _DEFAULT
|
|
241
|
+
prev = _VALUE
|
|
242
|
+
index += 1
|
|
243
|
+
continue
|
|
244
|
+
if char == "\n":
|
|
245
|
+
return None # an unterminated regex literal
|
|
246
|
+
index += 1
|
|
247
|
+
if state != _DEFAULT and state != _LINE_COMMENT:
|
|
248
|
+
return None # unterminated literal or block comment
|
|
249
|
+
if template_returns:
|
|
250
|
+
return None # unterminated `${` substitution
|
|
251
|
+
return found
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
__all__ = ["scan_tool_invocations"]
|
|
@@ -0,0 +1,309 @@
|
|
|
1
|
+
"""Read-time landmark derivation for the Codex conversation outline (#463 S4).
|
|
2
|
+
|
|
3
|
+
Pure kernel: no SQLite, no I/O, and nothing from the query layer. It consumes
|
|
4
|
+
the S3 card decoders in ``_lib_codex_conversation`` and the S2 heading
|
|
5
|
+
decomposition in ``_lib_codex_reasoning_headings``, and returns small derived
|
|
6
|
+
facts — a failure verdict per outcome-bearing row, the authored reasoning
|
|
7
|
+
headings of a reasoning row, and the per-file touches of a patch event.
|
|
8
|
+
|
|
9
|
+
**Everything here is read-time only.** Nothing it produces may reach
|
|
10
|
+
``codex_conversation_messages.detail_json``: those bytes feed ``_row_source_bytes``,
|
|
11
|
+
which drives ``PAGE_SOURCE_BYTE_BUDGET`` page boundaries, so persisting a derived
|
|
12
|
+
field would make two conversations with identical content paginate differently
|
|
13
|
+
according to which binary ingested them. The standing guard is
|
|
14
|
+
``tests/test_codex_conversation_normalization.py::test_s3_no_read_time_enrichment_is_ever_persisted``.
|
|
15
|
+
"""
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import dataclasses
|
|
19
|
+
import re
|
|
20
|
+
from typing import Any, Iterable
|
|
21
|
+
|
|
22
|
+
from _lib_codex_reasoning_headings import decompose_reasoning_headings
|
|
23
|
+
|
|
24
|
+
# A physical row address — ``(source_path, line_offset)``, the key every
|
|
25
|
+
# conversation table is unique on and the one the payload pass joins by.
|
|
26
|
+
Position = tuple[str, int]
|
|
27
|
+
|
|
28
|
+
# The two status strings that mean the call failed. `decode_tool_output_card`
|
|
29
|
+
# already sets `is_error` from exactly this set, and `classify_tool_failure`
|
|
30
|
+
# below applies it to the patch and completion cards too so one definition
|
|
31
|
+
# covers every family.
|
|
32
|
+
#
|
|
33
|
+
# `"error"` is deliberately in this set. The CLIENT's `OUTCOME_STATUSES`
|
|
34
|
+
# excludes it, so a card carrying `status: "error"` on a call whose call-side
|
|
35
|
+
# card is not `terminal` collapses to `unknown` and is not flagged there. Spec
|
|
36
|
+
# §6.4 takes correctness over bug-compatibility: the server states one
|
|
37
|
+
# enumerated classification and Task 9 brings the client to it, because
|
|
38
|
+
# asserting "the server matches the client exactly" would have frozen a defect.
|
|
39
|
+
#
|
|
40
|
+
# `"running"` and `"unknown"` are NOT failures. `unknown` is a real state
|
|
41
|
+
# covering 17.6% of outputs — 4,585 of them are open sessions, measured — rather
|
|
42
|
+
# than an absence, and reporting it as a failure would invent errors that the
|
|
43
|
+
# reader could then not find.
|
|
44
|
+
_FAILED_STATUSES = frozenset({"failed", "error"})
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def classify_tool_failure(cards: Any) -> bool:
|
|
48
|
+
"""Did this tool call fail? One enumerated definition, spec §6.4.
|
|
49
|
+
|
|
50
|
+
``cards`` maps a card family to the card the decoders produced for this
|
|
51
|
+
call — ``terminal_output``, ``patch``, and ``web``/``mcp`` whose value holds
|
|
52
|
+
the folded ``completion``. An absent family contributes nothing; an
|
|
53
|
+
unrecognized one is ignored rather than guessed at.
|
|
54
|
+
|
|
55
|
+
Each disjunct is named with the card it comes from:
|
|
56
|
+
"""
|
|
57
|
+
if not isinstance(cards, dict):
|
|
58
|
+
return False
|
|
59
|
+
|
|
60
|
+
# 1. The result side of a terminal call — `decode_tool_output_card`, whose
|
|
61
|
+
# own `is_error` is the resolved five-grammar verdict.
|
|
62
|
+
output = cards.get("terminal_output")
|
|
63
|
+
if isinstance(output, dict):
|
|
64
|
+
if output.get("is_error") is True:
|
|
65
|
+
return True
|
|
66
|
+
# 2. The same card's resolved status, which covers a card built by a
|
|
67
|
+
# caller that did not carry `is_error` forward.
|
|
68
|
+
if output.get("status") in _FAILED_STATUSES:
|
|
69
|
+
return True
|
|
70
|
+
|
|
71
|
+
# 3. A patch, from either side — `decode_patch_event_card`'s `success`, and
|
|
72
|
+
# the standalone patch event's own `status`, which the client treats as
|
|
73
|
+
# an error independently of `success`.
|
|
74
|
+
patch = cards.get("patch")
|
|
75
|
+
if isinstance(patch, dict):
|
|
76
|
+
if patch.get("success") is False:
|
|
77
|
+
return True
|
|
78
|
+
if patch.get("status") in _FAILED_STATUSES:
|
|
79
|
+
return True
|
|
80
|
+
|
|
81
|
+
# 4/5. The web-search and MCP completions — `decode_secondary_event_card`,
|
|
82
|
+
# reached through the call-side card's folded `completion`.
|
|
83
|
+
for family in ("web", "mcp"):
|
|
84
|
+
holder = cards.get(family)
|
|
85
|
+
if not isinstance(holder, dict):
|
|
86
|
+
continue
|
|
87
|
+
completion = holder.get("completion")
|
|
88
|
+
if isinstance(completion, dict) and completion.get("status") in _FAILED_STATUSES:
|
|
89
|
+
return True
|
|
90
|
+
|
|
91
|
+
return False
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def reasoning_heading_texts(payload: Any) -> list[str] | None:
|
|
95
|
+
"""The authored headings of one reasoning payload's ``summary``, or ``None``.
|
|
96
|
+
|
|
97
|
+
Lifted verbatim out of ``_lib_codex_conversation_query._reasoning_headings``
|
|
98
|
+
so the outline and the reader decompose by ONE rule. S2's §4.6 precedent is
|
|
99
|
+
that the wire publishes every heading and the render layer dedupes; a second
|
|
100
|
+
copy of this parse is exactly how the two would drift.
|
|
101
|
+
|
|
102
|
+
Headings come from ``payload["summary"]`` ONLY. ``payload["content"]`` is the
|
|
103
|
+
body, which stays disclosure content and is never decomposed.
|
|
104
|
+
|
|
105
|
+
All-or-nothing. When the payload is absent, unreadable or malformed this
|
|
106
|
+
returns ``None`` and the caller omits the field entirely, so the client falls
|
|
107
|
+
back to the stored ``title``/``summary`` rendering. Decomposition never fails
|
|
108
|
+
the request and never partially populates.
|
|
109
|
+
"""
|
|
110
|
+
if not isinstance(payload, dict):
|
|
111
|
+
return None
|
|
112
|
+
summary = payload.get("summary")
|
|
113
|
+
if not isinstance(summary, list) or not summary:
|
|
114
|
+
return None
|
|
115
|
+
entries: list[str] = []
|
|
116
|
+
for entry in summary:
|
|
117
|
+
if not isinstance(entry, dict):
|
|
118
|
+
return None
|
|
119
|
+
text = entry.get("text")
|
|
120
|
+
# Mirror `_join_content_texts`, which is what produced the stored
|
|
121
|
+
# summary: it keeps non-empty string `text` leaves and ignores the rest.
|
|
122
|
+
if text is None:
|
|
123
|
+
continue
|
|
124
|
+
if not isinstance(text, str):
|
|
125
|
+
return None
|
|
126
|
+
if text:
|
|
127
|
+
entries.append(text)
|
|
128
|
+
headings = decompose_reasoning_headings(entries)
|
|
129
|
+
return headings or None
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
# `@@ -<old start>[,<old count>] +<new start>[,<new count>] @@`. An omitted count
|
|
133
|
+
# means 1, which is what a single-line hunk emits.
|
|
134
|
+
_HUNK_HEADER_RE = re.compile(r"^@@ -\d+(?:,(\d+))? \+\d+(?:,(\d+))? @@")
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _diff_counts(diff: str) -> tuple[int, int] | None:
|
|
138
|
+
"""Added and removed line counts of one unified diff, or ``None``.
|
|
139
|
+
|
|
140
|
+
Counted INSIDE the ``@@`` hunks, never by prefix over the whole text. A
|
|
141
|
+
prefix filter cannot separate a file header from content that looks like
|
|
142
|
+
one: removing the SQL comment ``-- legacy`` renders ``--- legacy`` and
|
|
143
|
+
adding ``++i;`` renders ``+++i;``, so excluding every line that starts with
|
|
144
|
+
``---`` or ``+++`` drops real changed lines and silently understates the
|
|
145
|
+
diff — the same undercount §4.5 exists to prevent, reached by a different
|
|
146
|
+
route. A header can only appear before a hunk opens, and the hunk header
|
|
147
|
+
states how many old-side and new-side lines follow, so inside a hunk nothing
|
|
148
|
+
has to be recognised by its prefix alone.
|
|
149
|
+
|
|
150
|
+
``None`` when the text carries no hunk at all. Such a string is not a
|
|
151
|
+
unified diff, and counting its ``+``/``-`` prefixed lines would be the same
|
|
152
|
+
guess; the caller reports an undetermined count rather than a wrong one.
|
|
153
|
+
"""
|
|
154
|
+
added = removed = 0
|
|
155
|
+
old_left = new_left = 0
|
|
156
|
+
in_hunk = saw_hunk = False
|
|
157
|
+
for line in diff.split("\n"):
|
|
158
|
+
if not in_hunk:
|
|
159
|
+
header = _HUNK_HEADER_RE.match(line)
|
|
160
|
+
if header is None:
|
|
161
|
+
continue
|
|
162
|
+
saw_hunk = True
|
|
163
|
+
old_left = int(header.group(1)) if header.group(1) is not None else 1
|
|
164
|
+
new_left = int(header.group(2)) if header.group(2) is not None else 1
|
|
165
|
+
in_hunk = old_left > 0 or new_left > 0
|
|
166
|
+
continue
|
|
167
|
+
if line.startswith("\\"):
|
|
168
|
+
# "" belongs to neither side's line count.
|
|
169
|
+
continue
|
|
170
|
+
if line.startswith("+"):
|
|
171
|
+
added += 1
|
|
172
|
+
new_left -= 1
|
|
173
|
+
elif line.startswith("-"):
|
|
174
|
+
removed += 1
|
|
175
|
+
old_left -= 1
|
|
176
|
+
else:
|
|
177
|
+
old_left -= 1
|
|
178
|
+
new_left -= 1
|
|
179
|
+
if old_left <= 0 and new_left <= 0:
|
|
180
|
+
in_hunk = False
|
|
181
|
+
return (added, removed) if saw_hunk else None
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def patch_file_touches(payload: Any) -> list[dict]:
|
|
185
|
+
"""Per-file touches of one ``patch_apply_end`` payload, in provider order.
|
|
186
|
+
|
|
187
|
+
Each entry is ``{"path", "op", "added", "removed"}``; ``added``/``removed``
|
|
188
|
+
are ``None`` when the count cannot be determined, never 0, because 0 is a
|
|
189
|
+
claim and an undetermined count is not.
|
|
190
|
+
|
|
191
|
+
**Counted from the UNBOUNDED raw ``changes`` entry, before card allocation**
|
|
192
|
+
(spec §4.5). ``decode_patch_event_card`` shares one 16,000-character budget
|
|
193
|
+
across stdout, stderr and every file and caps the file list at 128, and S3
|
|
194
|
+
measured 85 real change entries over that budget — so counting `+`/`-` lines
|
|
195
|
+
out of a served ``unified_diff`` silently undercounts, and a wholly clipped
|
|
196
|
+
diff would read as no change at all despite complete evidence sitting in the
|
|
197
|
+
payload. Presentation cards and whole-session statistics have different
|
|
198
|
+
bounds and must not share one.
|
|
199
|
+
|
|
200
|
+
Only the DICT-shaped ``changes`` is a real production source. S3 measured
|
|
201
|
+
all 4,829 patch events arriving as a dict keyed by file path; #489 taught
|
|
202
|
+
the stored file-search projection to consume that shape too. The legacy
|
|
203
|
+
list shape stays supported here and at ingest so a provider change cannot
|
|
204
|
+
silently produce an empty file list.
|
|
205
|
+
"""
|
|
206
|
+
if not isinstance(payload, dict) or payload.get("type") != "patch_apply_end":
|
|
207
|
+
return []
|
|
208
|
+
changes = payload.get("changes")
|
|
209
|
+
touches: list[dict] = []
|
|
210
|
+
if isinstance(changes, dict):
|
|
211
|
+
items: Iterable[tuple[Any, Any]] = changes.items()
|
|
212
|
+
elif isinstance(changes, list):
|
|
213
|
+
# The path lives on the entry in this shape, and the kind is `status`
|
|
214
|
+
# rather than `type` — the two shapes disagree on both.
|
|
215
|
+
items = (
|
|
216
|
+
(change.get("path"), change) for change in changes
|
|
217
|
+
if isinstance(change, dict)
|
|
218
|
+
)
|
|
219
|
+
else:
|
|
220
|
+
return []
|
|
221
|
+
for path, change in items:
|
|
222
|
+
if not isinstance(path, str) or not isinstance(change, dict):
|
|
223
|
+
continue
|
|
224
|
+
kind = change.get("type")
|
|
225
|
+
if not isinstance(kind, str):
|
|
226
|
+
kind = change.get("status")
|
|
227
|
+
added = removed = None
|
|
228
|
+
diff = change.get("unified_diff")
|
|
229
|
+
if isinstance(diff, str):
|
|
230
|
+
counted = _diff_counts(diff)
|
|
231
|
+
if counted is not None:
|
|
232
|
+
added, removed = counted
|
|
233
|
+
elif kind in {"add", "delete"} and isinstance(change.get("content"), str):
|
|
234
|
+
lines = change["content"].split("\n")
|
|
235
|
+
if lines and lines[-1] == "":
|
|
236
|
+
lines.pop()
|
|
237
|
+
added, removed = ((len(lines), 0) if kind == "add"
|
|
238
|
+
else (0, len(lines)))
|
|
239
|
+
touches.append({
|
|
240
|
+
"path": path,
|
|
241
|
+
"op": kind if isinstance(kind, str) else None,
|
|
242
|
+
"added": added,
|
|
243
|
+
"removed": removed,
|
|
244
|
+
})
|
|
245
|
+
return touches
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def fold_owner_by_position(
|
|
249
|
+
groups: Iterable[Iterable[Position]],
|
|
250
|
+
) -> dict[Position, Position]:
|
|
251
|
+
"""Map every row position to the position of its fold group's first row.
|
|
252
|
+
|
|
253
|
+
That first row is the ``tool_call`` a failing ``tool_output`` or completion
|
|
254
|
+
event belongs to. ``_fold_groups_for_item`` computes this membership
|
|
255
|
+
payload-free and ``_build_segment_index`` discarded it, so the outline had no
|
|
256
|
+
way to say WHICH call a failure belongs to — only that the segment contained
|
|
257
|
+
one.
|
|
258
|
+
|
|
259
|
+
Grouping is a conservative SUPERSET of what the block builder actually folds,
|
|
260
|
+
which is the safe direction here: attributing a failure to the call it was
|
|
261
|
+
grouped with can at worst name a call that did not fold, and never splits a
|
|
262
|
+
call from an outcome that did.
|
|
263
|
+
"""
|
|
264
|
+
owners: dict[Position, Position] = {}
|
|
265
|
+
for group in groups:
|
|
266
|
+
head: Position | None = None
|
|
267
|
+
for position in group:
|
|
268
|
+
if head is None:
|
|
269
|
+
head = position
|
|
270
|
+
owners[position] = head
|
|
271
|
+
return owners
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
@dataclasses.dataclass
|
|
275
|
+
class EventDerivation:
|
|
276
|
+
"""Everything the outline derives from one scoped pass over the raw payloads.
|
|
277
|
+
|
|
278
|
+
Keyed by physical position throughout, because that is the only identity the
|
|
279
|
+
message table and the event table share, and because the item and segment
|
|
280
|
+
keys a landmark ultimately carries are minted later from the rows these
|
|
281
|
+
positions name.
|
|
282
|
+
|
|
283
|
+
**Mutable, and filled in by the pass as it streams.** The maps are complete
|
|
284
|
+
only once ``_derive_outline_events`` has returned; a partly-filled instance
|
|
285
|
+
is never handed to a consumer. The dataclass is deliberately not frozen:
|
|
286
|
+
freezing rebinding of the three attributes while the dicts they name are
|
|
287
|
+
mutated in place would advertise an immutability this object does not have.
|
|
288
|
+
|
|
289
|
+
``errors_by_position`` answers "did this call fail" for the outcome-bearing
|
|
290
|
+
rows the pass classified — a ``tool_output``, or a ``patch_apply_end`` /
|
|
291
|
+
``web_search_end`` / ``mcp_tool_call_end`` event. **Absence is a third state,
|
|
292
|
+
not ``False``.** A position is absent when the scope did not select it, and
|
|
293
|
+
equally when it was selected and could not be classified — its event row is
|
|
294
|
+
gone, its payload does not parse, or the decoder declined the shape. The
|
|
295
|
+
caller cannot tell those apart from this map and must not read absence as a
|
|
296
|
+
claim that the call succeeded: a count derived from it is a count of
|
|
297
|
+
failures FOUND, and a route that needs to distinguish "looked and found
|
|
298
|
+
none" from "could not look" compares these keys against the in-scope set it
|
|
299
|
+
supplied.
|
|
300
|
+
"""
|
|
301
|
+
|
|
302
|
+
errors_by_position: dict[Position, bool] = dataclasses.field(default_factory=dict)
|
|
303
|
+
patch_files_by_position: dict[Position, list[dict]] = dataclasses.field(
|
|
304
|
+
default_factory=dict)
|
|
305
|
+
headings_by_position: dict[Position, list[str]] = dataclasses.field(
|
|
306
|
+
default_factory=dict)
|
|
307
|
+
|
|
308
|
+
def failing_positions(self) -> set[Position]:
|
|
309
|
+
return {pos for pos, failed in self.errors_by_position.items() if failed}
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
"""Read-time cleaning of harness markup out of a Codex title (#463 S4 §5).
|
|
2
|
+
|
|
3
|
+
Pure kernel: one entry point, ``clean_codex_title(text) -> str``, and a CLOSED
|
|
4
|
+
allowlist of the grammars a census of the real store actually found.
|
|
5
|
+
|
|
6
|
+
**Read time, not ingest (D5).** The title is stored —
|
|
7
|
+
``codex_conversation_rollups.title`` is written at ingest and ``_rollup_fields``
|
|
8
|
+
returns it on its fast path — so repairing ``derive_title`` would heal nothing
|
|
9
|
+
for the conversations that are already wrong. Cleaning on the read path heals
|
|
10
|
+
all history with no migration and no reingest flag.
|
|
11
|
+
|
|
12
|
+
**The allowlist is a measurement, not a guess (§5.4).** Over the 438 stored
|
|
13
|
+
Codex rollup titles in the production store on 2026-08-04:
|
|
14
|
+
|
|
15
|
+
=========================================== ===== ===========
|
|
16
|
+
grammar count disposition
|
|
17
|
+
=========================================== ===== ===========
|
|
18
|
+
``[$name](<abs path>/SKILL.md) <rest>`` 165 unwrap
|
|
19
|
+
``<command-name>…</command-name> …`` 40 see below
|
|
20
|
+
``<recommended_plugins> …`` 6 strip
|
|
21
|
+
``<command-message>…</command-message> …`` 1 see below
|
|
22
|
+
=========================================== ===== ===========
|
|
23
|
+
|
|
24
|
+
Nothing else occurred. No title carried a tag anywhere but at its head (0 of
|
|
25
|
+
438), and the two ``CODEX_TITLE_SKIP_PREFIXES`` wrappers
|
|
26
|
+
(``<environment_context>``, ``<user_instructions>``) appeared 0 times, because
|
|
27
|
+
they are skipped at ingest by a different mechanism that stays where it is
|
|
28
|
+
(§5.3).
|
|
29
|
+
|
|
30
|
+
Within the command wrapper the three tags are dispositioned separately, from
|
|
31
|
+
what their content actually looks like: ``command-name`` is the slash command
|
|
32
|
+
and is STRIPPED, while ``command-message`` and ``command-args`` carry the human
|
|
33
|
+
text and are UNWRAPPED. On the corpus that turns
|
|
34
|
+
``<command-name>/model</command-name> <command-message>model</command-message>
|
|
35
|
+
<command-args>fable</command-args>`` into ``model fable``.
|
|
36
|
+
|
|
37
|
+
``recommended_plugins`` never closes in the data: titles are capped at 120
|
|
38
|
+
characters, so the stored value is the head of a plugin catalogue. Stripping it
|
|
39
|
+
leaves nothing, and ``_display_chain`` falls through to the project label and
|
|
40
|
+
then to a short native thread id, which the chain already does.
|
|
41
|
+
|
|
42
|
+
**Closed, and deliberately so.** A general tag stripper would eat user-authored
|
|
43
|
+
angle brackets in a title. An unrecognized construct passes through BYTE for
|
|
44
|
+
byte — the function returns its input unchanged when no rule fires, so the 226
|
|
45
|
+
titles the census found clean, and every prose label this is applied to, cannot
|
|
46
|
+
move.
|
|
47
|
+
"""
|
|
48
|
+
from __future__ import annotations
|
|
49
|
+
|
|
50
|
+
import re
|
|
51
|
+
|
|
52
|
+
# `[$skill-name](/abs/path/to/SKILL.md)` — the Codex skill invocation. The
|
|
53
|
+
# prompt text after it is real, and the link target is a private filesystem path
|
|
54
|
+
# that leaks into every title surface. Byte-identical to the client's
|
|
55
|
+
# `cleanQualifiedTitle` regex, so the two agree on the same input and applying
|
|
56
|
+
# both is a no-op.
|
|
57
|
+
#
|
|
58
|
+
# NO trailing lookahead. The first version required whitespace or end of string
|
|
59
|
+
# after the closing paren, so `…/SKILL.md)Task B of issue #450.` — prompt text
|
|
60
|
+
# written straight against the paren — did not match and the whole link,
|
|
61
|
+
# absolute path included, reached the reader header and the outline rail. Two of
|
|
62
|
+
# 300 served titles in the test store carry that form. Nothing is lost by
|
|
63
|
+
# dropping the lookahead: the pattern is head-anchored (`pattern.match`) and its
|
|
64
|
+
# target is the literal `/SKILL.md)`, so it cannot start matching mid-title or
|
|
65
|
+
# consume any other Markdown link.
|
|
66
|
+
_SKILL_LINK_RE = re.compile(
|
|
67
|
+
r"\[((?:\$)[^\]\r\n]+)\]\([^)\r\n]*/SKILL\.md\)")
|
|
68
|
+
|
|
69
|
+
_STRIP, _UNWRAP = "strip", "unwrap"
|
|
70
|
+
|
|
71
|
+
# Head-anchored, in match order. Each entry is (pattern, disposition, group) —
|
|
72
|
+
# `group` names the capture an `unwrap` keeps.
|
|
73
|
+
_GRAMMARS: tuple[tuple[re.Pattern, str, int], ...] = (
|
|
74
|
+
(_SKILL_LINK_RE, _UNWRAP, 1),
|
|
75
|
+
(re.compile(r"<command-name>(.*?)</command-name>", re.S), _STRIP, 1),
|
|
76
|
+
(re.compile(r"<command-message>(.*?)</command-message>", re.S), _UNWRAP, 1),
|
|
77
|
+
(re.compile(r"<command-args>(.*?)</command-args>", re.S), _UNWRAP, 1),
|
|
78
|
+
# The closing form first: alternation is ordered, and the open-ended arm
|
|
79
|
+
# would otherwise swallow a closed construct's tail.
|
|
80
|
+
(re.compile(r"<recommended_plugins>.*?</recommended_plugins>"
|
|
81
|
+
r"|<recommended_plugins>.*", re.S), _STRIP, 0),
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def clean_codex_title(text) -> str:
|
|
86
|
+
"""The title with recognized leading harness markup removed or unwrapped.
|
|
87
|
+
|
|
88
|
+
Returns the input unchanged when no grammar in the allowlist matches its
|
|
89
|
+
head, including for a non-string or empty input, which keeps every
|
|
90
|
+
untouched title and every prose label byte-stable.
|
|
91
|
+
|
|
92
|
+
A construct that strips to nothing yields ``""``, and the caller's fallback
|
|
93
|
+
chain takes over (§5.3).
|
|
94
|
+
"""
|
|
95
|
+
if not isinstance(text, str) or not text:
|
|
96
|
+
return text if isinstance(text, str) else ""
|
|
97
|
+
kept: list[str] = []
|
|
98
|
+
rest = text
|
|
99
|
+
matched = False
|
|
100
|
+
while rest:
|
|
101
|
+
head = rest.lstrip()
|
|
102
|
+
for pattern, disposition, group in _GRAMMARS:
|
|
103
|
+
found = pattern.match(head)
|
|
104
|
+
if found is None:
|
|
105
|
+
continue
|
|
106
|
+
matched = True
|
|
107
|
+
if disposition == _UNWRAP:
|
|
108
|
+
kept.append(found.group(group))
|
|
109
|
+
rest = head[found.end():]
|
|
110
|
+
break
|
|
111
|
+
else:
|
|
112
|
+
kept.append(head)
|
|
113
|
+
break
|
|
114
|
+
if not matched:
|
|
115
|
+
return text
|
|
116
|
+
return " ".join(" ".join(part.split()) for part in kept if part.strip()).strip()
|