cctally 1.90.1 → 1.92.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +74 -0
- package/README.md +2 -2
- package/bin/_cctally_cache.py +863 -74
- package/bin/_cctally_config.py +57 -0
- package/bin/_cctally_core.py +53 -8
- package/bin/_cctally_dashboard.py +146 -5
- package/bin/_cctally_dashboard_conversation.py +164 -18
- package/bin/_cctally_dashboard_envelope.py +69 -12
- package/bin/_cctally_dashboard_sources.py +27 -1
- package/bin/_cctally_db.py +372 -10
- package/bin/_cctally_doctor.py +18 -1
- package/bin/_cctally_journal.py +535 -13
- package/bin/_cctally_journal_repair.py +6 -0
- package/bin/_cctally_parser.py +6 -0
- package/bin/_cctally_quota.py +171 -55
- package/bin/_cctally_record.py +13 -1
- package/bin/_cctally_rederive.py +4 -0
- package/bin/_cctally_store.py +311 -6
- package/bin/_cctally_transcript.py +32 -2
- package/bin/_lib_cache_report.py +8 -3
- package/bin/_lib_cache_report_wire.py +8 -20
- package/bin/_lib_codex_conversation.py +959 -81
- package/bin/_lib_codex_conversation_query.py +2792 -167
- package/bin/_lib_codex_find_projection.py +370 -0
- package/bin/_lib_codex_harness_preamble.py +176 -0
- package/bin/_lib_codex_hooks.py +5 -3
- package/bin/_lib_codex_js_scan.py +254 -0
- package/bin/_lib_codex_landmarks.py +309 -0
- package/bin/_lib_codex_reasoning_headings.py +73 -0
- package/bin/_lib_codex_segments.py +259 -0
- package/bin/_lib_codex_title_clean.py +116 -0
- package/bin/_lib_conversation_dispatch.py +153 -21
- package/bin/_lib_conversation_watch.py +4 -2
- package/bin/_lib_dashboard_sources.py +33 -32
- package/bin/_lib_doctor.py +64 -0
- package/bin/_lib_quota_alert_axes.py +31 -34
- package/bin/_lib_stats_damage.py +523 -0
- package/bin/cctally +5 -0
- package/dashboard/static/assets/index-BEzzJtUd.js +97 -0
- package/dashboard/static/assets/index-DnWdv8um.css +1 -0
- package/dashboard/static/dashboard.html +2 -2
- package/package.json +9 -1
- package/dashboard/static/assets/index-Bar8-S1i.css +0 -1
- package/dashboard/static/assets/index-CRogVlEC.js +0 -92
|
@@ -0,0 +1,254 @@
|
|
|
1
|
+
"""#463 S3 — a bounded lexical scanner for `tools.<name>(` in Codex programs.
|
|
2
|
+
|
|
3
|
+
Pure kernel: no I/O, nothing imported beyond ``re``.
|
|
4
|
+
|
|
5
|
+
Why this is a tokenizer and not a search. 17,777 uncarded `exec` calls, 51% of
|
|
6
|
+
the uncarded set, are JavaScript programs that declare constants, filter
|
|
7
|
+
`ALL_TOOLS`, await `Promise.all` and invoke `tools.exec_command`,
|
|
8
|
+
`tools.write_stdin` or other tools. The obvious way to card them is a textual
|
|
9
|
+
scan for `tools.<name>(`, and it is not acceptable:
|
|
10
|
+
`tests/test_codex_conversation_normalization.py` asserts that a call written
|
|
11
|
+
inside a string literal, inside a `//` comment and inside a regex literal each
|
|
12
|
+
decode to `None`, and a textual scanner would decode all three — fabricating
|
|
13
|
+
tool activity from a comment. A `complete: false` flag communicates omission; it
|
|
14
|
+
cannot make a false positive truthful.
|
|
15
|
+
|
|
16
|
+
So the source is walked ONCE through a small state machine that skips string
|
|
17
|
+
literals, template literals (including `${}` substitutions, which return to code
|
|
18
|
+
state), regex literals and both comment forms, and an invocation is recognized
|
|
19
|
+
only at a genuine token position whose member path is exactly `tools.<name>`.
|
|
20
|
+
|
|
21
|
+
It stays closed, bounded and non-executing. Lexing is not evaluation: nothing
|
|
22
|
+
here runs, resolves an identifier or follows a reference, so an aliased
|
|
23
|
+
`alias.exec_command(…)` and a computed `tools[key](…)` are both invisible by
|
|
24
|
+
construction, which is correct — they are not provably `tools.<name>`.
|
|
25
|
+
|
|
26
|
+
Anything that cannot be lexed cleanly returns ``None``, and the caller then
|
|
27
|
+
produces no card at all rather than a partial one.
|
|
28
|
+
"""
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
import re
|
|
32
|
+
|
|
33
|
+
# The same ceiling `decode_tool_call_card` already applies to a harness body.
|
|
34
|
+
# Duplicated as a literal rather than imported, because importing the decoder
|
|
35
|
+
# module here would be circular — that module imports this one.
|
|
36
|
+
_PARSE_CAP = 1_000_000
|
|
37
|
+
|
|
38
|
+
_IDENT_START = re.compile(r"[A-Za-z_$]")
|
|
39
|
+
_IDENT_RE = re.compile(r"[A-Za-z_$][A-Za-z0-9_$]*")
|
|
40
|
+
|
|
41
|
+
# Anchored at a genuine `tools` token. Whitespace is tolerated around the dot and
|
|
42
|
+
# before the argument list because real programs wrap long chains across lines.
|
|
43
|
+
_MEMBER_CALL_RE = re.compile(
|
|
44
|
+
r"tools\s*\.\s*([A-Za-z_$][A-Za-z0-9_$]*)\s*\(")
|
|
45
|
+
|
|
46
|
+
# A `/` opens a regex literal only when the previous significant token is one of
|
|
47
|
+
# these, or when it is the first token in the source. After a value — an
|
|
48
|
+
# identifier, a number, a string, `)` or `]` — a `/` is division. Getting this
|
|
49
|
+
# wrong in the permissive direction would swallow the rest of the program into a
|
|
50
|
+
# regex and silently lose every invocation after it.
|
|
51
|
+
_REGEX_ALLOWED_PUNCTUATION = frozenset("(,=:[!&|?{};+-*%<>~^")
|
|
52
|
+
_REGEX_ALLOWED_KEYWORDS = frozenset({
|
|
53
|
+
"return", "typeof", "instanceof", "in", "of", "new", "delete", "void",
|
|
54
|
+
"case", "do", "else", "yield", "await", "throw",
|
|
55
|
+
})
|
|
56
|
+
|
|
57
|
+
# Sentinel `prev` value meaning "a completed value", after which `/` is division.
|
|
58
|
+
_VALUE = "\0value"
|
|
59
|
+
|
|
60
|
+
_DEFAULT = 0
|
|
61
|
+
_SINGLE = 1
|
|
62
|
+
_DOUBLE = 2
|
|
63
|
+
_TEMPLATE = 3
|
|
64
|
+
_LINE_COMMENT = 4
|
|
65
|
+
_BLOCK_COMMENT = 5
|
|
66
|
+
_REGEX = 6
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _regex_allowed(prev: str | None) -> bool:
|
|
70
|
+
if prev is None:
|
|
71
|
+
return True
|
|
72
|
+
if prev == _VALUE:
|
|
73
|
+
return False
|
|
74
|
+
if len(prev) == 1:
|
|
75
|
+
return prev in _REGEX_ALLOWED_PUNCTUATION
|
|
76
|
+
return prev in _REGEX_ALLOWED_KEYWORDS
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def scan_tool_invocations(
|
|
80
|
+
source: str, *, limit: int,
|
|
81
|
+
) -> list[tuple[str, int]] | None:
|
|
82
|
+
"""``[(tool_name, arg_open_paren_index), …]``, or ``None``.
|
|
83
|
+
|
|
84
|
+
``None`` means the source could not be lexed cleanly — an unterminated
|
|
85
|
+
string, template, regex or block comment, or a source past the parse cap —
|
|
86
|
+
and the caller must then produce no card.
|
|
87
|
+
|
|
88
|
+
An empty list means the source lexed but contains no `tools.<name>(` at a
|
|
89
|
+
genuine token position. A call inside a literal or a comment is therefore
|
|
90
|
+
indistinguishable from no call at all, which is the point.
|
|
91
|
+
|
|
92
|
+
``limit`` bounds how many invocations are COLLECTED, not how far the scan
|
|
93
|
+
runs: the walk always continues to the end of the source, because an
|
|
94
|
+
unterminated literal after the limit still means the source could not be
|
|
95
|
+
lexed and no card may be built from it.
|
|
96
|
+
"""
|
|
97
|
+
if not isinstance(source, str) or len(source) > _PARSE_CAP:
|
|
98
|
+
return None
|
|
99
|
+
found: list[tuple[str, int]] = []
|
|
100
|
+
state = _DEFAULT
|
|
101
|
+
prev: str | None = None
|
|
102
|
+
brace_depth = 0
|
|
103
|
+
template_returns: list[int] = []
|
|
104
|
+
in_char_class = False
|
|
105
|
+
index = 0
|
|
106
|
+
size = len(source)
|
|
107
|
+
while index < size:
|
|
108
|
+
char = source[index]
|
|
109
|
+
if state == _DEFAULT:
|
|
110
|
+
if char in " \t\r\n":
|
|
111
|
+
index += 1
|
|
112
|
+
continue
|
|
113
|
+
if char == "/" and index + 1 < size and source[index + 1] == "/":
|
|
114
|
+
state = _LINE_COMMENT
|
|
115
|
+
index += 2
|
|
116
|
+
continue
|
|
117
|
+
if char == "/" and index + 1 < size and source[index + 1] == "*":
|
|
118
|
+
state = _BLOCK_COMMENT
|
|
119
|
+
index += 2
|
|
120
|
+
continue
|
|
121
|
+
if char == "/":
|
|
122
|
+
if _regex_allowed(prev):
|
|
123
|
+
state = _REGEX
|
|
124
|
+
in_char_class = False
|
|
125
|
+
index += 1
|
|
126
|
+
continue
|
|
127
|
+
prev = "/"
|
|
128
|
+
index += 1
|
|
129
|
+
continue
|
|
130
|
+
if char == "'":
|
|
131
|
+
state = _SINGLE
|
|
132
|
+
index += 1
|
|
133
|
+
continue
|
|
134
|
+
if char == '"':
|
|
135
|
+
state = _DOUBLE
|
|
136
|
+
index += 1
|
|
137
|
+
continue
|
|
138
|
+
if char == "`":
|
|
139
|
+
state = _TEMPLATE
|
|
140
|
+
index += 1
|
|
141
|
+
continue
|
|
142
|
+
if char == "{":
|
|
143
|
+
brace_depth += 1
|
|
144
|
+
prev = "{"
|
|
145
|
+
index += 1
|
|
146
|
+
continue
|
|
147
|
+
if char == "}":
|
|
148
|
+
if template_returns and brace_depth == template_returns[-1]:
|
|
149
|
+
template_returns.pop()
|
|
150
|
+
state = _TEMPLATE
|
|
151
|
+
prev = _VALUE
|
|
152
|
+
index += 1
|
|
153
|
+
continue
|
|
154
|
+
brace_depth -= 1
|
|
155
|
+
prev = "}"
|
|
156
|
+
index += 1
|
|
157
|
+
continue
|
|
158
|
+
if _IDENT_START.match(char):
|
|
159
|
+
# A `tools` token is only a member path when nothing binds to it
|
|
160
|
+
# on the left. The test is the previous significant TOKEN, not
|
|
161
|
+
# the character immediately before `tools`: JavaScript permits
|
|
162
|
+
# whitespace and newlines around `.`, so `evil . tools.x` and
|
|
163
|
+
# `evil.\n tools.x` are member accesses on `evil` even though
|
|
164
|
+
# the character before `tools` is a space. `prev` is unchanged by
|
|
165
|
+
# the whitespace branch above, so it still holds the `.`.
|
|
166
|
+
if prev != ".":
|
|
167
|
+
match = _MEMBER_CALL_RE.match(source, index)
|
|
168
|
+
if match is not None:
|
|
169
|
+
if len(found) < limit:
|
|
170
|
+
found.append((match.group(1), match.end() - 1))
|
|
171
|
+
prev = "("
|
|
172
|
+
index = match.end()
|
|
173
|
+
continue
|
|
174
|
+
word = _IDENT_RE.match(source, index)
|
|
175
|
+
prev = word.group(0)
|
|
176
|
+
index = word.end()
|
|
177
|
+
continue
|
|
178
|
+
if char.isdigit():
|
|
179
|
+
index += 1
|
|
180
|
+
prev = _VALUE
|
|
181
|
+
continue
|
|
182
|
+
prev = char
|
|
183
|
+
index += 1
|
|
184
|
+
continue
|
|
185
|
+
if state in (_SINGLE, _DOUBLE):
|
|
186
|
+
if char == "\\":
|
|
187
|
+
index += 2
|
|
188
|
+
continue
|
|
189
|
+
if (state == _SINGLE and char == "'") or (state == _DOUBLE and char == '"'):
|
|
190
|
+
state = _DEFAULT
|
|
191
|
+
prev = _VALUE
|
|
192
|
+
index += 1
|
|
193
|
+
continue
|
|
194
|
+
if char == "\n":
|
|
195
|
+
return None # an unterminated ordinary string literal
|
|
196
|
+
index += 1
|
|
197
|
+
continue
|
|
198
|
+
if state == _TEMPLATE:
|
|
199
|
+
if char == "\\":
|
|
200
|
+
index += 2
|
|
201
|
+
continue
|
|
202
|
+
if char == "`":
|
|
203
|
+
state = _DEFAULT
|
|
204
|
+
prev = _VALUE
|
|
205
|
+
index += 1
|
|
206
|
+
continue
|
|
207
|
+
if char == "$" and index + 1 < size and source[index + 1] == "{":
|
|
208
|
+
template_returns.append(brace_depth)
|
|
209
|
+
state = _DEFAULT
|
|
210
|
+
prev = "{"
|
|
211
|
+
index += 2
|
|
212
|
+
continue
|
|
213
|
+
index += 1
|
|
214
|
+
continue
|
|
215
|
+
if state == _LINE_COMMENT:
|
|
216
|
+
if char == "\n":
|
|
217
|
+
state = _DEFAULT
|
|
218
|
+
index += 1
|
|
219
|
+
continue
|
|
220
|
+
if state == _BLOCK_COMMENT:
|
|
221
|
+
if char == "*" and index + 1 < size and source[index + 1] == "/":
|
|
222
|
+
state = _DEFAULT
|
|
223
|
+
index += 2
|
|
224
|
+
continue
|
|
225
|
+
index += 1
|
|
226
|
+
continue
|
|
227
|
+
# _REGEX
|
|
228
|
+
if char == "\\":
|
|
229
|
+
index += 2
|
|
230
|
+
continue
|
|
231
|
+
if char == "[":
|
|
232
|
+
in_char_class = True
|
|
233
|
+
index += 1
|
|
234
|
+
continue
|
|
235
|
+
if char == "]":
|
|
236
|
+
in_char_class = False
|
|
237
|
+
index += 1
|
|
238
|
+
continue
|
|
239
|
+
if char == "/" and not in_char_class:
|
|
240
|
+
state = _DEFAULT
|
|
241
|
+
prev = _VALUE
|
|
242
|
+
index += 1
|
|
243
|
+
continue
|
|
244
|
+
if char == "\n":
|
|
245
|
+
return None # an unterminated regex literal
|
|
246
|
+
index += 1
|
|
247
|
+
if state != _DEFAULT and state != _LINE_COMMENT:
|
|
248
|
+
return None # unterminated literal or block comment
|
|
249
|
+
if template_returns:
|
|
250
|
+
return None # unterminated `${` substitution
|
|
251
|
+
return found
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
__all__ = ["scan_tool_invocations"]
|
|
@@ -0,0 +1,309 @@
|
|
|
1
|
+
"""Read-time landmark derivation for the Codex conversation outline (#463 S4).
|
|
2
|
+
|
|
3
|
+
Pure kernel: no SQLite, no I/O, and nothing from the query layer. It consumes
|
|
4
|
+
the S3 card decoders in ``_lib_codex_conversation`` and the S2 heading
|
|
5
|
+
decomposition in ``_lib_codex_reasoning_headings``, and returns small derived
|
|
6
|
+
facts — a failure verdict per outcome-bearing row, the authored reasoning
|
|
7
|
+
headings of a reasoning row, and the per-file touches of a patch event.
|
|
8
|
+
|
|
9
|
+
**Everything here is read-time only.** Nothing it produces may reach
|
|
10
|
+
``codex_conversation_messages.detail_json``: those bytes feed ``_row_source_bytes``,
|
|
11
|
+
which drives ``PAGE_SOURCE_BYTE_BUDGET`` page boundaries, so persisting a derived
|
|
12
|
+
field would make two conversations with identical content paginate differently
|
|
13
|
+
according to which binary ingested them. The standing guard is
|
|
14
|
+
``tests/test_codex_conversation_normalization.py::test_s3_no_read_time_enrichment_is_ever_persisted``.
|
|
15
|
+
"""
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import dataclasses
|
|
19
|
+
import re
|
|
20
|
+
from typing import Any, Iterable
|
|
21
|
+
|
|
22
|
+
from _lib_codex_reasoning_headings import decompose_reasoning_headings
|
|
23
|
+
|
|
24
|
+
# A physical row address — ``(source_path, line_offset)``, the key every
|
|
25
|
+
# conversation table is unique on and the one the payload pass joins by.
|
|
26
|
+
Position = tuple[str, int]
|
|
27
|
+
|
|
28
|
+
# The two status strings that mean the call failed. `decode_tool_output_card`
|
|
29
|
+
# already sets `is_error` from exactly this set, and `classify_tool_failure`
|
|
30
|
+
# below applies it to the patch and completion cards too so one definition
|
|
31
|
+
# covers every family.
|
|
32
|
+
#
|
|
33
|
+
# `"error"` is deliberately in this set. The CLIENT's `OUTCOME_STATUSES`
|
|
34
|
+
# excludes it, so a card carrying `status: "error"` on a call whose call-side
|
|
35
|
+
# card is not `terminal` collapses to `unknown` and is not flagged there. Spec
|
|
36
|
+
# §6.4 takes correctness over bug-compatibility: the server states one
|
|
37
|
+
# enumerated classification and Task 9 brings the client to it, because
|
|
38
|
+
# asserting "the server matches the client exactly" would have frozen a defect.
|
|
39
|
+
#
|
|
40
|
+
# `"running"` and `"unknown"` are NOT failures. `unknown` is a real state
|
|
41
|
+
# covering 17.6% of outputs — 4,585 of them are open sessions, measured — rather
|
|
42
|
+
# than an absence, and reporting it as a failure would invent errors that the
|
|
43
|
+
# reader could then not find.
|
|
44
|
+
_FAILED_STATUSES = frozenset({"failed", "error"})
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def classify_tool_failure(cards: Any) -> bool:
|
|
48
|
+
"""Did this tool call fail? One enumerated definition, spec §6.4.
|
|
49
|
+
|
|
50
|
+
``cards`` maps a card family to the card the decoders produced for this
|
|
51
|
+
call — ``terminal_output``, ``patch``, and ``web``/``mcp`` whose value holds
|
|
52
|
+
the folded ``completion``. An absent family contributes nothing; an
|
|
53
|
+
unrecognized one is ignored rather than guessed at.
|
|
54
|
+
|
|
55
|
+
Each disjunct is named with the card it comes from:
|
|
56
|
+
"""
|
|
57
|
+
if not isinstance(cards, dict):
|
|
58
|
+
return False
|
|
59
|
+
|
|
60
|
+
# 1. The result side of a terminal call — `decode_tool_output_card`, whose
|
|
61
|
+
# own `is_error` is the resolved five-grammar verdict.
|
|
62
|
+
output = cards.get("terminal_output")
|
|
63
|
+
if isinstance(output, dict):
|
|
64
|
+
if output.get("is_error") is True:
|
|
65
|
+
return True
|
|
66
|
+
# 2. The same card's resolved status, which covers a card built by a
|
|
67
|
+
# caller that did not carry `is_error` forward.
|
|
68
|
+
if output.get("status") in _FAILED_STATUSES:
|
|
69
|
+
return True
|
|
70
|
+
|
|
71
|
+
# 3. A patch, from either side — `decode_patch_event_card`'s `success`, and
|
|
72
|
+
# the standalone patch event's own `status`, which the client treats as
|
|
73
|
+
# an error independently of `success`.
|
|
74
|
+
patch = cards.get("patch")
|
|
75
|
+
if isinstance(patch, dict):
|
|
76
|
+
if patch.get("success") is False:
|
|
77
|
+
return True
|
|
78
|
+
if patch.get("status") in _FAILED_STATUSES:
|
|
79
|
+
return True
|
|
80
|
+
|
|
81
|
+
# 4/5. The web-search and MCP completions — `decode_secondary_event_card`,
|
|
82
|
+
# reached through the call-side card's folded `completion`.
|
|
83
|
+
for family in ("web", "mcp"):
|
|
84
|
+
holder = cards.get(family)
|
|
85
|
+
if not isinstance(holder, dict):
|
|
86
|
+
continue
|
|
87
|
+
completion = holder.get("completion")
|
|
88
|
+
if isinstance(completion, dict) and completion.get("status") in _FAILED_STATUSES:
|
|
89
|
+
return True
|
|
90
|
+
|
|
91
|
+
return False
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def reasoning_heading_texts(payload: Any) -> list[str] | None:
|
|
95
|
+
"""The authored headings of one reasoning payload's ``summary``, or ``None``.
|
|
96
|
+
|
|
97
|
+
Lifted verbatim out of ``_lib_codex_conversation_query._reasoning_headings``
|
|
98
|
+
so the outline and the reader decompose by ONE rule. S2's §4.6 precedent is
|
|
99
|
+
that the wire publishes every heading and the render layer dedupes; a second
|
|
100
|
+
copy of this parse is exactly how the two would drift.
|
|
101
|
+
|
|
102
|
+
Headings come from ``payload["summary"]`` ONLY. ``payload["content"]`` is the
|
|
103
|
+
body, which stays disclosure content and is never decomposed.
|
|
104
|
+
|
|
105
|
+
All-or-nothing. When the payload is absent, unreadable or malformed this
|
|
106
|
+
returns ``None`` and the caller omits the field entirely, so the client falls
|
|
107
|
+
back to the stored ``title``/``summary`` rendering. Decomposition never fails
|
|
108
|
+
the request and never partially populates.
|
|
109
|
+
"""
|
|
110
|
+
if not isinstance(payload, dict):
|
|
111
|
+
return None
|
|
112
|
+
summary = payload.get("summary")
|
|
113
|
+
if not isinstance(summary, list) or not summary:
|
|
114
|
+
return None
|
|
115
|
+
entries: list[str] = []
|
|
116
|
+
for entry in summary:
|
|
117
|
+
if not isinstance(entry, dict):
|
|
118
|
+
return None
|
|
119
|
+
text = entry.get("text")
|
|
120
|
+
# Mirror `_join_content_texts`, which is what produced the stored
|
|
121
|
+
# summary: it keeps non-empty string `text` leaves and ignores the rest.
|
|
122
|
+
if text is None:
|
|
123
|
+
continue
|
|
124
|
+
if not isinstance(text, str):
|
|
125
|
+
return None
|
|
126
|
+
if text:
|
|
127
|
+
entries.append(text)
|
|
128
|
+
headings = decompose_reasoning_headings(entries)
|
|
129
|
+
return headings or None
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
# `@@ -<old start>[,<old count>] +<new start>[,<new count>] @@`. An omitted count
|
|
133
|
+
# means 1, which is what a single-line hunk emits.
|
|
134
|
+
_HUNK_HEADER_RE = re.compile(r"^@@ -\d+(?:,(\d+))? \+\d+(?:,(\d+))? @@")
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _diff_counts(diff: str) -> tuple[int, int] | None:
|
|
138
|
+
"""Added and removed line counts of one unified diff, or ``None``.
|
|
139
|
+
|
|
140
|
+
Counted INSIDE the ``@@`` hunks, never by prefix over the whole text. A
|
|
141
|
+
prefix filter cannot separate a file header from content that looks like
|
|
142
|
+
one: removing the SQL comment ``-- legacy`` renders ``--- legacy`` and
|
|
143
|
+
adding ``++i;`` renders ``+++i;``, so excluding every line that starts with
|
|
144
|
+
``---`` or ``+++`` drops real changed lines and silently understates the
|
|
145
|
+
diff — the same undercount §4.5 exists to prevent, reached by a different
|
|
146
|
+
route. A header can only appear before a hunk opens, and the hunk header
|
|
147
|
+
states how many old-side and new-side lines follow, so inside a hunk nothing
|
|
148
|
+
has to be recognised by its prefix alone.
|
|
149
|
+
|
|
150
|
+
``None`` when the text carries no hunk at all. Such a string is not a
|
|
151
|
+
unified diff, and counting its ``+``/``-`` prefixed lines would be the same
|
|
152
|
+
guess; the caller reports an undetermined count rather than a wrong one.
|
|
153
|
+
"""
|
|
154
|
+
added = removed = 0
|
|
155
|
+
old_left = new_left = 0
|
|
156
|
+
in_hunk = saw_hunk = False
|
|
157
|
+
for line in diff.split("\n"):
|
|
158
|
+
if not in_hunk:
|
|
159
|
+
header = _HUNK_HEADER_RE.match(line)
|
|
160
|
+
if header is None:
|
|
161
|
+
continue
|
|
162
|
+
saw_hunk = True
|
|
163
|
+
old_left = int(header.group(1)) if header.group(1) is not None else 1
|
|
164
|
+
new_left = int(header.group(2)) if header.group(2) is not None else 1
|
|
165
|
+
in_hunk = old_left > 0 or new_left > 0
|
|
166
|
+
continue
|
|
167
|
+
if line.startswith("\\"):
|
|
168
|
+
# "" belongs to neither side's line count.
|
|
169
|
+
continue
|
|
170
|
+
if line.startswith("+"):
|
|
171
|
+
added += 1
|
|
172
|
+
new_left -= 1
|
|
173
|
+
elif line.startswith("-"):
|
|
174
|
+
removed += 1
|
|
175
|
+
old_left -= 1
|
|
176
|
+
else:
|
|
177
|
+
old_left -= 1
|
|
178
|
+
new_left -= 1
|
|
179
|
+
if old_left <= 0 and new_left <= 0:
|
|
180
|
+
in_hunk = False
|
|
181
|
+
return (added, removed) if saw_hunk else None
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def patch_file_touches(payload: Any) -> list[dict]:
|
|
185
|
+
"""Per-file touches of one ``patch_apply_end`` payload, in provider order.
|
|
186
|
+
|
|
187
|
+
Each entry is ``{"path", "op", "added", "removed"}``; ``added``/``removed``
|
|
188
|
+
are ``None`` when the count cannot be determined, never 0, because 0 is a
|
|
189
|
+
claim and an undetermined count is not.
|
|
190
|
+
|
|
191
|
+
**Counted from the UNBOUNDED raw ``changes`` entry, before card allocation**
|
|
192
|
+
(spec §4.5). ``decode_patch_event_card`` shares one 16,000-character budget
|
|
193
|
+
across stdout, stderr and every file and caps the file list at 128, and S3
|
|
194
|
+
measured 85 real change entries over that budget — so counting `+`/`-` lines
|
|
195
|
+
out of a served ``unified_diff`` silently undercounts, and a wholly clipped
|
|
196
|
+
diff would read as no change at all despite complete evidence sitting in the
|
|
197
|
+
payload. Presentation cards and whole-session statistics have different
|
|
198
|
+
bounds and must not share one.
|
|
199
|
+
|
|
200
|
+
Only the DICT-shaped ``changes`` is a real production source. S3 measured
|
|
201
|
+
all 4,829 patch events arriving as a dict keyed by file path; #489 taught
|
|
202
|
+
the stored file-search projection to consume that shape too. The legacy
|
|
203
|
+
list shape stays supported here and at ingest so a provider change cannot
|
|
204
|
+
silently produce an empty file list.
|
|
205
|
+
"""
|
|
206
|
+
if not isinstance(payload, dict) or payload.get("type") != "patch_apply_end":
|
|
207
|
+
return []
|
|
208
|
+
changes = payload.get("changes")
|
|
209
|
+
touches: list[dict] = []
|
|
210
|
+
if isinstance(changes, dict):
|
|
211
|
+
items: Iterable[tuple[Any, Any]] = changes.items()
|
|
212
|
+
elif isinstance(changes, list):
|
|
213
|
+
# The path lives on the entry in this shape, and the kind is `status`
|
|
214
|
+
# rather than `type` — the two shapes disagree on both.
|
|
215
|
+
items = (
|
|
216
|
+
(change.get("path"), change) for change in changes
|
|
217
|
+
if isinstance(change, dict)
|
|
218
|
+
)
|
|
219
|
+
else:
|
|
220
|
+
return []
|
|
221
|
+
for path, change in items:
|
|
222
|
+
if not isinstance(path, str) or not isinstance(change, dict):
|
|
223
|
+
continue
|
|
224
|
+
kind = change.get("type")
|
|
225
|
+
if not isinstance(kind, str):
|
|
226
|
+
kind = change.get("status")
|
|
227
|
+
added = removed = None
|
|
228
|
+
diff = change.get("unified_diff")
|
|
229
|
+
if isinstance(diff, str):
|
|
230
|
+
counted = _diff_counts(diff)
|
|
231
|
+
if counted is not None:
|
|
232
|
+
added, removed = counted
|
|
233
|
+
elif kind in {"add", "delete"} and isinstance(change.get("content"), str):
|
|
234
|
+
lines = change["content"].split("\n")
|
|
235
|
+
if lines and lines[-1] == "":
|
|
236
|
+
lines.pop()
|
|
237
|
+
added, removed = ((len(lines), 0) if kind == "add"
|
|
238
|
+
else (0, len(lines)))
|
|
239
|
+
touches.append({
|
|
240
|
+
"path": path,
|
|
241
|
+
"op": kind if isinstance(kind, str) else None,
|
|
242
|
+
"added": added,
|
|
243
|
+
"removed": removed,
|
|
244
|
+
})
|
|
245
|
+
return touches
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def fold_owner_by_position(
|
|
249
|
+
groups: Iterable[Iterable[Position]],
|
|
250
|
+
) -> dict[Position, Position]:
|
|
251
|
+
"""Map every row position to the position of its fold group's first row.
|
|
252
|
+
|
|
253
|
+
That first row is the ``tool_call`` a failing ``tool_output`` or completion
|
|
254
|
+
event belongs to. ``_fold_groups_for_item`` computes this membership
|
|
255
|
+
payload-free and ``_build_segment_index`` discarded it, so the outline had no
|
|
256
|
+
way to say WHICH call a failure belongs to — only that the segment contained
|
|
257
|
+
one.
|
|
258
|
+
|
|
259
|
+
Grouping is a conservative SUPERSET of what the block builder actually folds,
|
|
260
|
+
which is the safe direction here: attributing a failure to the call it was
|
|
261
|
+
grouped with can at worst name a call that did not fold, and never splits a
|
|
262
|
+
call from an outcome that did.
|
|
263
|
+
"""
|
|
264
|
+
owners: dict[Position, Position] = {}
|
|
265
|
+
for group in groups:
|
|
266
|
+
head: Position | None = None
|
|
267
|
+
for position in group:
|
|
268
|
+
if head is None:
|
|
269
|
+
head = position
|
|
270
|
+
owners[position] = head
|
|
271
|
+
return owners
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
@dataclasses.dataclass
|
|
275
|
+
class EventDerivation:
|
|
276
|
+
"""Everything the outline derives from one scoped pass over the raw payloads.
|
|
277
|
+
|
|
278
|
+
Keyed by physical position throughout, because that is the only identity the
|
|
279
|
+
message table and the event table share, and because the item and segment
|
|
280
|
+
keys a landmark ultimately carries are minted later from the rows these
|
|
281
|
+
positions name.
|
|
282
|
+
|
|
283
|
+
**Mutable, and filled in by the pass as it streams.** The maps are complete
|
|
284
|
+
only once ``_derive_outline_events`` has returned; a partly-filled instance
|
|
285
|
+
is never handed to a consumer. The dataclass is deliberately not frozen:
|
|
286
|
+
freezing rebinding of the three attributes while the dicts they name are
|
|
287
|
+
mutated in place would advertise an immutability this object does not have.
|
|
288
|
+
|
|
289
|
+
``errors_by_position`` answers "did this call fail" for the outcome-bearing
|
|
290
|
+
rows the pass classified — a ``tool_output``, or a ``patch_apply_end`` /
|
|
291
|
+
``web_search_end`` / ``mcp_tool_call_end`` event. **Absence is a third state,
|
|
292
|
+
not ``False``.** A position is absent when the scope did not select it, and
|
|
293
|
+
equally when it was selected and could not be classified — its event row is
|
|
294
|
+
gone, its payload does not parse, or the decoder declined the shape. The
|
|
295
|
+
caller cannot tell those apart from this map and must not read absence as a
|
|
296
|
+
claim that the call succeeded: a count derived from it is a count of
|
|
297
|
+
failures FOUND, and a route that needs to distinguish "looked and found
|
|
298
|
+
none" from "could not look" compares these keys against the in-scope set it
|
|
299
|
+
supplied.
|
|
300
|
+
"""
|
|
301
|
+
|
|
302
|
+
errors_by_position: dict[Position, bool] = dataclasses.field(default_factory=dict)
|
|
303
|
+
patch_files_by_position: dict[Position, list[dict]] = dataclasses.field(
|
|
304
|
+
default_factory=dict)
|
|
305
|
+
headings_by_position: dict[Position, list[str]] = dataclasses.field(
|
|
306
|
+
default_factory=dict)
|
|
307
|
+
|
|
308
|
+
def failing_positions(self) -> set[Position]:
|
|
309
|
+
return {pos for pos, failed in self.errors_by_position.items() if failed}
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Pure decomposition of a Codex reasoning aggregate into its authored headings.
|
|
2
|
+
|
|
3
|
+
#463 S2 §2.4. Codex writes reasoning as short bold headings. When an aggregate
|
|
4
|
+
holds several of them joined by newlines, ``_REASONING_TITLE_RE`` cannot
|
|
5
|
+
fullmatch the whole summary and the blob becomes one ``summary``, which the
|
|
6
|
+
reader renders as one clipped line. Measured over a real store: 5,081 of the
|
|
7
|
+
5,234 summary-only reasoning blocks are entirely bold-heading lines, holding
|
|
8
|
+
12,323 individual headings between them, and every one of those headings is
|
|
9
|
+
today invisible past the first thirty characters of the first one.
|
|
10
|
+
|
|
11
|
+
This module recovers the individual headings WITHOUT touching the stored
|
|
12
|
+
projection. That seam is load-bearing and is the trap most likely to be walked
|
|
13
|
+
into again: ``_row_is_reasoning_title`` reads the stored reasoning projection to
|
|
14
|
+
decide whether a row is a title boundary, and that is one of segmentation's two
|
|
15
|
+
semantic boundaries. Decomposing inside ``_reasoning_projection`` would change
|
|
16
|
+
which rows are boundaries and therefore move segment boundaries, which #463 S1's
|
|
17
|
+
contract forbids. Decomposition therefore runs at READ time, over the retained
|
|
18
|
+
payload's ``summary`` entries, and the stored ``title``, ``summary`` and ``body``
|
|
19
|
+
keep exactly today's values.
|
|
20
|
+
|
|
21
|
+
No I/O, no database, no imports beyond ``re``.
|
|
22
|
+
"""
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import re
|
|
26
|
+
from collections.abc import Iterable
|
|
27
|
+
|
|
28
|
+
# The same shape ``_lib_codex_conversation._REASONING_TITLE_RE`` recognises,
|
|
29
|
+
# applied per LINE rather than to the whole summary. Deliberately NOT carrying
|
|
30
|
+
# the stored projection's additional guard (which rejects a title containing
|
|
31
|
+
# ``**``): the projection is a frozen segmentation input, this is a read-time
|
|
32
|
+
# rendering concern, and §2.4 states the per-line rule without that guard.
|
|
33
|
+
_HEADING_LINE_RE = re.compile(r"\A\*\*([^\n]+)\*\*\Z")
|
|
34
|
+
|
|
35
|
+
# Codex interleaves this literal between headings in a minority of aggregates
|
|
36
|
+
# (123 of 5,234 measured). It is layout, not content, so a decomposed entry
|
|
37
|
+
# discards it.
|
|
38
|
+
_SEPARATOR_LINE = "<!-- -->"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def decompose_reasoning_headings(entries: Iterable[object]) -> list[str]:
|
|
42
|
+
"""Heading texts for one reasoning aggregate, in entry order.
|
|
43
|
+
|
|
44
|
+
``entries`` is the ordered ``text`` values of the retained payload's
|
|
45
|
+
``summary`` entries. The unit of decision is the ENTRY and the decision is
|
|
46
|
+
all-or-nothing: an entry yields several headings only when, after discarding
|
|
47
|
+
separator lines, every remaining line is a heading line and at least one
|
|
48
|
+
remains. Otherwise it yields exactly one heading whose text is the entry
|
|
49
|
+
VERBATIM, separator lines included — a fallback that partially cleaned its
|
|
50
|
+
input would be a second transformation nobody reviewed.
|
|
51
|
+
|
|
52
|
+
Coverage invariant (§2.4, and the test): every line of the source aggregate
|
|
53
|
+
is either a discarded separator in a decomposed entry, or appears in exactly
|
|
54
|
+
one heading, exactly once. No line is dropped, none is duplicated, and no
|
|
55
|
+
path both extracts headings and re-renders the original text.
|
|
56
|
+
|
|
57
|
+
Total over its input: a non-string entry is skipped rather than raised on, so
|
|
58
|
+
no provider shape can fail detail assembly. The all-or-nothing rule for a
|
|
59
|
+
malformed payload is enforced at the call site, which validates the whole
|
|
60
|
+
entry list before calling.
|
|
61
|
+
"""
|
|
62
|
+
out: list[str] = []
|
|
63
|
+
for entry in entries or ():
|
|
64
|
+
if not isinstance(entry, str) or not entry.strip():
|
|
65
|
+
continue
|
|
66
|
+
lines = [line.strip() for line in entry.split("\n") if line.strip()]
|
|
67
|
+
kept = [line for line in lines if line != _SEPARATOR_LINE]
|
|
68
|
+
matches = [_HEADING_LINE_RE.fullmatch(line) for line in kept]
|
|
69
|
+
if kept and all(match is not None for match in matches):
|
|
70
|
+
out.extend(match.group(1) for match in matches)
|
|
71
|
+
else:
|
|
72
|
+
out.append(entry)
|
|
73
|
+
return out
|