cctally 1.90.1 → 1.92.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +74 -0
- package/README.md +2 -2
- package/bin/_cctally_cache.py +863 -74
- package/bin/_cctally_config.py +57 -0
- package/bin/_cctally_core.py +53 -8
- package/bin/_cctally_dashboard.py +146 -5
- package/bin/_cctally_dashboard_conversation.py +164 -18
- package/bin/_cctally_dashboard_envelope.py +69 -12
- package/bin/_cctally_dashboard_sources.py +27 -1
- package/bin/_cctally_db.py +372 -10
- package/bin/_cctally_doctor.py +18 -1
- package/bin/_cctally_journal.py +535 -13
- package/bin/_cctally_journal_repair.py +6 -0
- package/bin/_cctally_parser.py +6 -0
- package/bin/_cctally_quota.py +171 -55
- package/bin/_cctally_record.py +13 -1
- package/bin/_cctally_rederive.py +4 -0
- package/bin/_cctally_store.py +311 -6
- package/bin/_cctally_transcript.py +32 -2
- package/bin/_lib_cache_report.py +8 -3
- package/bin/_lib_cache_report_wire.py +8 -20
- package/bin/_lib_codex_conversation.py +959 -81
- package/bin/_lib_codex_conversation_query.py +2792 -167
- package/bin/_lib_codex_find_projection.py +370 -0
- package/bin/_lib_codex_harness_preamble.py +176 -0
- package/bin/_lib_codex_hooks.py +5 -3
- package/bin/_lib_codex_js_scan.py +254 -0
- package/bin/_lib_codex_landmarks.py +309 -0
- package/bin/_lib_codex_reasoning_headings.py +73 -0
- package/bin/_lib_codex_segments.py +259 -0
- package/bin/_lib_codex_title_clean.py +116 -0
- package/bin/_lib_conversation_dispatch.py +153 -21
- package/bin/_lib_conversation_watch.py +4 -2
- package/bin/_lib_dashboard_sources.py +33 -32
- package/bin/_lib_doctor.py +64 -0
- package/bin/_lib_quota_alert_axes.py +31 -34
- package/bin/_lib_stats_damage.py +523 -0
- package/bin/cctally +5 -0
- package/dashboard/static/assets/index-BEzzJtUd.js +97 -0
- package/dashboard/static/assets/index-DnWdv8um.css +1 -0
- package/dashboard/static/dashboard.html +2 -2
- package/package.json +9 -1
- package/dashboard/static/assets/index-Bar8-S1i.css +0 -1
- package/dashboard/static/assets/index-CRogVlEC.js +0 -92
|
@@ -0,0 +1,370 @@
|
|
|
1
|
+
"""Canonical visible-text projection and matching for Codex conversation find.
|
|
2
|
+
|
|
3
|
+
The module is deliberately stdlib-only. Its Markdown scanner models the
|
|
4
|
+
visible text-node boundaries used by the dashboard's ReactMarkdown/remark-gfm
|
|
5
|
+
surface; it is not an HTML renderer and never interprets raw HTML.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
import html
|
|
11
|
+
import re
|
|
12
|
+
from typing import Iterable, Iterator, Sequence
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True)
|
|
16
|
+
class RenderLeaf:
|
|
17
|
+
key: str
|
|
18
|
+
text: str
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True)
|
|
22
|
+
class ProjectedLeaf:
|
|
23
|
+
key: str
|
|
24
|
+
start: int
|
|
25
|
+
end: int
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass(frozen=True)
|
|
29
|
+
class FindRange:
|
|
30
|
+
start: int
|
|
31
|
+
end: int
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass(frozen=True)
|
|
35
|
+
class LeafFragment:
|
|
36
|
+
leaf_key: str
|
|
37
|
+
start: int
|
|
38
|
+
end: int
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class _ProjectionBuilder:
|
|
42
|
+
def __init__(self) -> None:
|
|
43
|
+
self.parts: list[str] = []
|
|
44
|
+
self.leaves: list[dict[str, int | str]] = []
|
|
45
|
+
self.length = 0
|
|
46
|
+
self._open_leaf: int | None = None
|
|
47
|
+
|
|
48
|
+
def boundary(self) -> None:
|
|
49
|
+
self._open_leaf = None
|
|
50
|
+
|
|
51
|
+
def separator(self, value: str) -> None:
|
|
52
|
+
if not value:
|
|
53
|
+
return
|
|
54
|
+
self.boundary()
|
|
55
|
+
self.parts.append(value)
|
|
56
|
+
self.length += len(value)
|
|
57
|
+
|
|
58
|
+
def emit(self, value: str, *, boundary: bool = False, key: str | None = None) -> None:
|
|
59
|
+
if not value:
|
|
60
|
+
return
|
|
61
|
+
if boundary:
|
|
62
|
+
self.boundary()
|
|
63
|
+
start = self.length
|
|
64
|
+
self.parts.append(value)
|
|
65
|
+
self.length += len(value)
|
|
66
|
+
if key is not None:
|
|
67
|
+
self.leaves.append({"key": key, "start": start, "end": self.length})
|
|
68
|
+
self._open_leaf = None
|
|
69
|
+
return
|
|
70
|
+
if self._open_leaf is None:
|
|
71
|
+
self._open_leaf = len(self.leaves)
|
|
72
|
+
self.leaves.append({
|
|
73
|
+
"key": f"t{self._open_leaf}",
|
|
74
|
+
"start": start,
|
|
75
|
+
"end": self.length,
|
|
76
|
+
})
|
|
77
|
+
else:
|
|
78
|
+
self.leaves[self._open_leaf]["end"] = self.length
|
|
79
|
+
|
|
80
|
+
def value(self) -> tuple[str, tuple[ProjectedLeaf, ...]]:
|
|
81
|
+
return (
|
|
82
|
+
"".join(self.parts),
|
|
83
|
+
tuple(ProjectedLeaf(**leaf) for leaf in self.leaves),
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
_TABLE_DELIMITER_RE = re.compile(
|
|
88
|
+
r"^\s*\|?\s*:?-{3,}:?\s*(?:\|\s*:?-{3,}:?\s*)+\|?\s*$"
|
|
89
|
+
)
|
|
90
|
+
_BLOCK_PREFIX_RE = re.compile(
|
|
91
|
+
r"^\s*(?:(?:#{1,6})\s+|>\s?|(?:[-+*]|\d+[.)])\s+)"
|
|
92
|
+
)
|
|
93
|
+
_TASK_MARKER_RE = re.compile(r"^\[[ xX]\]\s+")
|
|
94
|
+
_AUTOLINK_RE = re.compile(r"<((?:https?://|mailto:)[^ <>]+|[^ <>@]+@[^ <>@]+)>")
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _table_cells(line: str) -> list[str]:
|
|
98
|
+
value = line.strip()
|
|
99
|
+
if value.startswith("|"):
|
|
100
|
+
value = value[1:]
|
|
101
|
+
if value.endswith("|"):
|
|
102
|
+
value = value[:-1]
|
|
103
|
+
cells: list[str] = []
|
|
104
|
+
current: list[str] = []
|
|
105
|
+
escaped = False
|
|
106
|
+
for char in value:
|
|
107
|
+
if escaped:
|
|
108
|
+
current.append(char)
|
|
109
|
+
escaped = False
|
|
110
|
+
elif char == "\\":
|
|
111
|
+
current.append(char)
|
|
112
|
+
escaped = True
|
|
113
|
+
elif char == "|":
|
|
114
|
+
cells.append("".join(current).strip())
|
|
115
|
+
current = []
|
|
116
|
+
else:
|
|
117
|
+
current.append(char)
|
|
118
|
+
cells.append("".join(current).strip())
|
|
119
|
+
return cells
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _find_closing(source: str, token: str, start: int) -> int:
|
|
123
|
+
cursor = start
|
|
124
|
+
while True:
|
|
125
|
+
found = source.find(token, cursor)
|
|
126
|
+
if found < 0:
|
|
127
|
+
return -1
|
|
128
|
+
backslashes = 0
|
|
129
|
+
probe = found - 1
|
|
130
|
+
while probe >= 0 and source[probe] == "\\":
|
|
131
|
+
backslashes += 1
|
|
132
|
+
probe -= 1
|
|
133
|
+
if backslashes % 2 == 0:
|
|
134
|
+
return found
|
|
135
|
+
cursor = found + len(token)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _project_inline(source: str, builder: _ProjectionBuilder) -> None:
|
|
139
|
+
plain: list[str] = []
|
|
140
|
+
|
|
141
|
+
def flush() -> None:
|
|
142
|
+
if plain:
|
|
143
|
+
builder.emit(html.unescape("".join(plain)))
|
|
144
|
+
plain.clear()
|
|
145
|
+
|
|
146
|
+
cursor = 0
|
|
147
|
+
while cursor < len(source):
|
|
148
|
+
if source[cursor] == "\\" and cursor + 1 < len(source):
|
|
149
|
+
plain.append(source[cursor + 1])
|
|
150
|
+
cursor += 2
|
|
151
|
+
continue
|
|
152
|
+
|
|
153
|
+
if source[cursor] == "`":
|
|
154
|
+
run = 1
|
|
155
|
+
while cursor + run < len(source) and source[cursor + run] == "`":
|
|
156
|
+
run += 1
|
|
157
|
+
token = "`" * run
|
|
158
|
+
close = _find_closing(source, token, cursor + run)
|
|
159
|
+
if close >= 0:
|
|
160
|
+
flush()
|
|
161
|
+
builder.emit(source[cursor + run:close].strip(" "), boundary=True)
|
|
162
|
+
builder.boundary()
|
|
163
|
+
cursor = close + run
|
|
164
|
+
continue
|
|
165
|
+
|
|
166
|
+
if source.startswith("![", cursor) or source[cursor] == "[":
|
|
167
|
+
image = source.startswith("
|
|
170
|
+
if label_end >= 0:
|
|
171
|
+
destination_end = source.find(")", label_end + 2)
|
|
172
|
+
if destination_end >= 0:
|
|
173
|
+
flush()
|
|
174
|
+
builder.boundary()
|
|
175
|
+
if not image:
|
|
176
|
+
_project_inline(source[label_start:label_end], builder)
|
|
177
|
+
builder.boundary()
|
|
178
|
+
cursor = destination_end + 1
|
|
179
|
+
continue
|
|
180
|
+
|
|
181
|
+
if source[cursor] == "<":
|
|
182
|
+
autolink = _AUTOLINK_RE.match(source, cursor)
|
|
183
|
+
if autolink is not None:
|
|
184
|
+
flush()
|
|
185
|
+
builder.boundary()
|
|
186
|
+
label = autolink.group(1)
|
|
187
|
+
builder.emit(label[7:] if label.startswith("mailto:") else label)
|
|
188
|
+
builder.boundary()
|
|
189
|
+
cursor = autolink.end()
|
|
190
|
+
continue
|
|
191
|
+
|
|
192
|
+
matched_delimiter = False
|
|
193
|
+
for token in ("**", "__", "~~", "*", "_"):
|
|
194
|
+
if not source.startswith(token, cursor):
|
|
195
|
+
continue
|
|
196
|
+
close = _find_closing(source, token, cursor + len(token))
|
|
197
|
+
if close < 0 or close == cursor + len(token):
|
|
198
|
+
continue
|
|
199
|
+
flush()
|
|
200
|
+
builder.boundary()
|
|
201
|
+
_project_inline(source[cursor + len(token):close], builder)
|
|
202
|
+
builder.boundary()
|
|
203
|
+
cursor = close + len(token)
|
|
204
|
+
matched_delimiter = True
|
|
205
|
+
break
|
|
206
|
+
if matched_delimiter:
|
|
207
|
+
continue
|
|
208
|
+
|
|
209
|
+
plain.append(source[cursor])
|
|
210
|
+
cursor += 1
|
|
211
|
+
flush()
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def _strip_block_prefix(line: str) -> str:
|
|
215
|
+
value = _BLOCK_PREFIX_RE.sub("", line, count=1)
|
|
216
|
+
if _TASK_MARKER_RE.match(value):
|
|
217
|
+
value = _TASK_MARKER_RE.sub(" ", value, count=1)
|
|
218
|
+
if value.endswith(" "):
|
|
219
|
+
value = value[:-2]
|
|
220
|
+
elif value.endswith("\\"):
|
|
221
|
+
value = value[:-1]
|
|
222
|
+
return value
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def project_markdown(source: str) -> tuple[str, tuple[ProjectedLeaf, ...]]:
|
|
226
|
+
builder = _ProjectionBuilder()
|
|
227
|
+
lines = source.replace("\r\n", "\n").replace("\r", "\n").split("\n")
|
|
228
|
+
blocks: list[tuple[str, object]] = []
|
|
229
|
+
cursor = 0
|
|
230
|
+
while cursor < len(lines):
|
|
231
|
+
line = lines[cursor]
|
|
232
|
+
if not line.strip():
|
|
233
|
+
cursor += 1
|
|
234
|
+
continue
|
|
235
|
+
|
|
236
|
+
fence = re.match(r"^\s*(`{3,}|~{3,})(?:[^`]*)$", line)
|
|
237
|
+
if fence:
|
|
238
|
+
token = fence.group(1)
|
|
239
|
+
body: list[str] = []
|
|
240
|
+
cursor += 1
|
|
241
|
+
while cursor < len(lines) and not re.match(
|
|
242
|
+
rf"^\s*{re.escape(token[0])}{{{len(token)},}}\s*$", lines[cursor]
|
|
243
|
+
):
|
|
244
|
+
body.append(lines[cursor])
|
|
245
|
+
cursor += 1
|
|
246
|
+
closed = cursor < len(lines)
|
|
247
|
+
if closed:
|
|
248
|
+
cursor += 1
|
|
249
|
+
code = "\n".join(body)
|
|
250
|
+
if body:
|
|
251
|
+
code += "\n"
|
|
252
|
+
blocks.append(("code", code))
|
|
253
|
+
continue
|
|
254
|
+
|
|
255
|
+
if cursor + 1 < len(lines) and "|" in line and _TABLE_DELIMITER_RE.match(lines[cursor + 1]):
|
|
256
|
+
rows = [_table_cells(line)]
|
|
257
|
+
cursor += 2
|
|
258
|
+
while cursor < len(lines) and lines[cursor].strip() and "|" in lines[cursor]:
|
|
259
|
+
rows.append(_table_cells(lines[cursor]))
|
|
260
|
+
cursor += 1
|
|
261
|
+
blocks.append(("table", rows))
|
|
262
|
+
continue
|
|
263
|
+
|
|
264
|
+
paragraph = [_strip_block_prefix(line)]
|
|
265
|
+
cursor += 1
|
|
266
|
+
while cursor < len(lines) and lines[cursor].strip():
|
|
267
|
+
if re.match(r"^\s*(`{3,}|~{3,})", lines[cursor]):
|
|
268
|
+
break
|
|
269
|
+
paragraph.append(_strip_block_prefix(lines[cursor]))
|
|
270
|
+
cursor += 1
|
|
271
|
+
blocks.append(("paragraph", paragraph))
|
|
272
|
+
|
|
273
|
+
for block_index, (kind, value) in enumerate(blocks):
|
|
274
|
+
if block_index:
|
|
275
|
+
builder.separator("\n")
|
|
276
|
+
if kind == "code":
|
|
277
|
+
builder.emit(str(value), boundary=True)
|
|
278
|
+
builder.boundary()
|
|
279
|
+
elif kind == "table":
|
|
280
|
+
for row_index, row in enumerate(value):
|
|
281
|
+
if row_index:
|
|
282
|
+
builder.separator("\n")
|
|
283
|
+
for cell_index, cell in enumerate(row):
|
|
284
|
+
if cell_index:
|
|
285
|
+
builder.separator("\t")
|
|
286
|
+
builder.boundary()
|
|
287
|
+
_project_inline(cell, builder)
|
|
288
|
+
builder.boundary()
|
|
289
|
+
else:
|
|
290
|
+
for line_index, line in enumerate(value):
|
|
291
|
+
if line_index:
|
|
292
|
+
builder.separator("\n")
|
|
293
|
+
_project_inline(line, builder)
|
|
294
|
+
return builder.value()
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def project_plain(leaves: Sequence[RenderLeaf]) -> tuple[str, tuple[ProjectedLeaf, ...]]:
|
|
298
|
+
builder = _ProjectionBuilder()
|
|
299
|
+
for leaf in leaves:
|
|
300
|
+
builder.emit(leaf.text, key=leaf.key)
|
|
301
|
+
return builder.value()
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def _single_scalar_lower(value: str) -> str:
|
|
305
|
+
return "".join((lowered if len(lowered := scalar.lower()) == 1 else scalar) for scalar in value)
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def iter_literal_ranges(
|
|
309
|
+
text: str, query: str, *, case_sensitive: bool,
|
|
310
|
+
) -> Iterator[FindRange]:
|
|
311
|
+
if not query:
|
|
312
|
+
return
|
|
313
|
+
haystack = text if case_sensitive else _single_scalar_lower(text)
|
|
314
|
+
needle = query if case_sensitive else _single_scalar_lower(query)
|
|
315
|
+
if not needle:
|
|
316
|
+
return
|
|
317
|
+
cursor = 0
|
|
318
|
+
while cursor <= len(haystack) - len(needle):
|
|
319
|
+
found = haystack.find(needle, cursor)
|
|
320
|
+
if found < 0:
|
|
321
|
+
break
|
|
322
|
+
yield FindRange(found, found + len(needle))
|
|
323
|
+
cursor = found + len(needle)
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def literal_ranges(text: str, query: str, *, case_sensitive: bool) -> tuple[FindRange, ...]:
|
|
327
|
+
return tuple(iter_literal_ranges(text, query, case_sensitive=case_sensitive))
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def iter_regex_ranges(text: str, pattern: re.Pattern[str]) -> Iterator[FindRange]:
|
|
331
|
+
for match in pattern.finditer(text):
|
|
332
|
+
if match.end() > match.start():
|
|
333
|
+
yield FindRange(match.start(), match.end())
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
def regex_ranges(text: str, pattern: re.Pattern[str]) -> tuple[FindRange, ...]:
|
|
337
|
+
return tuple(iter_regex_ranges(text, pattern))
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def slice_range_to_leaves(
|
|
341
|
+
match: FindRange,
|
|
342
|
+
leaves: Sequence[ProjectedLeaf],
|
|
343
|
+
) -> tuple[LeafFragment, ...]:
|
|
344
|
+
fragments: list[LeafFragment] = []
|
|
345
|
+
for leaf in leaves:
|
|
346
|
+
start = max(match.start, leaf.start)
|
|
347
|
+
end = min(match.end, leaf.end)
|
|
348
|
+
if end <= start:
|
|
349
|
+
continue
|
|
350
|
+
fragments.append(LeafFragment(
|
|
351
|
+
leaf_key=leaf.key,
|
|
352
|
+
start=start - leaf.start,
|
|
353
|
+
end=end - leaf.start,
|
|
354
|
+
))
|
|
355
|
+
return tuple(fragments)
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
__all__ = [
|
|
359
|
+
"FindRange",
|
|
360
|
+
"LeafFragment",
|
|
361
|
+
"ProjectedLeaf",
|
|
362
|
+
"RenderLeaf",
|
|
363
|
+
"literal_ranges",
|
|
364
|
+
"iter_literal_ranges",
|
|
365
|
+
"iter_regex_ranges",
|
|
366
|
+
"project_markdown",
|
|
367
|
+
"project_plain",
|
|
368
|
+
"regex_ranges",
|
|
369
|
+
"slice_range_to_leaves",
|
|
370
|
+
]
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
"""#463 S3 — the closed line reader for Codex tool-output harness preambles.
|
|
2
|
+
|
|
3
|
+
Pure kernel: no I/O, no DB, no config, and nothing imported beyond ``re``.
|
|
4
|
+
|
|
5
|
+
Replaces ``_HARNESS_STATUS_RE``, which was one regex written against one assumed
|
|
6
|
+
shape. Executed against every retained tool output on a read-only copy of the
|
|
7
|
+
production store it examined 60,862 outputs, of which 39,942 carry the exact
|
|
8
|
+
``Script completed`` / ``Script failed`` preamble it targets and **0 match**: it
|
|
9
|
+
required a blank line before ``Output:`` and end-of-string after it, while the
|
|
10
|
+
harness writes a single newline and a trailing newline. Five distinct preamble
|
|
11
|
+
grammars exist; that regex targeted one of them.
|
|
12
|
+
|
|
13
|
+
The replacement is a line reader over a CLOSED vocabulary rather than one regex
|
|
14
|
+
per grammar, deliberately (spec section 4.1). The defect being fixed is a
|
|
15
|
+
pattern written against an assumed shape, and the grammar set was mis-described
|
|
16
|
+
twice while the design was written. A closed vocabulary degrades to ``unknown``
|
|
17
|
+
on an arrangement it has not seen, instead of silently matching nothing.
|
|
18
|
+
|
|
19
|
+
Two rules bound what may be consumed:
|
|
20
|
+
|
|
21
|
+
* **Anchored, never searched.** Matching starts at position zero, because the
|
|
22
|
+
preamble is positionally guaranteed and a search would match a user's own
|
|
23
|
+
output somewhere in the middle of a real result.
|
|
24
|
+
* **Terminated by an ``Output:`` line.** All five observed grammars end that
|
|
25
|
+
way. Without the terminator a result that legitimately begins with
|
|
26
|
+
``Exit code: 0`` would lose its first line; with it, the 10,177 outputs that
|
|
27
|
+
carry no preamble are untouched by construction.
|
|
28
|
+
|
|
29
|
+
Any malformed or over-limit field makes the WHOLE preamble unrecognized rather
|
|
30
|
+
than partially parsed, so a near-match degrades to today's behaviour instead of
|
|
31
|
+
stripping a line it did not understand.
|
|
32
|
+
"""
|
|
33
|
+
from __future__ import annotations
|
|
34
|
+
|
|
35
|
+
import re
|
|
36
|
+
|
|
37
|
+
# At most this many lines and this many characters are examined before an
|
|
38
|
+
# `Output:` line. Reaching either limit first leaves the text untouched.
|
|
39
|
+
_MAX_PREAMBLE_LINES = 8
|
|
40
|
+
_MAX_PREAMBLE_CHARS = 512
|
|
41
|
+
|
|
42
|
+
_TERMINATOR = "Output:"
|
|
43
|
+
|
|
44
|
+
# Value syntaxes (spec section 4.1). A token is 1-64 characters from
|
|
45
|
+
# `[A-Za-z0-9_-]`; an integer is 1-10 digits with an optional leading `-`; a wall
|
|
46
|
+
# time is a decimal number with a unit from a closed set.
|
|
47
|
+
_TOKEN = r"[A-Za-z0-9_-]{1,64}"
|
|
48
|
+
_INT = r"-?\d{1,10}"
|
|
49
|
+
_UNSIGNED_INT = r"\d{1,10}"
|
|
50
|
+
_NUMBER = r"\d+(?:\.\d+)?"
|
|
51
|
+
_UNITS = r"seconds|second|ms"
|
|
52
|
+
|
|
53
|
+
_CHUNK_ID_RE = re.compile(rf"Chunk ID: ({_TOKEN})")
|
|
54
|
+
_EXIT_CODE_RE = re.compile(rf"Exit code: ({_INT})")
|
|
55
|
+
_WALL_TIME_RE = re.compile(rf"Wall time:? ({_NUMBER}) ({_UNITS})")
|
|
56
|
+
_PROCESS_EXITED_RE = re.compile(rf"Process exited with code ({_INT})")
|
|
57
|
+
_PROCESS_RUNNING_RE = re.compile(rf"Process running with session ID ({_TOKEN})")
|
|
58
|
+
_TOKEN_COUNT_RE = re.compile(rf"Original token count: ({_UNSIGNED_INT})")
|
|
59
|
+
|
|
60
|
+
_MILLISECOND_UNITS = frozenset({"ms"})
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class _Reject(Exception):
|
|
64
|
+
"""The line vocabulary refused a line; the whole preamble is unrecognized."""
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _wall_time_seconds(number: str, unit: str) -> float:
|
|
68
|
+
value = float(number)
|
|
69
|
+
return value / 1000.0 if unit in _MILLISECOND_UNITS else value
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _read_line(line: str, observed: dict) -> bool:
|
|
73
|
+
"""Record one recognized preamble line, or raise ``_Reject``.
|
|
74
|
+
|
|
75
|
+
Returns True when the line carried a FIELD (so a bare terminator with nothing
|
|
76
|
+
before it can be refused) and False for a blank separator.
|
|
77
|
+
"""
|
|
78
|
+
if line == "":
|
|
79
|
+
# A blank separator carries no field, so tolerating it cannot mis-parse
|
|
80
|
+
# one, and the terminator plus the two caps still bound what is consumed.
|
|
81
|
+
# Zero production outputs carry it, but the shipped `session-b-card-wire`
|
|
82
|
+
# fixture does, and a reader that refused it would silently regress that
|
|
83
|
+
# fixture's resolved status to `unknown`.
|
|
84
|
+
return False
|
|
85
|
+
if line in ("Script completed", "Script failed"):
|
|
86
|
+
observed["script"] = "completed" if line.endswith("completed") else "failed"
|
|
87
|
+
return True
|
|
88
|
+
match = _CHUNK_ID_RE.fullmatch(line)
|
|
89
|
+
if match is not None:
|
|
90
|
+
return True # deliberately not published (section 4.3)
|
|
91
|
+
match = _EXIT_CODE_RE.fullmatch(line)
|
|
92
|
+
if match is not None:
|
|
93
|
+
observed["exit_code"] = int(match.group(1))
|
|
94
|
+
return True
|
|
95
|
+
match = _WALL_TIME_RE.fullmatch(line)
|
|
96
|
+
if match is not None:
|
|
97
|
+
observed["wall_time_seconds"] = _wall_time_seconds(
|
|
98
|
+
match.group(1), match.group(2))
|
|
99
|
+
return True
|
|
100
|
+
match = _PROCESS_EXITED_RE.fullmatch(line)
|
|
101
|
+
if match is not None:
|
|
102
|
+
observed["exit_code"] = int(match.group(1))
|
|
103
|
+
return True
|
|
104
|
+
match = _PROCESS_RUNNING_RE.fullmatch(line)
|
|
105
|
+
if match is not None:
|
|
106
|
+
observed["session_announcement"] = match.group(1)
|
|
107
|
+
observed["running"] = True
|
|
108
|
+
return True
|
|
109
|
+
match = _TOKEN_COUNT_RE.fullmatch(line)
|
|
110
|
+
if match is not None:
|
|
111
|
+
return True
|
|
112
|
+
raise _Reject(line)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _resolve_status(observed: dict) -> str:
|
|
116
|
+
"""Spec section 4.2, in priority order.
|
|
117
|
+
|
|
118
|
+
``running`` is explicitly NOT an error: 4,585 outputs are open sessions, and
|
|
119
|
+
treating them as failures would be worse than the present silence.
|
|
120
|
+
"""
|
|
121
|
+
if observed.get("script") == "failed":
|
|
122
|
+
return "failed"
|
|
123
|
+
if "exit_code" in observed:
|
|
124
|
+
return "completed" if observed["exit_code"] == 0 else "failed"
|
|
125
|
+
if observed.get("running"):
|
|
126
|
+
return "running"
|
|
127
|
+
if observed.get("script") == "completed":
|
|
128
|
+
return "completed"
|
|
129
|
+
return "unknown"
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def parse_harness_preamble(text: str) -> tuple[dict, str] | None:
|
|
133
|
+
"""``(fields, remainder)`` for a recognized preamble, else ``None``.
|
|
134
|
+
|
|
135
|
+
``fields`` carries ``status`` (one of ``completed``, ``failed``, ``running``,
|
|
136
|
+
``unknown``), ``exit_code``, ``wall_time_seconds`` and
|
|
137
|
+
``session_announcement``. ``remainder`` is ``text`` with the consumed run,
|
|
138
|
+
terminator included, removed.
|
|
139
|
+
|
|
140
|
+
``None`` means the caller leaves the text exactly as it found it.
|
|
141
|
+
|
|
142
|
+
``session_announcement`` carries the provider's raw session id for the
|
|
143
|
+
conversation-level session index ONLY. It is never published on a card:
|
|
144
|
+
the reader sees a conversation-local ordinal instead (spec section 4.3).
|
|
145
|
+
"""
|
|
146
|
+
if not isinstance(text, str) or not text:
|
|
147
|
+
return None
|
|
148
|
+
observed: dict = {}
|
|
149
|
+
consumed = 0
|
|
150
|
+
field_lines = 0
|
|
151
|
+
for index, raw in enumerate(text.split("\n")):
|
|
152
|
+
line = raw[:-1] if raw.endswith("\r") else raw
|
|
153
|
+
consumed += len(raw) + 1
|
|
154
|
+
if index >= _MAX_PREAMBLE_LINES or consumed > _MAX_PREAMBLE_CHARS:
|
|
155
|
+
return None
|
|
156
|
+
if line == _TERMINATOR:
|
|
157
|
+
# A bare terminator with no field before it is not a preamble — it is
|
|
158
|
+
# a tool's own first line, and consuming it would delete real output.
|
|
159
|
+
if field_lines == 0:
|
|
160
|
+
return None
|
|
161
|
+
fields = {
|
|
162
|
+
"status": _resolve_status(observed),
|
|
163
|
+
"exit_code": observed.get("exit_code"),
|
|
164
|
+
"wall_time_seconds": observed.get("wall_time_seconds"),
|
|
165
|
+
"session_announcement": observed.get("session_announcement"),
|
|
166
|
+
}
|
|
167
|
+
return fields, text[consumed:]
|
|
168
|
+
try:
|
|
169
|
+
if _read_line(line, observed):
|
|
170
|
+
field_lines += 1
|
|
171
|
+
except _Reject:
|
|
172
|
+
return None
|
|
173
|
+
return None # ran out of text before a terminator
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
__all__ = ["parse_harness_preamble"]
|
package/bin/_lib_codex_hooks.py
CHANGED
|
@@ -57,14 +57,16 @@ def _command_tokens(command: object) -> list[str] | None:
|
|
|
57
57
|
|
|
58
58
|
def is_owned_codex_hook_command(command: object, binary: str) -> bool:
|
|
59
59
|
# The absolute installed path changes across package-manager upgrades, so
|
|
60
|
-
# ownership is the exact argument tail plus
|
|
61
|
-
#
|
|
60
|
+
# ownership is the exact argument tail plus either the canonical ``cctally``
|
|
61
|
+
# basename or today's resolved hook-target basename. The latter admits the
|
|
62
|
+
# npm shim while still tolerating version-directory changes.
|
|
62
63
|
tokens = _command_tokens(command)
|
|
64
|
+
owned_basenames = {"cctally", pathlib.Path(binary).name}
|
|
63
65
|
return bool(
|
|
64
66
|
tokens
|
|
65
67
|
and len(tokens) == 5
|
|
66
68
|
and pathlib.Path(tokens[0]).is_absolute()
|
|
67
|
-
and pathlib.Path(tokens[0]).name
|
|
69
|
+
and pathlib.Path(tokens[0]).name in owned_basenames
|
|
68
70
|
and tokens[1:] == ["hook-tick", "--foreground", "--source", "codex"]
|
|
69
71
|
)
|
|
70
72
|
|