cctally 1.91.0 → 1.92.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +51 -0
- package/README.md +4 -2
- package/bin/_cctally_cache.py +903 -74
- package/bin/_cctally_config.py +57 -0
- package/bin/_cctally_core.py +94 -14
- package/bin/_cctally_dashboard.py +217 -19
- package/bin/_cctally_dashboard_conversation.py +170 -20
- package/bin/_cctally_dashboard_envelope.py +2 -0
- package/bin/_cctally_db.py +481 -19
- package/bin/_cctally_doctor.py +18 -1
- package/bin/_cctally_journal.py +1156 -21
- package/bin/_cctally_journal_repair.py +6 -0
- package/bin/_cctally_parser.py +26 -0
- package/bin/_cctally_quota.py +171 -55
- package/bin/_cctally_record.py +13 -1
- package/bin/_cctally_rederive.py +4 -0
- package/bin/_cctally_statusline.py +6 -6
- package/bin/_cctally_store.py +1061 -40
- package/bin/_cctally_transcript.py +32 -2
- package/bin/_cctally_tui.py +54 -6
- package/bin/_lib_cache_report.py +8 -3
- package/bin/_lib_codex_conversation.py +851 -81
- package/bin/_lib_codex_conversation_query.py +2031 -96
- package/bin/_lib_codex_find_projection.py +517 -0
- package/bin/_lib_codex_harness_preamble.py +176 -0
- package/bin/_lib_codex_hooks.py +5 -3
- package/bin/_lib_codex_js_scan.py +254 -0
- package/bin/_lib_codex_landmarks.py +309 -0
- package/bin/_lib_codex_title_clean.py +116 -0
- package/bin/_lib_conversation_dispatch.py +168 -22
- package/bin/_lib_conversation_query.py +62 -2
- package/bin/_lib_conversation_watch.py +4 -2
- package/bin/_lib_doctor.py +64 -0
- package/bin/_lib_quota_alert_axes.py +31 -34
- package/bin/_lib_stats_damage.py +523 -0
- package/bin/_lib_stats_publish.py +243 -0
- package/bin/cctally +17 -3
- package/dashboard/static/assets/index-Dat-mza6.js +97 -0
- package/dashboard/static/assets/{index-Dwirao3Y.css → index-DnWdv8um.css} +1 -1
- package/dashboard/static/dashboard.html +2 -2
- package/package.json +8 -1
- package/dashboard/static/assets/index-CILAoEja.js +0 -90
|
@@ -0,0 +1,517 @@
|
|
|
1
|
+
"""Canonical visible-text projection and matching for Codex conversation find.
|
|
2
|
+
|
|
3
|
+
The module is deliberately stdlib-only. Its Markdown scanner models the
|
|
4
|
+
visible text-node boundaries used by the dashboard's ReactMarkdown/remark-gfm
|
|
5
|
+
surface; it is not an HTML renderer and never interprets raw HTML.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
import html
|
|
11
|
+
import re
|
|
12
|
+
from typing import Iterable, Iterator, Sequence
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
CODEX_FIND_PROJECTION_VERSION = 2
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass(frozen=True)
|
|
19
|
+
class RenderLeaf:
|
|
20
|
+
key: str
|
|
21
|
+
text: str
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True)
|
|
25
|
+
class ProjectedLeaf:
|
|
26
|
+
key: str
|
|
27
|
+
start: int
|
|
28
|
+
end: int
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass(frozen=True)
|
|
32
|
+
class FindRange:
|
|
33
|
+
start: int
|
|
34
|
+
end: int
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass(frozen=True)
|
|
38
|
+
class LeafFragment:
|
|
39
|
+
leaf_key: str
|
|
40
|
+
start: int
|
|
41
|
+
end: int
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class _ProjectionBuilder:
|
|
45
|
+
def __init__(self) -> None:
|
|
46
|
+
self.parts: list[str] = []
|
|
47
|
+
self.leaves: list[dict[str, int | str]] = []
|
|
48
|
+
self.length = 0
|
|
49
|
+
self._open_leaf: int | None = None
|
|
50
|
+
|
|
51
|
+
def boundary(self) -> None:
|
|
52
|
+
self._open_leaf = None
|
|
53
|
+
|
|
54
|
+
def separator(self, value: str) -> None:
|
|
55
|
+
if not value:
|
|
56
|
+
return
|
|
57
|
+
self.boundary()
|
|
58
|
+
self.parts.append(value)
|
|
59
|
+
self.length += len(value)
|
|
60
|
+
|
|
61
|
+
def emit(self, value: str, *, boundary: bool = False, key: str | None = None) -> None:
|
|
62
|
+
if not value:
|
|
63
|
+
return
|
|
64
|
+
if boundary:
|
|
65
|
+
self.boundary()
|
|
66
|
+
start = self.length
|
|
67
|
+
self.parts.append(value)
|
|
68
|
+
self.length += len(value)
|
|
69
|
+
if key is not None:
|
|
70
|
+
self.leaves.append({"key": key, "start": start, "end": self.length})
|
|
71
|
+
self._open_leaf = None
|
|
72
|
+
return
|
|
73
|
+
if self._open_leaf is None:
|
|
74
|
+
self._open_leaf = len(self.leaves)
|
|
75
|
+
self.leaves.append({
|
|
76
|
+
"key": f"t{self._open_leaf}",
|
|
77
|
+
"start": start,
|
|
78
|
+
"end": self.length,
|
|
79
|
+
})
|
|
80
|
+
else:
|
|
81
|
+
self.leaves[self._open_leaf]["end"] = self.length
|
|
82
|
+
|
|
83
|
+
def value(self) -> tuple[str, tuple[ProjectedLeaf, ...]]:
|
|
84
|
+
return (
|
|
85
|
+
"".join(self.parts),
|
|
86
|
+
tuple(ProjectedLeaf(**leaf) for leaf in self.leaves),
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
_TABLE_DELIMITER_RE = re.compile(
|
|
91
|
+
r"^\s*\|?\s*:?-{3,}:?\s*(?:\|\s*:?-{3,}:?\s*)+\|?\s*$"
|
|
92
|
+
)
|
|
93
|
+
_BLOCK_PREFIX_RE = re.compile(
|
|
94
|
+
r"^\s*(?:(?:#{1,6})\s+|>\s?|(?:[-+*]|\d+[.)])\s+)"
|
|
95
|
+
)
|
|
96
|
+
_TASK_MARKER_RE = re.compile(r"^\[[ xX]\]\s+")
|
|
97
|
+
_AUTOLINK_RE = re.compile(r"<((?:https?://|mailto:)[^ <>]+|[^ <>@]+@[^ <>@]+)>")
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _table_cells(line: str) -> list[str]:
|
|
101
|
+
value = line.strip()
|
|
102
|
+
if value.startswith("|"):
|
|
103
|
+
value = value[1:]
|
|
104
|
+
if value.endswith("|"):
|
|
105
|
+
value = value[:-1]
|
|
106
|
+
cells: list[str] = []
|
|
107
|
+
current: list[str] = []
|
|
108
|
+
escaped = False
|
|
109
|
+
for char in value:
|
|
110
|
+
if escaped:
|
|
111
|
+
current.append(char)
|
|
112
|
+
escaped = False
|
|
113
|
+
elif char == "\\":
|
|
114
|
+
current.append(char)
|
|
115
|
+
escaped = True
|
|
116
|
+
elif char == "|":
|
|
117
|
+
cells.append("".join(current).strip())
|
|
118
|
+
current = []
|
|
119
|
+
else:
|
|
120
|
+
current.append(char)
|
|
121
|
+
cells.append("".join(current).strip())
|
|
122
|
+
return cells
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _find_closing(source: str, token: str, start: int) -> int:
|
|
126
|
+
cursor = start
|
|
127
|
+
while True:
|
|
128
|
+
found = source.find(token, cursor)
|
|
129
|
+
if found < 0:
|
|
130
|
+
return -1
|
|
131
|
+
backslashes = 0
|
|
132
|
+
probe = found - 1
|
|
133
|
+
while probe >= 0 and source[probe] == "\\":
|
|
134
|
+
backslashes += 1
|
|
135
|
+
probe -= 1
|
|
136
|
+
if backslashes % 2 == 0:
|
|
137
|
+
return found
|
|
138
|
+
cursor = found + len(token)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _project_inline(source: str, builder: _ProjectionBuilder) -> None:
|
|
142
|
+
plain: list[str] = []
|
|
143
|
+
|
|
144
|
+
def flush() -> None:
|
|
145
|
+
if plain:
|
|
146
|
+
builder.emit(html.unescape("".join(plain)))
|
|
147
|
+
plain.clear()
|
|
148
|
+
|
|
149
|
+
cursor = 0
|
|
150
|
+
while cursor < len(source):
|
|
151
|
+
if source[cursor] == "\\" and cursor + 1 < len(source):
|
|
152
|
+
plain.append(source[cursor + 1])
|
|
153
|
+
cursor += 2
|
|
154
|
+
continue
|
|
155
|
+
|
|
156
|
+
if source[cursor] == "`":
|
|
157
|
+
run = 1
|
|
158
|
+
while cursor + run < len(source) and source[cursor + run] == "`":
|
|
159
|
+
run += 1
|
|
160
|
+
token = "`" * run
|
|
161
|
+
close = _find_closing(source, token, cursor + run)
|
|
162
|
+
if close >= 0:
|
|
163
|
+
flush()
|
|
164
|
+
builder.emit(source[cursor + run:close].strip(" "), boundary=True)
|
|
165
|
+
builder.boundary()
|
|
166
|
+
cursor = close + run
|
|
167
|
+
continue
|
|
168
|
+
|
|
169
|
+
if source.startswith("![", cursor) or source[cursor] == "[":
|
|
170
|
+
image = source.startswith("
|
|
173
|
+
if label_end >= 0:
|
|
174
|
+
destination_end = source.find(")", label_end + 2)
|
|
175
|
+
if destination_end >= 0:
|
|
176
|
+
flush()
|
|
177
|
+
builder.boundary()
|
|
178
|
+
if not image:
|
|
179
|
+
_project_inline(source[label_start:label_end], builder)
|
|
180
|
+
builder.boundary()
|
|
181
|
+
cursor = destination_end + 1
|
|
182
|
+
continue
|
|
183
|
+
|
|
184
|
+
if source[cursor] == "<":
|
|
185
|
+
autolink = _AUTOLINK_RE.match(source, cursor)
|
|
186
|
+
if autolink is not None:
|
|
187
|
+
flush()
|
|
188
|
+
builder.boundary()
|
|
189
|
+
label = autolink.group(1)
|
|
190
|
+
builder.emit(label[7:] if label.startswith("mailto:") else label)
|
|
191
|
+
builder.boundary()
|
|
192
|
+
cursor = autolink.end()
|
|
193
|
+
continue
|
|
194
|
+
|
|
195
|
+
matched_delimiter = False
|
|
196
|
+
for token in ("**", "__", "~~", "*", "_"):
|
|
197
|
+
if not source.startswith(token, cursor):
|
|
198
|
+
continue
|
|
199
|
+
close = _find_closing(source, token, cursor + len(token))
|
|
200
|
+
if close < 0 or close == cursor + len(token):
|
|
201
|
+
continue
|
|
202
|
+
flush()
|
|
203
|
+
builder.boundary()
|
|
204
|
+
_project_inline(source[cursor + len(token):close], builder)
|
|
205
|
+
builder.boundary()
|
|
206
|
+
cursor = close + len(token)
|
|
207
|
+
matched_delimiter = True
|
|
208
|
+
break
|
|
209
|
+
if matched_delimiter:
|
|
210
|
+
continue
|
|
211
|
+
|
|
212
|
+
plain.append(source[cursor])
|
|
213
|
+
cursor += 1
|
|
214
|
+
flush()
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _strip_block_prefix(line: str) -> str:
|
|
218
|
+
value = _BLOCK_PREFIX_RE.sub("", line, count=1)
|
|
219
|
+
if _TASK_MARKER_RE.match(value):
|
|
220
|
+
value = _TASK_MARKER_RE.sub(" ", value, count=1)
|
|
221
|
+
if value.endswith(" "):
|
|
222
|
+
value = value[:-2]
|
|
223
|
+
elif value.endswith("\\"):
|
|
224
|
+
value = value[:-1]
|
|
225
|
+
return value
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def project_markdown(source: str) -> tuple[str, tuple[ProjectedLeaf, ...]]:
|
|
229
|
+
builder = _ProjectionBuilder()
|
|
230
|
+
lines = source.replace("\r\n", "\n").replace("\r", "\n").split("\n")
|
|
231
|
+
blocks: list[tuple[str, object]] = []
|
|
232
|
+
cursor = 0
|
|
233
|
+
while cursor < len(lines):
|
|
234
|
+
line = lines[cursor]
|
|
235
|
+
if not line.strip():
|
|
236
|
+
cursor += 1
|
|
237
|
+
continue
|
|
238
|
+
|
|
239
|
+
fence = re.match(r"^\s*(`{3,}|~{3,})(?:[^`]*)$", line)
|
|
240
|
+
if fence:
|
|
241
|
+
token = fence.group(1)
|
|
242
|
+
body: list[str] = []
|
|
243
|
+
cursor += 1
|
|
244
|
+
while cursor < len(lines) and not re.match(
|
|
245
|
+
rf"^\s*{re.escape(token[0])}{{{len(token)},}}\s*$", lines[cursor]
|
|
246
|
+
):
|
|
247
|
+
body.append(lines[cursor])
|
|
248
|
+
cursor += 1
|
|
249
|
+
closed = cursor < len(lines)
|
|
250
|
+
if closed:
|
|
251
|
+
cursor += 1
|
|
252
|
+
code = "\n".join(body)
|
|
253
|
+
if body:
|
|
254
|
+
code += "\n"
|
|
255
|
+
blocks.append(("code", code))
|
|
256
|
+
continue
|
|
257
|
+
|
|
258
|
+
if cursor + 1 < len(lines) and "|" in line and _TABLE_DELIMITER_RE.match(lines[cursor + 1]):
|
|
259
|
+
rows = [_table_cells(line)]
|
|
260
|
+
cursor += 2
|
|
261
|
+
while cursor < len(lines) and lines[cursor].strip() and "|" in lines[cursor]:
|
|
262
|
+
rows.append(_table_cells(lines[cursor]))
|
|
263
|
+
cursor += 1
|
|
264
|
+
blocks.append(("table", rows))
|
|
265
|
+
continue
|
|
266
|
+
|
|
267
|
+
paragraph = [_strip_block_prefix(line)]
|
|
268
|
+
cursor += 1
|
|
269
|
+
while cursor < len(lines) and lines[cursor].strip():
|
|
270
|
+
if re.match(r"^\s*(`{3,}|~{3,})", lines[cursor]):
|
|
271
|
+
break
|
|
272
|
+
paragraph.append(_strip_block_prefix(lines[cursor]))
|
|
273
|
+
cursor += 1
|
|
274
|
+
blocks.append(("paragraph", paragraph))
|
|
275
|
+
|
|
276
|
+
for block_index, (kind, value) in enumerate(blocks):
|
|
277
|
+
if block_index:
|
|
278
|
+
builder.separator("\n")
|
|
279
|
+
if kind == "code":
|
|
280
|
+
builder.emit(str(value), boundary=True)
|
|
281
|
+
builder.boundary()
|
|
282
|
+
elif kind == "table":
|
|
283
|
+
for row_index, row in enumerate(value):
|
|
284
|
+
if row_index:
|
|
285
|
+
builder.separator("\n")
|
|
286
|
+
for cell_index, cell in enumerate(row):
|
|
287
|
+
if cell_index:
|
|
288
|
+
builder.separator("\t")
|
|
289
|
+
builder.boundary()
|
|
290
|
+
_project_inline(cell, builder)
|
|
291
|
+
builder.boundary()
|
|
292
|
+
else:
|
|
293
|
+
for line_index, line in enumerate(value):
|
|
294
|
+
if line_index:
|
|
295
|
+
builder.separator("\n")
|
|
296
|
+
_project_inline(line, builder)
|
|
297
|
+
return builder.value()
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def project_plain(leaves: Sequence[RenderLeaf]) -> tuple[str, tuple[ProjectedLeaf, ...]]:
|
|
301
|
+
builder = _ProjectionBuilder()
|
|
302
|
+
for leaf in leaves:
|
|
303
|
+
builder.emit(leaf.text, key=leaf.key)
|
|
304
|
+
return builder.value()
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
_CONTEXT_DIFF_GIT_RE = re.compile(r"diff --git a/\S+ b/\S+")
|
|
308
|
+
_CONTEXT_HUNK_RE = re.compile(r"^@@ -(\d+)(?:,\d+)? \+(\d+)(?:,\d+)? @@")
|
|
309
|
+
_CONTEXT_EXTENDED_HEADER_PREFIXES = (
|
|
310
|
+
"old mode ", "new mode ", "new file mode ", "deleted file mode ",
|
|
311
|
+
"rename from ", "rename to ", "copy from ", "copy to ",
|
|
312
|
+
"similarity index ", "dissimilarity index ", "index ",
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
def _context_is_diff_line(line: str) -> bool:
|
|
317
|
+
if _CONTEXT_DIFF_GIT_RE.search(line):
|
|
318
|
+
return True
|
|
319
|
+
if line.startswith(("--- ", "+++ ", "@@")):
|
|
320
|
+
return True
|
|
321
|
+
if line.startswith(_CONTEXT_EXTENDED_HEADER_PREFIXES):
|
|
322
|
+
return True
|
|
323
|
+
return line == "" or line[0] in "+- \\"
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def _segment_context_body(text: str) -> list[tuple[str, str]]:
|
|
327
|
+
"""Mirror ``contextDiff.ts::segmentContextBody`` without rendering HTML."""
|
|
328
|
+
lines = text.split("\n")
|
|
329
|
+
if lines and lines[-1] == "":
|
|
330
|
+
lines.pop()
|
|
331
|
+
segments: list[tuple[str, str]] = []
|
|
332
|
+
prose: list[str] = []
|
|
333
|
+
diff: list[str] = []
|
|
334
|
+
in_diff = False
|
|
335
|
+
|
|
336
|
+
def flush(kind: str, values: list[str]) -> None:
|
|
337
|
+
if values:
|
|
338
|
+
segments.append((kind, "\n".join(values)))
|
|
339
|
+
values.clear()
|
|
340
|
+
|
|
341
|
+
for line in lines:
|
|
342
|
+
if not in_diff:
|
|
343
|
+
match = _CONTEXT_DIFF_GIT_RE.search(line)
|
|
344
|
+
if match is None:
|
|
345
|
+
prose.append(line)
|
|
346
|
+
continue
|
|
347
|
+
before = line[:match.start()].rstrip()
|
|
348
|
+
if before:
|
|
349
|
+
prose.append(before)
|
|
350
|
+
flush("prose", prose)
|
|
351
|
+
in_diff = True
|
|
352
|
+
diff.append(line[match.start():])
|
|
353
|
+
elif _context_is_diff_line(line):
|
|
354
|
+
diff.append(line)
|
|
355
|
+
else:
|
|
356
|
+
flush("diff", diff)
|
|
357
|
+
in_diff = False
|
|
358
|
+
prose.append(line)
|
|
359
|
+
flush("prose", prose)
|
|
360
|
+
flush("diff", diff)
|
|
361
|
+
return segments
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def _context_diff_rows(text: str) -> list[tuple[int, int, int, str]]:
|
|
365
|
+
"""Mirror the visible row walk in ``contextDiff.ts::parseUnifiedDiff``."""
|
|
366
|
+
rows: list[tuple[int, int, int, str]] = []
|
|
367
|
+
file_index = -1
|
|
368
|
+
hunk_index = -1
|
|
369
|
+
row_index = 0
|
|
370
|
+
in_hunk = False
|
|
371
|
+
for line in text.split("\n"):
|
|
372
|
+
if _CONTEXT_DIFF_GIT_RE.search(line):
|
|
373
|
+
file_index += 1
|
|
374
|
+
hunk_index = -1
|
|
375
|
+
row_index = 0
|
|
376
|
+
in_hunk = False
|
|
377
|
+
continue
|
|
378
|
+
if _CONTEXT_HUNK_RE.match(line):
|
|
379
|
+
hunk_index += 1
|
|
380
|
+
row_index = 0
|
|
381
|
+
in_hunk = True
|
|
382
|
+
continue
|
|
383
|
+
if not in_hunk:
|
|
384
|
+
continue
|
|
385
|
+
if line.startswith(("--- ", "+++ ")) or line.startswith(
|
|
386
|
+
_CONTEXT_EXTENDED_HEADER_PREFIXES
|
|
387
|
+
):
|
|
388
|
+
continue
|
|
389
|
+
if line == "" or line.startswith("\\"):
|
|
390
|
+
continue
|
|
391
|
+
rows.append((file_index, hunk_index, row_index, line[1:]))
|
|
392
|
+
row_index += 1
|
|
393
|
+
return rows
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
def _append_projected(
|
|
397
|
+
builder: _ProjectionBuilder,
|
|
398
|
+
projected: tuple[str, tuple[ProjectedLeaf, ...]],
|
|
399
|
+
*,
|
|
400
|
+
prefix: str,
|
|
401
|
+
) -> None:
|
|
402
|
+
text, leaves = projected
|
|
403
|
+
if not text:
|
|
404
|
+
return
|
|
405
|
+
start = builder.length
|
|
406
|
+
builder.parts.append(text)
|
|
407
|
+
builder.length += len(text)
|
|
408
|
+
builder.boundary()
|
|
409
|
+
builder.leaves.extend({
|
|
410
|
+
"key": f"{prefix}/{leaf.key}",
|
|
411
|
+
"start": start + leaf.start,
|
|
412
|
+
"end": start + leaf.end,
|
|
413
|
+
} for leaf in leaves)
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
def project_context(source: str) -> tuple[str, tuple[ProjectedLeaf, ...]]:
|
|
417
|
+
"""Project visible prose and diff-row leaves from one context body.
|
|
418
|
+
|
|
419
|
+
File headers and +/- statistics are derived card chrome, matching #482's
|
|
420
|
+
rule that only provider-authored render leaves enter the search surface.
|
|
421
|
+
"""
|
|
422
|
+
builder = _ProjectionBuilder()
|
|
423
|
+
for segment_index, (kind, text) in enumerate(_segment_context_body(source)):
|
|
424
|
+
if kind == "prose":
|
|
425
|
+
projected = project_markdown(text)
|
|
426
|
+
if not projected[0]:
|
|
427
|
+
continue
|
|
428
|
+
if builder.parts:
|
|
429
|
+
builder.separator("\n")
|
|
430
|
+
_append_projected(
|
|
431
|
+
builder, projected, prefix=f"segments.{segment_index}.prose"
|
|
432
|
+
)
|
|
433
|
+
continue
|
|
434
|
+
for file_index, hunk_index, row_index, row_text in _context_diff_rows(text):
|
|
435
|
+
if not row_text:
|
|
436
|
+
continue
|
|
437
|
+
if builder.parts:
|
|
438
|
+
builder.separator("\n")
|
|
439
|
+
builder.emit(
|
|
440
|
+
row_text,
|
|
441
|
+
key=(
|
|
442
|
+
f"segments.{segment_index}.files.{file_index}."
|
|
443
|
+
f"hunks.{hunk_index}.rows.{row_index}"
|
|
444
|
+
),
|
|
445
|
+
)
|
|
446
|
+
return builder.value()
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
def _single_scalar_lower(value: str) -> str:
|
|
450
|
+
return "".join((lowered if len(lowered := scalar.lower()) == 1 else scalar) for scalar in value)
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
def iter_literal_ranges(
|
|
454
|
+
text: str, query: str, *, case_sensitive: bool,
|
|
455
|
+
) -> Iterator[FindRange]:
|
|
456
|
+
if not query:
|
|
457
|
+
return
|
|
458
|
+
haystack = text if case_sensitive else _single_scalar_lower(text)
|
|
459
|
+
needle = query if case_sensitive else _single_scalar_lower(query)
|
|
460
|
+
if not needle:
|
|
461
|
+
return
|
|
462
|
+
cursor = 0
|
|
463
|
+
while cursor <= len(haystack) - len(needle):
|
|
464
|
+
found = haystack.find(needle, cursor)
|
|
465
|
+
if found < 0:
|
|
466
|
+
break
|
|
467
|
+
yield FindRange(found, found + len(needle))
|
|
468
|
+
cursor = found + len(needle)
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
def literal_ranges(text: str, query: str, *, case_sensitive: bool) -> tuple[FindRange, ...]:
|
|
472
|
+
return tuple(iter_literal_ranges(text, query, case_sensitive=case_sensitive))
|
|
473
|
+
|
|
474
|
+
|
|
475
|
+
def iter_regex_ranges(text: str, pattern: re.Pattern[str]) -> Iterator[FindRange]:
|
|
476
|
+
for match in pattern.finditer(text):
|
|
477
|
+
if match.end() > match.start():
|
|
478
|
+
yield FindRange(match.start(), match.end())
|
|
479
|
+
|
|
480
|
+
|
|
481
|
+
def regex_ranges(text: str, pattern: re.Pattern[str]) -> tuple[FindRange, ...]:
|
|
482
|
+
return tuple(iter_regex_ranges(text, pattern))
|
|
483
|
+
|
|
484
|
+
|
|
485
|
+
def slice_range_to_leaves(
|
|
486
|
+
match: FindRange,
|
|
487
|
+
leaves: Sequence[ProjectedLeaf],
|
|
488
|
+
) -> tuple[LeafFragment, ...]:
|
|
489
|
+
fragments: list[LeafFragment] = []
|
|
490
|
+
for leaf in leaves:
|
|
491
|
+
start = max(match.start, leaf.start)
|
|
492
|
+
end = min(match.end, leaf.end)
|
|
493
|
+
if end <= start:
|
|
494
|
+
continue
|
|
495
|
+
fragments.append(LeafFragment(
|
|
496
|
+
leaf_key=leaf.key,
|
|
497
|
+
start=start - leaf.start,
|
|
498
|
+
end=end - leaf.start,
|
|
499
|
+
))
|
|
500
|
+
return tuple(fragments)
|
|
501
|
+
|
|
502
|
+
|
|
503
|
+
__all__ = [
|
|
504
|
+
"CODEX_FIND_PROJECTION_VERSION",
|
|
505
|
+
"FindRange",
|
|
506
|
+
"LeafFragment",
|
|
507
|
+
"ProjectedLeaf",
|
|
508
|
+
"RenderLeaf",
|
|
509
|
+
"literal_ranges",
|
|
510
|
+
"iter_literal_ranges",
|
|
511
|
+
"iter_regex_ranges",
|
|
512
|
+
"project_markdown",
|
|
513
|
+
"project_context",
|
|
514
|
+
"project_plain",
|
|
515
|
+
"regex_ranges",
|
|
516
|
+
"slice_range_to_leaves",
|
|
517
|
+
]
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
"""#463 S3 — the closed line reader for Codex tool-output harness preambles.
|
|
2
|
+
|
|
3
|
+
Pure kernel: no I/O, no DB, no config, and nothing imported beyond ``re``.
|
|
4
|
+
|
|
5
|
+
Replaces ``_HARNESS_STATUS_RE``, which was one regex written against one assumed
|
|
6
|
+
shape. Executed against every retained tool output on a read-only copy of the
|
|
7
|
+
production store it examined 60,862 outputs, of which 39,942 carry the exact
|
|
8
|
+
``Script completed`` / ``Script failed`` preamble it targets and **0 match**: it
|
|
9
|
+
required a blank line before ``Output:`` and end-of-string after it, while the
|
|
10
|
+
harness writes a single newline and a trailing newline. Five distinct preamble
|
|
11
|
+
grammars exist; that regex targeted one of them.
|
|
12
|
+
|
|
13
|
+
The replacement is a line reader over a CLOSED vocabulary rather than one regex
|
|
14
|
+
per grammar, deliberately (spec section 4.1). The defect being fixed is a
|
|
15
|
+
pattern written against an assumed shape, and the grammar set was mis-described
|
|
16
|
+
twice while the design was written. A closed vocabulary degrades to ``unknown``
|
|
17
|
+
on an arrangement it has not seen, instead of silently matching nothing.
|
|
18
|
+
|
|
19
|
+
Two rules bound what may be consumed:
|
|
20
|
+
|
|
21
|
+
* **Anchored, never searched.** Matching starts at position zero, because the
|
|
22
|
+
preamble is positionally guaranteed and a search would match a user's own
|
|
23
|
+
output somewhere in the middle of a real result.
|
|
24
|
+
* **Terminated by an ``Output:`` line.** All five observed grammars end that
|
|
25
|
+
way. Without the terminator a result that legitimately begins with
|
|
26
|
+
``Exit code: 0`` would lose its first line; with it, the 10,177 outputs that
|
|
27
|
+
carry no preamble are untouched by construction.
|
|
28
|
+
|
|
29
|
+
Any malformed or over-limit field makes the WHOLE preamble unrecognized rather
|
|
30
|
+
than partially parsed, so a near-match degrades to today's behaviour instead of
|
|
31
|
+
stripping a line it did not understand.
|
|
32
|
+
"""
|
|
33
|
+
from __future__ import annotations
|
|
34
|
+
|
|
35
|
+
import re
|
|
36
|
+
|
|
37
|
+
# At most this many lines and this many characters are examined before an
|
|
38
|
+
# `Output:` line. Reaching either limit first leaves the text untouched.
|
|
39
|
+
_MAX_PREAMBLE_LINES = 8
|
|
40
|
+
_MAX_PREAMBLE_CHARS = 512
|
|
41
|
+
|
|
42
|
+
_TERMINATOR = "Output:"
|
|
43
|
+
|
|
44
|
+
# Value syntaxes (spec section 4.1). A token is 1-64 characters from
|
|
45
|
+
# `[A-Za-z0-9_-]`; an integer is 1-10 digits with an optional leading `-`; a wall
|
|
46
|
+
# time is a decimal number with a unit from a closed set.
|
|
47
|
+
_TOKEN = r"[A-Za-z0-9_-]{1,64}"
|
|
48
|
+
_INT = r"-?\d{1,10}"
|
|
49
|
+
_UNSIGNED_INT = r"\d{1,10}"
|
|
50
|
+
_NUMBER = r"\d+(?:\.\d+)?"
|
|
51
|
+
_UNITS = r"seconds|second|ms"
|
|
52
|
+
|
|
53
|
+
_CHUNK_ID_RE = re.compile(rf"Chunk ID: ({_TOKEN})")
|
|
54
|
+
_EXIT_CODE_RE = re.compile(rf"Exit code: ({_INT})")
|
|
55
|
+
_WALL_TIME_RE = re.compile(rf"Wall time:? ({_NUMBER}) ({_UNITS})")
|
|
56
|
+
_PROCESS_EXITED_RE = re.compile(rf"Process exited with code ({_INT})")
|
|
57
|
+
_PROCESS_RUNNING_RE = re.compile(rf"Process running with session ID ({_TOKEN})")
|
|
58
|
+
_TOKEN_COUNT_RE = re.compile(rf"Original token count: ({_UNSIGNED_INT})")
|
|
59
|
+
|
|
60
|
+
_MILLISECOND_UNITS = frozenset({"ms"})
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class _Reject(Exception):
|
|
64
|
+
"""The line vocabulary refused a line; the whole preamble is unrecognized."""
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _wall_time_seconds(number: str, unit: str) -> float:
|
|
68
|
+
value = float(number)
|
|
69
|
+
return value / 1000.0 if unit in _MILLISECOND_UNITS else value
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _read_line(line: str, observed: dict) -> bool:
|
|
73
|
+
"""Record one recognized preamble line, or raise ``_Reject``.
|
|
74
|
+
|
|
75
|
+
Returns True when the line carried a FIELD (so a bare terminator with nothing
|
|
76
|
+
before it can be refused) and False for a blank separator.
|
|
77
|
+
"""
|
|
78
|
+
if line == "":
|
|
79
|
+
# A blank separator carries no field, so tolerating it cannot mis-parse
|
|
80
|
+
# one, and the terminator plus the two caps still bound what is consumed.
|
|
81
|
+
# Zero production outputs carry it, but the shipped `session-b-card-wire`
|
|
82
|
+
# fixture does, and a reader that refused it would silently regress that
|
|
83
|
+
# fixture's resolved status to `unknown`.
|
|
84
|
+
return False
|
|
85
|
+
if line in ("Script completed", "Script failed"):
|
|
86
|
+
observed["script"] = "completed" if line.endswith("completed") else "failed"
|
|
87
|
+
return True
|
|
88
|
+
match = _CHUNK_ID_RE.fullmatch(line)
|
|
89
|
+
if match is not None:
|
|
90
|
+
return True # deliberately not published (section 4.3)
|
|
91
|
+
match = _EXIT_CODE_RE.fullmatch(line)
|
|
92
|
+
if match is not None:
|
|
93
|
+
observed["exit_code"] = int(match.group(1))
|
|
94
|
+
return True
|
|
95
|
+
match = _WALL_TIME_RE.fullmatch(line)
|
|
96
|
+
if match is not None:
|
|
97
|
+
observed["wall_time_seconds"] = _wall_time_seconds(
|
|
98
|
+
match.group(1), match.group(2))
|
|
99
|
+
return True
|
|
100
|
+
match = _PROCESS_EXITED_RE.fullmatch(line)
|
|
101
|
+
if match is not None:
|
|
102
|
+
observed["exit_code"] = int(match.group(1))
|
|
103
|
+
return True
|
|
104
|
+
match = _PROCESS_RUNNING_RE.fullmatch(line)
|
|
105
|
+
if match is not None:
|
|
106
|
+
observed["session_announcement"] = match.group(1)
|
|
107
|
+
observed["running"] = True
|
|
108
|
+
return True
|
|
109
|
+
match = _TOKEN_COUNT_RE.fullmatch(line)
|
|
110
|
+
if match is not None:
|
|
111
|
+
return True
|
|
112
|
+
raise _Reject(line)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _resolve_status(observed: dict) -> str:
|
|
116
|
+
"""Spec section 4.2, in priority order.
|
|
117
|
+
|
|
118
|
+
``running`` is explicitly NOT an error: 4,585 outputs are open sessions, and
|
|
119
|
+
treating them as failures would be worse than the present silence.
|
|
120
|
+
"""
|
|
121
|
+
if observed.get("script") == "failed":
|
|
122
|
+
return "failed"
|
|
123
|
+
if "exit_code" in observed:
|
|
124
|
+
return "completed" if observed["exit_code"] == 0 else "failed"
|
|
125
|
+
if observed.get("running"):
|
|
126
|
+
return "running"
|
|
127
|
+
if observed.get("script") == "completed":
|
|
128
|
+
return "completed"
|
|
129
|
+
return "unknown"
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def parse_harness_preamble(text: str) -> tuple[dict, str] | None:
|
|
133
|
+
"""``(fields, remainder)`` for a recognized preamble, else ``None``.
|
|
134
|
+
|
|
135
|
+
``fields`` carries ``status`` (one of ``completed``, ``failed``, ``running``,
|
|
136
|
+
``unknown``), ``exit_code``, ``wall_time_seconds`` and
|
|
137
|
+
``session_announcement``. ``remainder`` is ``text`` with the consumed run,
|
|
138
|
+
terminator included, removed.
|
|
139
|
+
|
|
140
|
+
``None`` means the caller leaves the text exactly as it found it.
|
|
141
|
+
|
|
142
|
+
``session_announcement`` carries the provider's raw session id for the
|
|
143
|
+
conversation-level session index ONLY. It is never published on a card:
|
|
144
|
+
the reader sees a conversation-local ordinal instead (spec section 4.3).
|
|
145
|
+
"""
|
|
146
|
+
if not isinstance(text, str) or not text:
|
|
147
|
+
return None
|
|
148
|
+
observed: dict = {}
|
|
149
|
+
consumed = 0
|
|
150
|
+
field_lines = 0
|
|
151
|
+
for index, raw in enumerate(text.split("\n")):
|
|
152
|
+
line = raw[:-1] if raw.endswith("\r") else raw
|
|
153
|
+
consumed += len(raw) + 1
|
|
154
|
+
if index >= _MAX_PREAMBLE_LINES or consumed > _MAX_PREAMBLE_CHARS:
|
|
155
|
+
return None
|
|
156
|
+
if line == _TERMINATOR:
|
|
157
|
+
# A bare terminator with no field before it is not a preamble — it is
|
|
158
|
+
# a tool's own first line, and consuming it would delete real output.
|
|
159
|
+
if field_lines == 0:
|
|
160
|
+
return None
|
|
161
|
+
fields = {
|
|
162
|
+
"status": _resolve_status(observed),
|
|
163
|
+
"exit_code": observed.get("exit_code"),
|
|
164
|
+
"wall_time_seconds": observed.get("wall_time_seconds"),
|
|
165
|
+
"session_announcement": observed.get("session_announcement"),
|
|
166
|
+
}
|
|
167
|
+
return fields, text[consumed:]
|
|
168
|
+
try:
|
|
169
|
+
if _read_line(line, observed):
|
|
170
|
+
field_lines += 1
|
|
171
|
+
except _Reject:
|
|
172
|
+
return None
|
|
173
|
+
return None # ran out of text before a terminator
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
__all__ = ["parse_harness_preamble"]
|