cctally 1.90.1 → 1.92.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +74 -0
- package/README.md +2 -2
- package/bin/_cctally_cache.py +863 -74
- package/bin/_cctally_config.py +57 -0
- package/bin/_cctally_core.py +53 -8
- package/bin/_cctally_dashboard.py +146 -5
- package/bin/_cctally_dashboard_conversation.py +164 -18
- package/bin/_cctally_dashboard_envelope.py +69 -12
- package/bin/_cctally_dashboard_sources.py +27 -1
- package/bin/_cctally_db.py +372 -10
- package/bin/_cctally_doctor.py +18 -1
- package/bin/_cctally_journal.py +535 -13
- package/bin/_cctally_journal_repair.py +6 -0
- package/bin/_cctally_parser.py +6 -0
- package/bin/_cctally_quota.py +171 -55
- package/bin/_cctally_record.py +13 -1
- package/bin/_cctally_rederive.py +4 -0
- package/bin/_cctally_store.py +311 -6
- package/bin/_cctally_transcript.py +32 -2
- package/bin/_lib_cache_report.py +8 -3
- package/bin/_lib_cache_report_wire.py +8 -20
- package/bin/_lib_codex_conversation.py +959 -81
- package/bin/_lib_codex_conversation_query.py +2792 -167
- package/bin/_lib_codex_find_projection.py +370 -0
- package/bin/_lib_codex_harness_preamble.py +176 -0
- package/bin/_lib_codex_hooks.py +5 -3
- package/bin/_lib_codex_js_scan.py +254 -0
- package/bin/_lib_codex_landmarks.py +309 -0
- package/bin/_lib_codex_reasoning_headings.py +73 -0
- package/bin/_lib_codex_segments.py +259 -0
- package/bin/_lib_codex_title_clean.py +116 -0
- package/bin/_lib_conversation_dispatch.py +153 -21
- package/bin/_lib_conversation_watch.py +4 -2
- package/bin/_lib_dashboard_sources.py +33 -32
- package/bin/_lib_doctor.py +64 -0
- package/bin/_lib_quota_alert_axes.py +31 -34
- package/bin/_lib_stats_damage.py +523 -0
- package/bin/cctally +5 -0
- package/dashboard/static/assets/index-BEzzJtUd.js +97 -0
- package/dashboard/static/assets/index-DnWdv8um.css +1 -0
- package/dashboard/static/dashboard.html +2 -2
- package/package.json +9 -1
- package/dashboard/static/assets/index-Bar8-S1i.css +0 -1
- package/dashboard/static/assets/index-CRogVlEC.js +0 -92
|
@@ -0,0 +1,259 @@
|
|
|
1
|
+
"""#463 S1 — the pure segmentation kernel for Codex conversation turns.
|
|
2
|
+
|
|
3
|
+
A **segment** is a run of consecutive fold groups inside one ``klass ==
|
|
4
|
+
"response"`` canonical item, taken greedily from the start of the turn, closed
|
|
5
|
+
when the block budget is reached, and closed earlier when a semantic boundary
|
|
6
|
+
falls inside the budget window. Items whose class is not ``response`` already
|
|
7
|
+
contain a single row; each is exactly one segment and its key does not change.
|
|
8
|
+
|
|
9
|
+
**The unit is a fold group, not a row, and that is not a detail.**
|
|
10
|
+
``_item_blocks_with_rows`` folds a ``tool_output`` into a preceding ``tool_call``
|
|
11
|
+
whenever the call identifier is non-empty, owned by exactly one call in the item,
|
|
12
|
+
and already seen — with **no adjacency requirement**. Patch, web-search and MCP
|
|
13
|
+
completion events fold the same way. A boundary drawn between a call and its
|
|
14
|
+
folded output would make the page-local builder emit a different block structure
|
|
15
|
+
than the whole-turn builder does, so a fold group is atomic here. Because folds
|
|
16
|
+
are non-adjacent a group can span intervening blocks, so a group that exceeds the
|
|
17
|
+
budget becomes its own segment: the budget is a target with fold-group atomicity
|
|
18
|
+
as a hard floor, and the ceiling is the budget plus at most one maximal group.
|
|
19
|
+
|
|
20
|
+
This module is deliberately pure — no SQLite, no I/O, and no import from the
|
|
21
|
+
query layer. The caller derives fold groups and the turn-scoped
|
|
22
|
+
``call_owner_count`` and passes them in.
|
|
23
|
+
"""
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import dataclasses
|
|
27
|
+
from typing import Any
|
|
28
|
+
|
|
29
|
+
# Per-segment block budget (spec section 2). At the measured 27.5 DOM nodes per
|
|
30
|
+
# block that is roughly 1,100 nodes, close to the 776 nodes per mounted row the
|
|
31
|
+
# Claude control paints in 449 ms.
|
|
32
|
+
SEGMENT_BLOCK_BUDGET = 40
|
|
33
|
+
|
|
34
|
+
# Per-page block budget, applied alongside ``limit`` (spec section 2). Roughly
|
|
35
|
+
# fifty full segments, in the same range as the Claude control's 2.58 MB page.
|
|
36
|
+
# The per-page bound is not optional: the profiled response was
|
|
37
|
+
# ``total: 78, returned: 78, has_after: false`` — 13.3 MB in one page, because
|
|
38
|
+
# 78 is fewer than the requested 500 — so a change that capped items alone would
|
|
39
|
+
# not bound that conversation at all.
|
|
40
|
+
PAGE_BLOCK_BUDGET = 2000
|
|
41
|
+
|
|
42
|
+
# Per-page SOURCE-byte budget, applied alongside ``limit`` and
|
|
43
|
+
# ``PAGE_BLOCK_BUDGET``; the first bound reached closes the page. It bounds
|
|
44
|
+
# TRANSFER and PARSE cost, which is a byte cost the block budget does not
|
|
45
|
+
# express, because a Codex block is far heavier than a Claude block.
|
|
46
|
+
#
|
|
47
|
+
# It is not redundant with PAGE_BLOCK_BUDGET. After segmentation the profiled
|
|
48
|
+
# conversation is 128 segments carrying 1,906 blocks, so a whole-conversation
|
|
49
|
+
# page holds 1,713 blocks — BELOW the 2,000-block budget. The block bound never
|
|
50
|
+
# fires on it, and without this one the response is still 13.24 MB in one page.
|
|
51
|
+
#
|
|
52
|
+
# CALIBRATED BY MEASUREMENT, not by arithmetic, on 2026-08-02 against a
|
|
53
|
+
# read-only copy of the production store. At 3,000,000 source bytes the
|
|
54
|
+
# profiled conversation serves 16 of its 128 segments, 274 blocks and 2.54 MB
|
|
55
|
+
# on the wire — the 2 to 3 MB target, and effectively the Claude control's
|
|
56
|
+
# 2.58 MB page. Across the six heaviest conversations the served page at this
|
|
57
|
+
# budget ranges from 0.75 MB to 2.54 MB of wire. At 4,000,000 the maximum rises
|
|
58
|
+
# to 3.16 MB; at 2,000,000 the profiled conversation falls to 1.45 MB.
|
|
59
|
+
#
|
|
60
|
+
# Do NOT re-derive this figure by dividing a wire target by the
|
|
61
|
+
# whole-conversation source-to-wire ratio. That ratio is not uniform and not
|
|
62
|
+
# even close: the profiled conversation is 6.91x source-to-wire taken whole
|
|
63
|
+
# (91.52 MB source, 13.24 MB wire), but only 1.11x over the segments a 3 MB page
|
|
64
|
+
# actually serves (2.83 MB source, 2.54 MB wire), because its heaviest rows sit
|
|
65
|
+
# in the tail and are clipped hardest. Re-calibrate by measuring served pages.
|
|
66
|
+
PAGE_SOURCE_BYTE_BUDGET = 3_000_000
|
|
67
|
+
|
|
68
|
+
# Fraction of the budget the boundary snap may give up. Bounding it at a quarter
|
|
69
|
+
# guarantees a segment is never smaller than 75 percent of the budget, so the
|
|
70
|
+
# rule cannot produce a run of very small segments, and it preserves the ceiling
|
|
71
|
+
# because it only ever closes a segment EARLIER than the budget would.
|
|
72
|
+
LOOKBACK_FRACTION = 0.25
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
# Both dataclasses below are ``frozen=True`` with ``eq=True``, which makes
|
|
76
|
+
# Python synthesize a ``__hash__`` — and that synthesized hash raises TypeError
|
|
77
|
+
# here, because every instance carries a ``list`` field. Nothing hashes a
|
|
78
|
+
# FoldGroup or a Segment today and nothing should: they are records passed
|
|
79
|
+
# between two functions in one call, never dict keys or set members, and their
|
|
80
|
+
# identity is positional rather than structural. ``unsafe_hash`` is deliberately
|
|
81
|
+
# NOT set, and ``__hash__`` is set to None so the failure is an explicit
|
|
82
|
+
# "unhashable type" at the call site rather than a TypeError from inside a
|
|
83
|
+
# generated method.
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@dataclasses.dataclass(frozen=True)
|
|
87
|
+
class FoldGroup:
|
|
88
|
+
"""A ``tool_call`` together with every row that folds into it, or a single
|
|
89
|
+
non-folding row. Never divided across segments.
|
|
90
|
+
|
|
91
|
+
``is_title_boundary`` is true when the group's first row is a reasoning row
|
|
92
|
+
whose stored projection produces a ``title``. ``is_tool_transition`` is true
|
|
93
|
+
when it is the first ``tool_call`` following a run of assistant or reasoning
|
|
94
|
+
rows. Those are the two semantic boundaries, in that priority order.
|
|
95
|
+
|
|
96
|
+
``first_pos`` and ``last_pos`` are the group's physical row positions inside
|
|
97
|
+
its item. Because folds are non-adjacent, ``last_pos`` can be far past
|
|
98
|
+
``first_pos`` and can bracket a LATER group entirely, which is what
|
|
99
|
+
``plan_segments`` uses to keep a segment physically contiguous. ``None``
|
|
100
|
+
disables that extension, for callers that have no positional information.
|
|
101
|
+
"""
|
|
102
|
+
|
|
103
|
+
rows: list
|
|
104
|
+
block_count: int
|
|
105
|
+
source_bytes: int
|
|
106
|
+
is_title_boundary: bool = False
|
|
107
|
+
is_tool_transition: bool = False
|
|
108
|
+
first_pos: int | None = None
|
|
109
|
+
last_pos: int | None = None
|
|
110
|
+
|
|
111
|
+
__hash__ = None
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
@dataclasses.dataclass(frozen=True)
|
|
115
|
+
class Segment:
|
|
116
|
+
"""One bounded run of fold groups inside a turn."""
|
|
117
|
+
|
|
118
|
+
ordinal: int
|
|
119
|
+
groups: list
|
|
120
|
+
block_count: int
|
|
121
|
+
source_bytes: int
|
|
122
|
+
anchor_row: Any
|
|
123
|
+
|
|
124
|
+
__hash__ = None
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _boundary_rank(group: FoldGroup) -> int:
|
|
128
|
+
"""Priority of the boundary a cut before this group would land on.
|
|
129
|
+
|
|
130
|
+
Lower is better. 0 = a reasoning title, 1 = a tool transition, 2 = not a
|
|
131
|
+
boundary at all.
|
|
132
|
+
"""
|
|
133
|
+
if group.is_title_boundary:
|
|
134
|
+
return 0
|
|
135
|
+
if group.is_tool_transition:
|
|
136
|
+
return 1
|
|
137
|
+
return 2
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _extend_to_contiguous(groups: list, start: int, end: int, total: int) -> int:
|
|
141
|
+
"""Grow ``end`` until the segment covers a CONTIGUOUS physical row range.
|
|
142
|
+
|
|
143
|
+
Fold-group atomicity alone does not give this (spec section 1). Because
|
|
144
|
+
folds are non-adjacent, a group's rows can bracket a later group's rows —
|
|
145
|
+
a native patch completion event sits between its call and that call's
|
|
146
|
+
output, for instance. Cutting between the two groups would then produce
|
|
147
|
+
segments whose physical ranges overlap: the earlier segment would render
|
|
148
|
+
rows out of physical order, and the later segment's rows would fall inside
|
|
149
|
+
the earlier one's time span.
|
|
150
|
+
|
|
151
|
+
Groups are created in the physical order of their FIRST row, so ``first_pos``
|
|
152
|
+
increases across the list. A cut before group ``end`` is therefore legal
|
|
153
|
+
exactly when every chosen group ends before ``groups[end]`` begins; if it
|
|
154
|
+
does not, that group is absorbed and the test repeats.
|
|
155
|
+
|
|
156
|
+
The extension only ever GROWS a segment, so the 75 percent lookback floor is
|
|
157
|
+
preserved. The ceiling becomes the budget plus the physical span of one
|
|
158
|
+
maximal fold group, which is what spec section 2 states.
|
|
159
|
+
"""
|
|
160
|
+
ends = [groups[i].last_pos for i in range(start, end)
|
|
161
|
+
if groups[i].last_pos is not None]
|
|
162
|
+
if not ends:
|
|
163
|
+
return end
|
|
164
|
+
max_last = max(ends)
|
|
165
|
+
while end < total:
|
|
166
|
+
nxt = groups[end].first_pos
|
|
167
|
+
if nxt is None or nxt > max_last:
|
|
168
|
+
break
|
|
169
|
+
if groups[end].last_pos is not None:
|
|
170
|
+
max_last = max(max_last, groups[end].last_pos)
|
|
171
|
+
end += 1
|
|
172
|
+
return end
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def plan_segments(
|
|
176
|
+
fold_groups: list,
|
|
177
|
+
*,
|
|
178
|
+
block_budget: int | None = None,
|
|
179
|
+
lookback_fraction: float | None = None,
|
|
180
|
+
) -> list[Segment]:
|
|
181
|
+
"""Divide a turn's fold groups into ordered segments.
|
|
182
|
+
|
|
183
|
+
Fills greedily from index 0. A segment closes at the last group that fits
|
|
184
|
+
inside ``block_budget``, or earlier at the highest-priority semantic
|
|
185
|
+
boundary whose cut point falls inside the lookback window — the range from
|
|
186
|
+
``(1 - lookback_fraction) * block_budget`` blocks up to the budget. A group
|
|
187
|
+
that does not fit even into an empty segment becomes its own segment.
|
|
188
|
+
|
|
189
|
+
**Greedy-from-start is the mechanism, not a convenience.** Every segment
|
|
190
|
+
depends only on the groups before it, so appending groups to a growing turn
|
|
191
|
+
leaves earlier segments — and therefore earlier segment keys — untouched.
|
|
192
|
+
Computing boundaries from the end would renumber a conversation's history on
|
|
193
|
+
every append. Segment keys 1..N remain durable only under that tail append;
|
|
194
|
+
inserting or deleting a row before a boundary shifts every later boundary in
|
|
195
|
+
the turn, and a former anchor becomes an interior row.
|
|
196
|
+
|
|
197
|
+
``block_budget`` and ``lookback_fraction`` resolve to the module constants
|
|
198
|
+
at CALL time when omitted. They are deliberately not default ARGUMENT values:
|
|
199
|
+
a default argument binds once at import, so a test that lowers or raises
|
|
200
|
+
``SEGMENT_BLOCK_BUDGET`` would silently keep the imported figure and pass
|
|
201
|
+
vacuously.
|
|
202
|
+
"""
|
|
203
|
+
if block_budget is None:
|
|
204
|
+
block_budget = SEGMENT_BLOCK_BUDGET
|
|
205
|
+
if lookback_fraction is None:
|
|
206
|
+
lookback_fraction = LOOKBACK_FRACTION
|
|
207
|
+
if block_budget <= 0:
|
|
208
|
+
raise ValueError("block_budget must be positive")
|
|
209
|
+
if not 0.0 <= lookback_fraction < 1.0:
|
|
210
|
+
raise ValueError("lookback_fraction must be in [0.0, 1.0)")
|
|
211
|
+
|
|
212
|
+
groups = list(fold_groups)
|
|
213
|
+
total = len(groups)
|
|
214
|
+
floor_blocks = block_budget - int(block_budget * lookback_fraction)
|
|
215
|
+
|
|
216
|
+
segments: list[Segment] = []
|
|
217
|
+
start = 0
|
|
218
|
+
while start < total:
|
|
219
|
+
# How far the budget alone reaches. At least one group always fits, so a
|
|
220
|
+
# single oversized group becomes its own segment rather than stalling.
|
|
221
|
+
end = start
|
|
222
|
+
blocks = 0
|
|
223
|
+
while end < total:
|
|
224
|
+
candidate = blocks + groups[end].block_count
|
|
225
|
+
if end > start and candidate > block_budget:
|
|
226
|
+
break
|
|
227
|
+
blocks = candidate
|
|
228
|
+
end += 1
|
|
229
|
+
|
|
230
|
+
# Look for a boundary inside the window. A cut happens BEFORE group
|
|
231
|
+
# ``cut``, so that group must exist and must itself be a boundary, and
|
|
232
|
+
# the blocks kept must already clear the lookback floor.
|
|
233
|
+
best_cut = None
|
|
234
|
+
best_rank = 2
|
|
235
|
+
kept = 0
|
|
236
|
+
for cut in range(start, end):
|
|
237
|
+
if cut > start and kept >= floor_blocks and cut < total:
|
|
238
|
+
rank = _boundary_rank(groups[cut])
|
|
239
|
+
if rank < best_rank:
|
|
240
|
+
best_rank = rank
|
|
241
|
+
best_cut = cut
|
|
242
|
+
if rank == 0:
|
|
243
|
+
break
|
|
244
|
+
kept += groups[cut].block_count
|
|
245
|
+
if best_cut is not None:
|
|
246
|
+
end = best_cut
|
|
247
|
+
|
|
248
|
+
end = _extend_to_contiguous(groups, start, end, total)
|
|
249
|
+
|
|
250
|
+
chosen = groups[start:end]
|
|
251
|
+
segments.append(Segment(
|
|
252
|
+
ordinal=len(segments),
|
|
253
|
+
groups=chosen,
|
|
254
|
+
block_count=sum(group.block_count for group in chosen),
|
|
255
|
+
source_bytes=sum(group.source_bytes for group in chosen),
|
|
256
|
+
anchor_row=chosen[0].rows[0] if chosen and chosen[0].rows else None,
|
|
257
|
+
))
|
|
258
|
+
start = end
|
|
259
|
+
return segments
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
"""Read-time cleaning of harness markup out of a Codex title (#463 S4 §5).
|
|
2
|
+
|
|
3
|
+
Pure kernel: one entry point, ``clean_codex_title(text) -> str``, and a CLOSED
|
|
4
|
+
allowlist of the grammars a census of the real store actually found.
|
|
5
|
+
|
|
6
|
+
**Read time, not ingest (D5).** The title is stored —
|
|
7
|
+
``codex_conversation_rollups.title`` is written at ingest and ``_rollup_fields``
|
|
8
|
+
returns it on its fast path — so repairing ``derive_title`` would heal nothing
|
|
9
|
+
for the conversations that are already wrong. Cleaning on the read path heals
|
|
10
|
+
all history with no migration and no reingest flag.
|
|
11
|
+
|
|
12
|
+
**The allowlist is a measurement, not a guess (§5.4).** Over the 438 stored
|
|
13
|
+
Codex rollup titles in the production store on 2026-08-04:
|
|
14
|
+
|
|
15
|
+
=========================================== ===== ===========
|
|
16
|
+
grammar count disposition
|
|
17
|
+
=========================================== ===== ===========
|
|
18
|
+
``[$name](<abs path>/SKILL.md) <rest>`` 165 unwrap
|
|
19
|
+
``<command-name>…</command-name> …`` 40 see below
|
|
20
|
+
``<recommended_plugins> …`` 6 strip
|
|
21
|
+
``<command-message>…</command-message> …`` 1 see below
|
|
22
|
+
=========================================== ===== ===========
|
|
23
|
+
|
|
24
|
+
Nothing else occurred. No title carried a tag anywhere but at its head (0 of
|
|
25
|
+
438), and the two ``CODEX_TITLE_SKIP_PREFIXES`` wrappers
|
|
26
|
+
(``<environment_context>``, ``<user_instructions>``) appeared 0 times, because
|
|
27
|
+
they are skipped at ingest by a different mechanism that stays where it is
|
|
28
|
+
(§5.3).
|
|
29
|
+
|
|
30
|
+
Within the command wrapper the three tags are dispositioned separately, from
|
|
31
|
+
what their content actually looks like: ``command-name`` is the slash command
|
|
32
|
+
and is STRIPPED, while ``command-message`` and ``command-args`` carry the human
|
|
33
|
+
text and are UNWRAPPED. On the corpus that turns
|
|
34
|
+
``<command-name>/model</command-name> <command-message>model</command-message>
|
|
35
|
+
<command-args>fable</command-args>`` into ``model fable``.
|
|
36
|
+
|
|
37
|
+
``recommended_plugins`` never closes in the data: titles are capped at 120
|
|
38
|
+
characters, so the stored value is the head of a plugin catalogue. Stripping it
|
|
39
|
+
leaves nothing, and ``_display_chain`` falls through to the project label and
|
|
40
|
+
then to a short native thread id, which the chain already does.
|
|
41
|
+
|
|
42
|
+
**Closed, and deliberately so.** A general tag stripper would eat user-authored
|
|
43
|
+
angle brackets in a title. An unrecognized construct passes through BYTE for
|
|
44
|
+
byte — the function returns its input unchanged when no rule fires, so the 226
|
|
45
|
+
titles the census found clean, and every prose label this is applied to, cannot
|
|
46
|
+
move.
|
|
47
|
+
"""
|
|
48
|
+
from __future__ import annotations
|
|
49
|
+
|
|
50
|
+
import re
|
|
51
|
+
|
|
52
|
+
# `[$skill-name](/abs/path/to/SKILL.md)` — the Codex skill invocation. The
|
|
53
|
+
# prompt text after it is real, and the link target is a private filesystem path
|
|
54
|
+
# that leaks into every title surface. Byte-identical to the client's
|
|
55
|
+
# `cleanQualifiedTitle` regex, so the two agree on the same input and applying
|
|
56
|
+
# both is a no-op.
|
|
57
|
+
#
|
|
58
|
+
# NO trailing lookahead. The first version required whitespace or end of string
|
|
59
|
+
# after the closing paren, so `…/SKILL.md)Task B of issue #450.` — prompt text
|
|
60
|
+
# written straight against the paren — did not match and the whole link,
|
|
61
|
+
# absolute path included, reached the reader header and the outline rail. Two of
|
|
62
|
+
# 300 served titles in the test store carry that form. Nothing is lost by
|
|
63
|
+
# dropping the lookahead: the pattern is head-anchored (`pattern.match`) and its
|
|
64
|
+
# target is the literal `/SKILL.md)`, so it cannot start matching mid-title or
|
|
65
|
+
# consume any other Markdown link.
|
|
66
|
+
_SKILL_LINK_RE = re.compile(
|
|
67
|
+
r"\[((?:\$)[^\]\r\n]+)\]\([^)\r\n]*/SKILL\.md\)")
|
|
68
|
+
|
|
69
|
+
_STRIP, _UNWRAP = "strip", "unwrap"
|
|
70
|
+
|
|
71
|
+
# Head-anchored, in match order. Each entry is (pattern, disposition, group) —
|
|
72
|
+
# `group` names the capture an `unwrap` keeps.
|
|
73
|
+
_GRAMMARS: tuple[tuple[re.Pattern, str, int], ...] = (
|
|
74
|
+
(_SKILL_LINK_RE, _UNWRAP, 1),
|
|
75
|
+
(re.compile(r"<command-name>(.*?)</command-name>", re.S), _STRIP, 1),
|
|
76
|
+
(re.compile(r"<command-message>(.*?)</command-message>", re.S), _UNWRAP, 1),
|
|
77
|
+
(re.compile(r"<command-args>(.*?)</command-args>", re.S), _UNWRAP, 1),
|
|
78
|
+
# The closing form first: alternation is ordered, and the open-ended arm
|
|
79
|
+
# would otherwise swallow a closed construct's tail.
|
|
80
|
+
(re.compile(r"<recommended_plugins>.*?</recommended_plugins>"
|
|
81
|
+
r"|<recommended_plugins>.*", re.S), _STRIP, 0),
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def clean_codex_title(text) -> str:
|
|
86
|
+
"""The title with recognized leading harness markup removed or unwrapped.
|
|
87
|
+
|
|
88
|
+
Returns the input unchanged when no grammar in the allowlist matches its
|
|
89
|
+
head, including for a non-string or empty input, which keeps every
|
|
90
|
+
untouched title and every prose label byte-stable.
|
|
91
|
+
|
|
92
|
+
A construct that strips to nothing yields ``""``, and the caller's fallback
|
|
93
|
+
chain takes over (§5.3).
|
|
94
|
+
"""
|
|
95
|
+
if not isinstance(text, str) or not text:
|
|
96
|
+
return text if isinstance(text, str) else ""
|
|
97
|
+
kept: list[str] = []
|
|
98
|
+
rest = text
|
|
99
|
+
matched = False
|
|
100
|
+
while rest:
|
|
101
|
+
head = rest.lstrip()
|
|
102
|
+
for pattern, disposition, group in _GRAMMARS:
|
|
103
|
+
found = pattern.match(head)
|
|
104
|
+
if found is None:
|
|
105
|
+
continue
|
|
106
|
+
matched = True
|
|
107
|
+
if disposition == _UNWRAP:
|
|
108
|
+
kept.append(found.group(group))
|
|
109
|
+
rest = head[found.end():]
|
|
110
|
+
break
|
|
111
|
+
else:
|
|
112
|
+
kept.append(head)
|
|
113
|
+
break
|
|
114
|
+
if not matched:
|
|
115
|
+
return text
|
|
116
|
+
return " ".join(" ".join(part.split()) for part in kept if part.strip()).strip()
|
|
@@ -306,14 +306,43 @@ def _map_claude_item(session_id: str, it: dict) -> dict:
|
|
|
306
306
|
"""One Claude assembled item → the neutral detail item shape (§5.6). Claude's
|
|
307
307
|
kinds/blocks pass through untranslated (both vocabularies are provider-truthful
|
|
308
308
|
values of the same required field)."""
|
|
309
|
+
own_uuid = it["anchor"]["uuid"]
|
|
309
310
|
return {
|
|
310
|
-
"item_key": _claude_item_key(session_id,
|
|
311
|
+
"item_key": _claude_item_key(session_id, own_uuid),
|
|
311
312
|
"kind": it["kind"],
|
|
312
313
|
"timestamp_utc": it.get("ts"),
|
|
313
314
|
"model": it.get("model"),
|
|
314
315
|
"blocks": it.get("blocks", []),
|
|
315
316
|
"cost_usd": it.get("cost_usd"),
|
|
316
317
|
"tokens": _claude_tokens_union(it.get("tokens")),
|
|
318
|
+
"member_item_keys": [
|
|
319
|
+
_claude_item_key(session_id, uuid)
|
|
320
|
+
for uuid in it.get("member_uuids", []) if uuid != own_uuid
|
|
321
|
+
],
|
|
322
|
+
"subagent_key": it.get("subagent_key"),
|
|
323
|
+
"parent_item_key": (
|
|
324
|
+
_claude_item_key(session_id, it["parent_uuid"])
|
|
325
|
+
if it.get("parent_uuid") is not None else None
|
|
326
|
+
),
|
|
327
|
+
"is_sidechain": bool(it.get("is_sidechain")),
|
|
328
|
+
"meta_kind": it.get("meta_kind"),
|
|
329
|
+
"skill_name": it.get("skill_name"),
|
|
330
|
+
"command_name": it.get("command_name"),
|
|
331
|
+
"cache_failure": it.get("cache_failure"),
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
def _map_claude_subagent_meta(session_id: str, values: dict | None) -> dict:
|
|
336
|
+
"""Translate the one identity-bearing field in legacy subagent metadata."""
|
|
337
|
+
return {
|
|
338
|
+
key: ({
|
|
339
|
+
**meta,
|
|
340
|
+
"spawn_uuid": (
|
|
341
|
+
_claude_item_key(session_id, meta["spawn_uuid"])
|
|
342
|
+
if meta.get("spawn_uuid") is not None else None
|
|
343
|
+
),
|
|
344
|
+
} if isinstance(meta, dict) else meta)
|
|
345
|
+
for key, meta in (values or {}).items()
|
|
317
346
|
}
|
|
318
347
|
|
|
319
348
|
|
|
@@ -386,6 +415,8 @@ def _claude_detail(
|
|
|
386
415
|
"title": res["title"],
|
|
387
416
|
"items": neutral_items,
|
|
388
417
|
"page": page,
|
|
418
|
+
"subagent_meta": _map_claude_subagent_meta(
|
|
419
|
+
session_id, res.get("subagent_meta")),
|
|
389
420
|
"children": [],
|
|
390
421
|
"parent": None,
|
|
391
422
|
"total_cost_usd": res["cost_usd"],
|
|
@@ -399,33 +430,98 @@ def _claude_detail(
|
|
|
399
430
|
def _claude_outline(
|
|
400
431
|
conn: sqlite3.Connection, session_id: str, conversation_key: str,
|
|
401
432
|
) -> dict:
|
|
402
|
-
"""Claude outline envelope (§5.6)
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
433
|
+
"""Claude outline envelope (§5.6), without weakening the native outline.
|
|
434
|
+
|
|
435
|
+
The legacy kernel is the authority for navigation: it already derives tool
|
|
436
|
+
failures, subagent topology, cache rebuilds, files, task completion, and the
|
|
437
|
+
complete turn skeleton from the same assembled items as detail. This
|
|
438
|
+
adapter changes only identity-bearing fields from native UUIDs to neutral
|
|
439
|
+
item keys and names the provider-neutral file fields. Re-deriving a smaller
|
|
440
|
+
outline from assembled blocks here previously stripped precisely the facts
|
|
441
|
+
the qualified reader needs (#491).
|
|
442
|
+
"""
|
|
406
443
|
o = lcq.get_conversation_outline(conn, session_id)
|
|
407
444
|
if o is None:
|
|
408
445
|
return {"status": "not_found", "conversation_key": conversation_key}
|
|
409
|
-
|
|
446
|
+
|
|
447
|
+
def item_key(uuid):
|
|
448
|
+
return _claude_item_key(session_id, uuid) if uuid is not None else None
|
|
449
|
+
|
|
410
450
|
turns = []
|
|
411
|
-
for
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
"
|
|
419
|
-
"
|
|
420
|
-
|
|
421
|
-
|
|
451
|
+
for turn in o["turns"]:
|
|
452
|
+
own_key = item_key(turn["uuid"])
|
|
453
|
+
neutral = {
|
|
454
|
+
"item_key": own_key,
|
|
455
|
+
"kind": turn["kind"],
|
|
456
|
+
"label": turn["label"],
|
|
457
|
+
"timestamp_utc": turn.get("ts"),
|
|
458
|
+
"kinds": {turn["kind"]: 1},
|
|
459
|
+
"member_item_keys": [
|
|
460
|
+
item_key(uuid) for uuid in turn.get("member_uuids", [])
|
|
461
|
+
if uuid != turn["uuid"]
|
|
462
|
+
],
|
|
463
|
+
"subagent_key": turn.get("subagent_key"),
|
|
464
|
+
"parent_item_key": item_key(turn.get("parent_uuid")),
|
|
465
|
+
"is_sidechain": bool(turn.get("is_sidechain")),
|
|
466
|
+
}
|
|
467
|
+
for field in (
|
|
468
|
+
"tools", "thinking", "model", "tokens", "meta_kind",
|
|
469
|
+
"skill_name", "cache_failure"):
|
|
470
|
+
if field in turn:
|
|
471
|
+
neutral[field] = turn[field]
|
|
472
|
+
turns.append(neutral)
|
|
473
|
+
|
|
474
|
+
stats = dict(o["stats"])
|
|
475
|
+
cache_failures = stats.get("cache_failures")
|
|
476
|
+
if isinstance(cache_failures, dict):
|
|
477
|
+
cache_failures = dict(cache_failures)
|
|
478
|
+
cache_failures["rebuilds"] = [
|
|
479
|
+
{**row, "uuid": item_key(row.get("uuid"))}
|
|
480
|
+
for row in cache_failures.get("rebuilds", [])
|
|
481
|
+
]
|
|
482
|
+
stats["cache_failures"] = cache_failures
|
|
483
|
+
|
|
484
|
+
files = []
|
|
485
|
+
for file in o.get("files", []):
|
|
486
|
+
touches = [{
|
|
487
|
+
"item_key": item_key(touch.get("uuid")),
|
|
488
|
+
"timestamp_utc": None,
|
|
489
|
+
"tool_use_id": touch.get("tool_use_id"),
|
|
490
|
+
"op": touch.get("op"),
|
|
491
|
+
"added": touch.get("add"),
|
|
492
|
+
"removed": touch.get("del"),
|
|
493
|
+
} for touch in file.get("touches", [])]
|
|
494
|
+
tools = list(dict.fromkeys(
|
|
495
|
+
touch["op"] for touch in touches if touch.get("op")))
|
|
496
|
+
files.append({
|
|
497
|
+
"file_path": file.get("path"),
|
|
498
|
+
# The count-only neutral file shape requires a tool label. Rich
|
|
499
|
+
# Claude entries retain every native operation instead of claiming
|
|
500
|
+
# that Write/MultiEdit touches were Edit calls.
|
|
501
|
+
"tool": ",".join(tools),
|
|
502
|
+
"count": len(touches),
|
|
503
|
+
"added": file.get("add"),
|
|
504
|
+
"removed": file.get("del"),
|
|
505
|
+
"touches": touches,
|
|
422
506
|
})
|
|
507
|
+
|
|
508
|
+
subagent_meta = _map_claude_subagent_meta(
|
|
509
|
+
session_id, o.get("subagent_meta"))
|
|
510
|
+
task_completion = o.get("task_completion")
|
|
511
|
+
if isinstance(task_completion, dict):
|
|
512
|
+
task_completion = {
|
|
513
|
+
**task_completion,
|
|
514
|
+
"anchor_uuid": item_key(task_completion.get("anchor_uuid")),
|
|
515
|
+
}
|
|
423
516
|
return {
|
|
424
517
|
"status": "ok",
|
|
425
518
|
"conversation_key": conversation_key,
|
|
426
519
|
"turns": turns,
|
|
427
|
-
"
|
|
428
|
-
"
|
|
520
|
+
"subagent_meta": subagent_meta,
|
|
521
|
+
"subagent_costs": o.get("subagent_costs", {}),
|
|
522
|
+
"stats": stats,
|
|
523
|
+
"files": files,
|
|
524
|
+
"task_completion": task_completion,
|
|
429
525
|
"children": [],
|
|
430
526
|
}
|
|
431
527
|
|
|
@@ -531,6 +627,25 @@ def neutral_browse(
|
|
|
531
627
|
raise ValueError(f"unknown source: {source!r}")
|
|
532
628
|
|
|
533
629
|
|
|
630
|
+
def neutral_facets(
|
|
631
|
+
conn: sqlite3.Connection, *, source: str, effective_speed: str | None = None,
|
|
632
|
+
) -> dict:
|
|
633
|
+
"""Facet-only collection envelope for one source.
|
|
634
|
+
|
|
635
|
+
Codex has a dedicated rollup projection so the facets request never builds
|
|
636
|
+
or prices a browse page that its transport discards. Claude retains its
|
|
637
|
+
established browse-derived facet contract.
|
|
638
|
+
"""
|
|
639
|
+
speed = effective_speed or _DEFAULT_SPEED
|
|
640
|
+
if source == "codex":
|
|
641
|
+
return q.list_codex_conversation_facets(conn)
|
|
642
|
+
if source == "claude":
|
|
643
|
+
env = _claude_browse(conn, effective_speed=speed)
|
|
644
|
+
return {"status": env.get("status"), "facets": env.get("facets") or {
|
|
645
|
+
"projects": [], "models": []}}
|
|
646
|
+
raise ValueError(f"unknown source: {source!r}")
|
|
647
|
+
|
|
648
|
+
|
|
534
649
|
def neutral_detail(
|
|
535
650
|
conn: sqlite3.Connection, ref: str, *, effective_speed: str | None = None,
|
|
536
651
|
after: str | None = None, before: str | None = None,
|
|
@@ -725,6 +840,8 @@ def _claude_export(
|
|
|
725
840
|
def neutral_find(
|
|
726
841
|
conn: sqlite3.Connection, ref: str, query: str, *, kind: str = "all",
|
|
727
842
|
regex: bool = False, case: bool = False, effective_speed: str | None = None,
|
|
843
|
+
limit: int = 100, cursor: str | None = None, direction: str = "next",
|
|
844
|
+
around: str | None = None,
|
|
728
845
|
) -> dict:
|
|
729
846
|
"""In-conversation find for a neutral reference (§3.1). An unknown ``kind``
|
|
730
847
|
raises ``ValueError`` (route → 400); an unknown/garbage ref → ``not_found``.
|
|
@@ -736,8 +853,23 @@ def neutral_find(
|
|
|
736
853
|
if cref is None:
|
|
737
854
|
return {"status": "not_found", "conversation_key": ref}
|
|
738
855
|
if cref.source == "codex":
|
|
739
|
-
|
|
740
|
-
|
|
856
|
+
try:
|
|
857
|
+
return q.find_occurrences_in_codex_conversation(
|
|
858
|
+
conn,
|
|
859
|
+
cref.conversation_key,
|
|
860
|
+
query,
|
|
861
|
+
kind=kind,
|
|
862
|
+
regex=regex,
|
|
863
|
+
case_sensitive=case,
|
|
864
|
+
limit=limit,
|
|
865
|
+
cursor=cursor,
|
|
866
|
+
direction=direction,
|
|
867
|
+
around=around,
|
|
868
|
+
)
|
|
869
|
+
except q.InvalidFindCursor:
|
|
870
|
+
return {"status": "invalid_find_cursor"}
|
|
871
|
+
except q.StaleFindCursor:
|
|
872
|
+
return {"status": "stale_find_cursor"}
|
|
741
873
|
return _claude_find(
|
|
742
874
|
conn, cref.native_key, cref.conversation_key, query,
|
|
743
875
|
kind=kind, regex=regex, case=case)
|
|
@@ -51,7 +51,9 @@ def watch_step(files, seen, *, stat_fn=file_sig, ingest_fn, committed_sig_fn=Non
|
|
|
51
51
|
cursor, in session_files, lags the new disk size). committed_sig_fn defaults
|
|
52
52
|
to stat_fn for pure unit tests with no cache. A contended/declined/failed
|
|
53
53
|
ingest leaves `seen` untouched so the next cycle retries (the 5s backstop is
|
|
54
|
-
the floor).
|
|
54
|
+
the floor). Account-scoped callers may additionally expose
|
|
55
|
+
``stats.targeted_visible``: the clean ingest still advances ``seen``, but a
|
|
56
|
+
false value suppresses the tail frame when only another account changed."""
|
|
55
57
|
committed_sig_fn = committed_sig_fn or stat_fn
|
|
56
58
|
changed = changed_paths(files, seen, stat_fn)
|
|
57
59
|
if not changed:
|
|
@@ -64,4 +66,4 @@ def watch_step(files, seen, *, stat_fn=file_sig, ingest_fn, committed_sig_fn=Non
|
|
|
64
66
|
sig = committed_sig_fn(p)
|
|
65
67
|
if sig is not None:
|
|
66
68
|
new_seen[p] = sig
|
|
67
|
-
return new_seen, True
|
|
69
|
+
return new_seen, bool(getattr(stats, "targeted_visible", True))
|