cctally 1.90.1 → 1.92.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/CHANGELOG.md +74 -0
  2. package/README.md +2 -2
  3. package/bin/_cctally_cache.py +863 -74
  4. package/bin/_cctally_config.py +57 -0
  5. package/bin/_cctally_core.py +53 -8
  6. package/bin/_cctally_dashboard.py +146 -5
  7. package/bin/_cctally_dashboard_conversation.py +164 -18
  8. package/bin/_cctally_dashboard_envelope.py +69 -12
  9. package/bin/_cctally_dashboard_sources.py +27 -1
  10. package/bin/_cctally_db.py +372 -10
  11. package/bin/_cctally_doctor.py +18 -1
  12. package/bin/_cctally_journal.py +535 -13
  13. package/bin/_cctally_journal_repair.py +6 -0
  14. package/bin/_cctally_parser.py +6 -0
  15. package/bin/_cctally_quota.py +171 -55
  16. package/bin/_cctally_record.py +13 -1
  17. package/bin/_cctally_rederive.py +4 -0
  18. package/bin/_cctally_store.py +311 -6
  19. package/bin/_cctally_transcript.py +32 -2
  20. package/bin/_lib_cache_report.py +8 -3
  21. package/bin/_lib_cache_report_wire.py +8 -20
  22. package/bin/_lib_codex_conversation.py +959 -81
  23. package/bin/_lib_codex_conversation_query.py +2792 -167
  24. package/bin/_lib_codex_find_projection.py +370 -0
  25. package/bin/_lib_codex_harness_preamble.py +176 -0
  26. package/bin/_lib_codex_hooks.py +5 -3
  27. package/bin/_lib_codex_js_scan.py +254 -0
  28. package/bin/_lib_codex_landmarks.py +309 -0
  29. package/bin/_lib_codex_reasoning_headings.py +73 -0
  30. package/bin/_lib_codex_segments.py +259 -0
  31. package/bin/_lib_codex_title_clean.py +116 -0
  32. package/bin/_lib_conversation_dispatch.py +153 -21
  33. package/bin/_lib_conversation_watch.py +4 -2
  34. package/bin/_lib_dashboard_sources.py +33 -32
  35. package/bin/_lib_doctor.py +64 -0
  36. package/bin/_lib_quota_alert_axes.py +31 -34
  37. package/bin/_lib_stats_damage.py +523 -0
  38. package/bin/cctally +5 -0
  39. package/dashboard/static/assets/index-BEzzJtUd.js +97 -0
  40. package/dashboard/static/assets/index-DnWdv8um.css +1 -0
  41. package/dashboard/static/dashboard.html +2 -2
  42. package/package.json +9 -1
  43. package/dashboard/static/assets/index-Bar8-S1i.css +0 -1
  44. package/dashboard/static/assets/index-CRogVlEC.js +0 -92
@@ -0,0 +1,259 @@
1
+ """#463 S1 — the pure segmentation kernel for Codex conversation turns.
2
+
3
+ A **segment** is a run of consecutive fold groups inside one ``klass ==
4
+ "response"`` canonical item, taken greedily from the start of the turn, closed
5
+ when the block budget is reached, and closed earlier when a semantic boundary
6
+ falls inside the budget window. Items whose class is not ``response`` already
7
+ contain a single row; each is exactly one segment and its key does not change.
8
+
9
+ **The unit is a fold group, not a row, and that is not a detail.**
10
+ ``_item_blocks_with_rows`` folds a ``tool_output`` into a preceding ``tool_call``
11
+ whenever the call identifier is non-empty, owned by exactly one call in the item,
12
+ and already seen — with **no adjacency requirement**. Patch, web-search and MCP
13
+ completion events fold the same way. A boundary drawn between a call and its
14
+ folded output would make the page-local builder emit a different block structure
15
+ than the whole-turn builder does, so a fold group is atomic here. Because folds
16
+ are non-adjacent a group can span intervening blocks, so a group that exceeds the
17
+ budget becomes its own segment: the budget is a target with fold-group atomicity
18
+ as a hard floor, and the ceiling is the budget plus at most one maximal group.
19
+
20
+ This module is deliberately pure — no SQLite, no I/O, and no import from the
21
+ query layer. The caller derives fold groups and the turn-scoped
22
+ ``call_owner_count`` and passes them in.
23
+ """
24
+ from __future__ import annotations
25
+
26
+ import dataclasses
27
+ from typing import Any
28
+
29
+ # Per-segment block budget (spec section 2). At the measured 27.5 DOM nodes per
30
+ # block that is roughly 1,100 nodes, close to the 776 nodes per mounted row the
31
+ # Claude control paints in 449 ms.
32
+ SEGMENT_BLOCK_BUDGET = 40
33
+
34
+ # Per-page block budget, applied alongside ``limit`` (spec section 2). Roughly
35
+ # fifty full segments, in the same range as the Claude control's 2.58 MB page.
36
+ # The per-page bound is not optional: the profiled response was
37
+ # ``total: 78, returned: 78, has_after: false`` — 13.3 MB in one page, because
38
+ # 78 is fewer than the requested 500 — so a change that capped items alone would
39
+ # not bound that conversation at all.
40
+ PAGE_BLOCK_BUDGET = 2000
41
+
42
+ # Per-page SOURCE-byte budget, applied alongside ``limit`` and
43
+ # ``PAGE_BLOCK_BUDGET``; the first bound reached closes the page. It bounds
44
+ # TRANSFER and PARSE cost, which is a byte cost the block budget does not
45
+ # express, because a Codex block is far heavier than a Claude block.
46
+ #
47
+ # It is not redundant with PAGE_BLOCK_BUDGET. After segmentation the profiled
48
+ # conversation is 128 segments carrying 1,906 blocks, so a whole-conversation
49
+ # page holds 1,713 blocks — BELOW the 2,000-block budget. The block bound never
50
+ # fires on it, and without this one the response is still 13.24 MB in one page.
51
+ #
52
+ # CALIBRATED BY MEASUREMENT, not by arithmetic, on 2026-08-02 against a
53
+ # read-only copy of the production store. At 3,000,000 source bytes the
54
+ # profiled conversation serves 16 of its 128 segments, 274 blocks and 2.54 MB
55
+ # on the wire — the 2 to 3 MB target, and effectively the Claude control's
56
+ # 2.58 MB page. Across the six heaviest conversations the served page at this
57
+ # budget ranges from 0.75 MB to 2.54 MB of wire. At 4,000,000 the maximum rises
58
+ # to 3.16 MB; at 2,000,000 the profiled conversation falls to 1.45 MB.
59
+ #
60
+ # Do NOT re-derive this figure by dividing a wire target by the
61
+ # whole-conversation source-to-wire ratio. That ratio is not uniform and not
62
+ # even close: the profiled conversation is 6.91x source-to-wire taken whole
63
+ # (91.52 MB source, 13.24 MB wire), but only 1.11x over the segments a 3 MB page
64
+ # actually serves (2.83 MB source, 2.54 MB wire), because its heaviest rows sit
65
+ # in the tail and are clipped hardest. Re-calibrate by measuring served pages.
66
+ PAGE_SOURCE_BYTE_BUDGET = 3_000_000
67
+
68
+ # Fraction of the budget the boundary snap may give up. Bounding it at a quarter
69
+ # guarantees a segment is never smaller than 75 percent of the budget, so the
70
+ # rule cannot produce a run of very small segments, and it preserves the ceiling
71
+ # because it only ever closes a segment EARLIER than the budget would.
72
+ LOOKBACK_FRACTION = 0.25
73
+
74
+
75
+ # Both dataclasses below are ``frozen=True`` with ``eq=True``, which makes
76
+ # Python synthesize a ``__hash__`` — and that synthesized hash raises TypeError
77
+ # here, because every instance carries a ``list`` field. Nothing hashes a
78
+ # FoldGroup or a Segment today and nothing should: they are records passed
79
+ # between two functions in one call, never dict keys or set members, and their
80
+ # identity is positional rather than structural. ``unsafe_hash`` is deliberately
81
+ # NOT set, and ``__hash__`` is set to None so the failure is an explicit
82
+ # "unhashable type" at the call site rather than a TypeError from inside a
83
+ # generated method.
84
+
85
+
86
+ @dataclasses.dataclass(frozen=True)
87
+ class FoldGroup:
88
+ """A ``tool_call`` together with every row that folds into it, or a single
89
+ non-folding row. Never divided across segments.
90
+
91
+ ``is_title_boundary`` is true when the group's first row is a reasoning row
92
+ whose stored projection produces a ``title``. ``is_tool_transition`` is true
93
+ when it is the first ``tool_call`` following a run of assistant or reasoning
94
+ rows. Those are the two semantic boundaries, in that priority order.
95
+
96
+ ``first_pos`` and ``last_pos`` are the group's physical row positions inside
97
+ its item. Because folds are non-adjacent, ``last_pos`` can be far past
98
+ ``first_pos`` and can bracket a LATER group entirely, which is what
99
+ ``plan_segments`` uses to keep a segment physically contiguous. ``None``
100
+ disables that extension, for callers that have no positional information.
101
+ """
102
+
103
+ rows: list
104
+ block_count: int
105
+ source_bytes: int
106
+ is_title_boundary: bool = False
107
+ is_tool_transition: bool = False
108
+ first_pos: int | None = None
109
+ last_pos: int | None = None
110
+
111
+ __hash__ = None
112
+
113
+
114
+ @dataclasses.dataclass(frozen=True)
115
+ class Segment:
116
+ """One bounded run of fold groups inside a turn."""
117
+
118
+ ordinal: int
119
+ groups: list
120
+ block_count: int
121
+ source_bytes: int
122
+ anchor_row: Any
123
+
124
+ __hash__ = None
125
+
126
+
127
+ def _boundary_rank(group: FoldGroup) -> int:
128
+ """Priority of the boundary a cut before this group would land on.
129
+
130
+ Lower is better. 0 = a reasoning title, 1 = a tool transition, 2 = not a
131
+ boundary at all.
132
+ """
133
+ if group.is_title_boundary:
134
+ return 0
135
+ if group.is_tool_transition:
136
+ return 1
137
+ return 2
138
+
139
+
140
+ def _extend_to_contiguous(groups: list, start: int, end: int, total: int) -> int:
141
+ """Grow ``end`` until the segment covers a CONTIGUOUS physical row range.
142
+
143
+ Fold-group atomicity alone does not give this (spec section 1). Because
144
+ folds are non-adjacent, a group's rows can bracket a later group's rows —
145
+ a native patch completion event sits between its call and that call's
146
+ output, for instance. Cutting between the two groups would then produce
147
+ segments whose physical ranges overlap: the earlier segment would render
148
+ rows out of physical order, and the later segment's rows would fall inside
149
+ the earlier one's time span.
150
+
151
+ Groups are created in the physical order of their FIRST row, so ``first_pos``
152
+ increases across the list. A cut before group ``end`` is therefore legal
153
+ exactly when every chosen group ends before ``groups[end]`` begins; if it
154
+ does not, that group is absorbed and the test repeats.
155
+
156
+ The extension only ever GROWS a segment, so the 75 percent lookback floor is
157
+ preserved. The ceiling becomes the budget plus the physical span of one
158
+ maximal fold group, which is what spec section 2 states.
159
+ """
160
+ ends = [groups[i].last_pos for i in range(start, end)
161
+ if groups[i].last_pos is not None]
162
+ if not ends:
163
+ return end
164
+ max_last = max(ends)
165
+ while end < total:
166
+ nxt = groups[end].first_pos
167
+ if nxt is None or nxt > max_last:
168
+ break
169
+ if groups[end].last_pos is not None:
170
+ max_last = max(max_last, groups[end].last_pos)
171
+ end += 1
172
+ return end
173
+
174
+
175
+ def plan_segments(
176
+ fold_groups: list,
177
+ *,
178
+ block_budget: int | None = None,
179
+ lookback_fraction: float | None = None,
180
+ ) -> list[Segment]:
181
+ """Divide a turn's fold groups into ordered segments.
182
+
183
+ Fills greedily from index 0. A segment closes at the last group that fits
184
+ inside ``block_budget``, or earlier at the highest-priority semantic
185
+ boundary whose cut point falls inside the lookback window — the range from
186
+ ``(1 - lookback_fraction) * block_budget`` blocks up to the budget. A group
187
+ that does not fit even into an empty segment becomes its own segment.
188
+
189
+ **Greedy-from-start is the mechanism, not a convenience.** Every segment
190
+ depends only on the groups before it, so appending groups to a growing turn
191
+ leaves earlier segments — and therefore earlier segment keys — untouched.
192
+ Computing boundaries from the end would renumber a conversation's history on
193
+ every append. Segment keys 1..N remain durable only under that tail append;
194
+ inserting or deleting a row before a boundary shifts every later boundary in
195
+ the turn, and a former anchor becomes an interior row.
196
+
197
+ ``block_budget`` and ``lookback_fraction`` resolve to the module constants
198
+ at CALL time when omitted. They are deliberately not default ARGUMENT values:
199
+ a default argument binds once at import, so a test that lowers or raises
200
+ ``SEGMENT_BLOCK_BUDGET`` would silently keep the imported figure and pass
201
+ vacuously.
202
+ """
203
+ if block_budget is None:
204
+ block_budget = SEGMENT_BLOCK_BUDGET
205
+ if lookback_fraction is None:
206
+ lookback_fraction = LOOKBACK_FRACTION
207
+ if block_budget <= 0:
208
+ raise ValueError("block_budget must be positive")
209
+ if not 0.0 <= lookback_fraction < 1.0:
210
+ raise ValueError("lookback_fraction must be in [0.0, 1.0)")
211
+
212
+ groups = list(fold_groups)
213
+ total = len(groups)
214
+ floor_blocks = block_budget - int(block_budget * lookback_fraction)
215
+
216
+ segments: list[Segment] = []
217
+ start = 0
218
+ while start < total:
219
+ # How far the budget alone reaches. At least one group always fits, so a
220
+ # single oversized group becomes its own segment rather than stalling.
221
+ end = start
222
+ blocks = 0
223
+ while end < total:
224
+ candidate = blocks + groups[end].block_count
225
+ if end > start and candidate > block_budget:
226
+ break
227
+ blocks = candidate
228
+ end += 1
229
+
230
+ # Look for a boundary inside the window. A cut happens BEFORE group
231
+ # ``cut``, so that group must exist and must itself be a boundary, and
232
+ # the blocks kept must already clear the lookback floor.
233
+ best_cut = None
234
+ best_rank = 2
235
+ kept = 0
236
+ for cut in range(start, end):
237
+ if cut > start and kept >= floor_blocks and cut < total:
238
+ rank = _boundary_rank(groups[cut])
239
+ if rank < best_rank:
240
+ best_rank = rank
241
+ best_cut = cut
242
+ if rank == 0:
243
+ break
244
+ kept += groups[cut].block_count
245
+ if best_cut is not None:
246
+ end = best_cut
247
+
248
+ end = _extend_to_contiguous(groups, start, end, total)
249
+
250
+ chosen = groups[start:end]
251
+ segments.append(Segment(
252
+ ordinal=len(segments),
253
+ groups=chosen,
254
+ block_count=sum(group.block_count for group in chosen),
255
+ source_bytes=sum(group.source_bytes for group in chosen),
256
+ anchor_row=chosen[0].rows[0] if chosen and chosen[0].rows else None,
257
+ ))
258
+ start = end
259
+ return segments
@@ -0,0 +1,116 @@
1
+ """Read-time cleaning of harness markup out of a Codex title (#463 S4 §5).
2
+
3
+ Pure kernel: one entry point, ``clean_codex_title(text) -> str``, and a CLOSED
4
+ allowlist of the grammars a census of the real store actually found.
5
+
6
+ **Read time, not ingest (D5).** The title is stored —
7
+ ``codex_conversation_rollups.title`` is written at ingest and ``_rollup_fields``
8
+ returns it on its fast path — so repairing ``derive_title`` would heal nothing
9
+ for the conversations that are already wrong. Cleaning on the read path heals
10
+ all history with no migration and no reingest flag.
11
+
12
+ **The allowlist is a measurement, not a guess (§5.4).** Over the 438 stored
13
+ Codex rollup titles in the production store on 2026-08-04:
14
+
15
+ =========================================== ===== ===========
16
+ grammar count disposition
17
+ =========================================== ===== ===========
18
+ ``[$name](<abs path>/SKILL.md) <rest>`` 165 unwrap
19
+ ``<command-name>…</command-name> …`` 40 see below
20
+ ``<recommended_plugins> …`` 6 strip
21
+ ``<command-message>…</command-message> …`` 1 see below
22
+ =========================================== ===== ===========
23
+
24
+ Nothing else occurred. No title carried a tag anywhere but at its head (0 of
25
+ 438), and the two ``CODEX_TITLE_SKIP_PREFIXES`` wrappers
26
+ (``<environment_context>``, ``<user_instructions>``) appeared 0 times, because
27
+ they are skipped at ingest by a different mechanism that stays where it is
28
+ (§5.3).
29
+
30
+ Within the command wrapper the three tags are dispositioned separately, from
31
+ what their content actually looks like: ``command-name`` is the slash command
32
+ and is STRIPPED, while ``command-message`` and ``command-args`` carry the human
33
+ text and are UNWRAPPED. On the corpus that turns
34
+ ``<command-name>/model</command-name> <command-message>model</command-message>
35
+ <command-args>fable</command-args>`` into ``model fable``.
36
+
37
+ ``recommended_plugins`` never closes in the data: titles are capped at 120
38
+ characters, so the stored value is the head of a plugin catalogue. Stripping it
39
+ leaves nothing, and ``_display_chain`` falls through to the project label and
40
+ then to a short native thread id, which the chain already does.
41
+
42
+ **Closed, and deliberately so.** A general tag stripper would eat user-authored
43
+ angle brackets in a title. An unrecognized construct passes through BYTE for
44
+ byte — the function returns its input unchanged when no rule fires, so the 226
45
+ titles the census found clean, and every prose label this is applied to, cannot
46
+ move.
47
+ """
48
+ from __future__ import annotations
49
+
50
+ import re
51
+
52
+ # `[$skill-name](/abs/path/to/SKILL.md)` — the Codex skill invocation. The
53
+ # prompt text after it is real, and the link target is a private filesystem path
54
+ # that leaks into every title surface. Byte-identical to the client's
55
+ # `cleanQualifiedTitle` regex, so the two agree on the same input and applying
56
+ # both is a no-op.
57
+ #
58
+ # NO trailing lookahead. The first version required whitespace or end of string
59
+ # after the closing paren, so `…/SKILL.md)Task B of issue #450.` — prompt text
60
+ # written straight against the paren — did not match and the whole link,
61
+ # absolute path included, reached the reader header and the outline rail. Two of
62
+ # 300 served titles in the test store carry that form. Nothing is lost by
63
+ # dropping the lookahead: the pattern is head-anchored (`pattern.match`) and its
64
+ # target is the literal `/SKILL.md)`, so it cannot start matching mid-title or
65
+ # consume any other Markdown link.
66
+ _SKILL_LINK_RE = re.compile(
67
+ r"\[((?:\$)[^\]\r\n]+)\]\([^)\r\n]*/SKILL\.md\)")
68
+
69
+ _STRIP, _UNWRAP = "strip", "unwrap"
70
+
71
+ # Head-anchored, in match order. Each entry is (pattern, disposition, group) —
72
+ # `group` names the capture an `unwrap` keeps.
73
+ _GRAMMARS: tuple[tuple[re.Pattern, str, int], ...] = (
74
+ (_SKILL_LINK_RE, _UNWRAP, 1),
75
+ (re.compile(r"<command-name>(.*?)</command-name>", re.S), _STRIP, 1),
76
+ (re.compile(r"<command-message>(.*?)</command-message>", re.S), _UNWRAP, 1),
77
+ (re.compile(r"<command-args>(.*?)</command-args>", re.S), _UNWRAP, 1),
78
+ # The closing form first: alternation is ordered, and the open-ended arm
79
+ # would otherwise swallow a closed construct's tail.
80
+ (re.compile(r"<recommended_plugins>.*?</recommended_plugins>"
81
+ r"|<recommended_plugins>.*", re.S), _STRIP, 0),
82
+ )
83
+
84
+
85
+ def clean_codex_title(text) -> str:
86
+ """The title with recognized leading harness markup removed or unwrapped.
87
+
88
+ Returns the input unchanged when no grammar in the allowlist matches its
89
+ head, including for a non-string or empty input, which keeps every
90
+ untouched title and every prose label byte-stable.
91
+
92
+ A construct that strips to nothing yields ``""``, and the caller's fallback
93
+ chain takes over (§5.3).
94
+ """
95
+ if not isinstance(text, str) or not text:
96
+ return text if isinstance(text, str) else ""
97
+ kept: list[str] = []
98
+ rest = text
99
+ matched = False
100
+ while rest:
101
+ head = rest.lstrip()
102
+ for pattern, disposition, group in _GRAMMARS:
103
+ found = pattern.match(head)
104
+ if found is None:
105
+ continue
106
+ matched = True
107
+ if disposition == _UNWRAP:
108
+ kept.append(found.group(group))
109
+ rest = head[found.end():]
110
+ break
111
+ else:
112
+ kept.append(head)
113
+ break
114
+ if not matched:
115
+ return text
116
+ return " ".join(" ".join(part.split()) for part in kept if part.strip()).strip()
@@ -306,14 +306,43 @@ def _map_claude_item(session_id: str, it: dict) -> dict:
306
306
  """One Claude assembled item → the neutral detail item shape (§5.6). Claude's
307
307
  kinds/blocks pass through untranslated (both vocabularies are provider-truthful
308
308
  values of the same required field)."""
309
+ own_uuid = it["anchor"]["uuid"]
309
310
  return {
310
- "item_key": _claude_item_key(session_id, it["anchor"]["uuid"]),
311
+ "item_key": _claude_item_key(session_id, own_uuid),
311
312
  "kind": it["kind"],
312
313
  "timestamp_utc": it.get("ts"),
313
314
  "model": it.get("model"),
314
315
  "blocks": it.get("blocks", []),
315
316
  "cost_usd": it.get("cost_usd"),
316
317
  "tokens": _claude_tokens_union(it.get("tokens")),
318
+ "member_item_keys": [
319
+ _claude_item_key(session_id, uuid)
320
+ for uuid in it.get("member_uuids", []) if uuid != own_uuid
321
+ ],
322
+ "subagent_key": it.get("subagent_key"),
323
+ "parent_item_key": (
324
+ _claude_item_key(session_id, it["parent_uuid"])
325
+ if it.get("parent_uuid") is not None else None
326
+ ),
327
+ "is_sidechain": bool(it.get("is_sidechain")),
328
+ "meta_kind": it.get("meta_kind"),
329
+ "skill_name": it.get("skill_name"),
330
+ "command_name": it.get("command_name"),
331
+ "cache_failure": it.get("cache_failure"),
332
+ }
333
+
334
+
335
+ def _map_claude_subagent_meta(session_id: str, values: dict | None) -> dict:
336
+ """Translate the one identity-bearing field in legacy subagent metadata."""
337
+ return {
338
+ key: ({
339
+ **meta,
340
+ "spawn_uuid": (
341
+ _claude_item_key(session_id, meta["spawn_uuid"])
342
+ if meta.get("spawn_uuid") is not None else None
343
+ ),
344
+ } if isinstance(meta, dict) else meta)
345
+ for key, meta in (values or {}).items()
317
346
  }
318
347
 
319
348
 
@@ -386,6 +415,8 @@ def _claude_detail(
386
415
  "title": res["title"],
387
416
  "items": neutral_items,
388
417
  "page": page,
418
+ "subagent_meta": _map_claude_subagent_meta(
419
+ session_id, res.get("subagent_meta")),
389
420
  "children": [],
390
421
  "parent": None,
391
422
  "total_cost_usd": res["cost_usd"],
@@ -399,33 +430,98 @@ def _claude_detail(
399
430
  def _claude_outline(
400
431
  conn: sqlite3.Connection, session_id: str, conversation_key: str,
401
432
  ) -> dict:
402
- """Claude outline envelope (§5.6). Reuses the existing kernel outline for
403
- ``stats``/``files``; the per-turn ``item_key`` + block-kind counts are built
404
- from the SAME assembled items the detail pages, so outline and detail item
405
- keys align exactly. ``children`` is empty (no native threading)."""
433
+ """Claude outline envelope (§5.6), without weakening the native outline.
434
+
435
+ The legacy kernel is the authority for navigation: it already derives tool
436
+ failures, subagent topology, cache rebuilds, files, task completion, and the
437
+ complete turn skeleton from the same assembled items as detail. This
438
+ adapter changes only identity-bearing fields from native UUIDs to neutral
439
+ item keys and names the provider-neutral file fields. Re-deriving a smaller
440
+ outline from assembled blocks here previously stripped precisely the facts
441
+ the qualified reader needs (#491).
442
+ """
406
443
  o = lcq.get_conversation_outline(conn, session_id)
407
444
  if o is None:
408
445
  return {"status": "not_found", "conversation_key": conversation_key}
409
- asm = lcq._assemble_session_memoized(conn, session_id)
446
+
447
+ def item_key(uuid):
448
+ return _claude_item_key(session_id, uuid) if uuid is not None else None
449
+
410
450
  turns = []
411
- for it in asm["items"]:
412
- kinds: dict[str, int] = {}
413
- for b in it.get("blocks", []):
414
- bk = b.get("kind")
415
- if bk:
416
- kinds[bk] = kinds.get(bk, 0) + 1
417
- turns.append({
418
- "item_key": _claude_item_key(session_id, it["anchor"]["uuid"]),
419
- "label": lcq._outline_label(it.get("text", "")),
420
- "timestamp_utc": it.get("ts"),
421
- "kinds": kinds,
451
+ for turn in o["turns"]:
452
+ own_key = item_key(turn["uuid"])
453
+ neutral = {
454
+ "item_key": own_key,
455
+ "kind": turn["kind"],
456
+ "label": turn["label"],
457
+ "timestamp_utc": turn.get("ts"),
458
+ "kinds": {turn["kind"]: 1},
459
+ "member_item_keys": [
460
+ item_key(uuid) for uuid in turn.get("member_uuids", [])
461
+ if uuid != turn["uuid"]
462
+ ],
463
+ "subagent_key": turn.get("subagent_key"),
464
+ "parent_item_key": item_key(turn.get("parent_uuid")),
465
+ "is_sidechain": bool(turn.get("is_sidechain")),
466
+ }
467
+ for field in (
468
+ "tools", "thinking", "model", "tokens", "meta_kind",
469
+ "skill_name", "cache_failure"):
470
+ if field in turn:
471
+ neutral[field] = turn[field]
472
+ turns.append(neutral)
473
+
474
+ stats = dict(o["stats"])
475
+ cache_failures = stats.get("cache_failures")
476
+ if isinstance(cache_failures, dict):
477
+ cache_failures = dict(cache_failures)
478
+ cache_failures["rebuilds"] = [
479
+ {**row, "uuid": item_key(row.get("uuid"))}
480
+ for row in cache_failures.get("rebuilds", [])
481
+ ]
482
+ stats["cache_failures"] = cache_failures
483
+
484
+ files = []
485
+ for file in o.get("files", []):
486
+ touches = [{
487
+ "item_key": item_key(touch.get("uuid")),
488
+ "timestamp_utc": None,
489
+ "tool_use_id": touch.get("tool_use_id"),
490
+ "op": touch.get("op"),
491
+ "added": touch.get("add"),
492
+ "removed": touch.get("del"),
493
+ } for touch in file.get("touches", [])]
494
+ tools = list(dict.fromkeys(
495
+ touch["op"] for touch in touches if touch.get("op")))
496
+ files.append({
497
+ "file_path": file.get("path"),
498
+ # The count-only neutral file shape requires a tool label. Rich
499
+ # Claude entries retain every native operation instead of claiming
500
+ # that Write/MultiEdit touches were Edit calls.
501
+ "tool": ",".join(tools),
502
+ "count": len(touches),
503
+ "added": file.get("add"),
504
+ "removed": file.get("del"),
505
+ "touches": touches,
422
506
  })
507
+
508
+ subagent_meta = _map_claude_subagent_meta(
509
+ session_id, o.get("subagent_meta"))
510
+ task_completion = o.get("task_completion")
511
+ if isinstance(task_completion, dict):
512
+ task_completion = {
513
+ **task_completion,
514
+ "anchor_uuid": item_key(task_completion.get("anchor_uuid")),
515
+ }
423
516
  return {
424
517
  "status": "ok",
425
518
  "conversation_key": conversation_key,
426
519
  "turns": turns,
427
- "stats": o["stats"],
428
- "files": o["files"],
520
+ "subagent_meta": subagent_meta,
521
+ "subagent_costs": o.get("subagent_costs", {}),
522
+ "stats": stats,
523
+ "files": files,
524
+ "task_completion": task_completion,
429
525
  "children": [],
430
526
  }
431
527
 
@@ -531,6 +627,25 @@ def neutral_browse(
531
627
  raise ValueError(f"unknown source: {source!r}")
532
628
 
533
629
 
630
+ def neutral_facets(
631
+ conn: sqlite3.Connection, *, source: str, effective_speed: str | None = None,
632
+ ) -> dict:
633
+ """Facet-only collection envelope for one source.
634
+
635
+ Codex has a dedicated rollup projection so the facets request never builds
636
+ or prices a browse page that its transport discards. Claude retains its
637
+ established browse-derived facet contract.
638
+ """
639
+ speed = effective_speed or _DEFAULT_SPEED
640
+ if source == "codex":
641
+ return q.list_codex_conversation_facets(conn)
642
+ if source == "claude":
643
+ env = _claude_browse(conn, effective_speed=speed)
644
+ return {"status": env.get("status"), "facets": env.get("facets") or {
645
+ "projects": [], "models": []}}
646
+ raise ValueError(f"unknown source: {source!r}")
647
+
648
+
534
649
  def neutral_detail(
535
650
  conn: sqlite3.Connection, ref: str, *, effective_speed: str | None = None,
536
651
  after: str | None = None, before: str | None = None,
@@ -725,6 +840,8 @@ def _claude_export(
725
840
  def neutral_find(
726
841
  conn: sqlite3.Connection, ref: str, query: str, *, kind: str = "all",
727
842
  regex: bool = False, case: bool = False, effective_speed: str | None = None,
843
+ limit: int = 100, cursor: str | None = None, direction: str = "next",
844
+ around: str | None = None,
728
845
  ) -> dict:
729
846
  """In-conversation find for a neutral reference (§3.1). An unknown ``kind``
730
847
  raises ``ValueError`` (route → 400); an unknown/garbage ref → ``not_found``.
@@ -736,8 +853,23 @@ def neutral_find(
736
853
  if cref is None:
737
854
  return {"status": "not_found", "conversation_key": ref}
738
855
  if cref.source == "codex":
739
- return q.find_in_codex_conversation(
740
- conn, cref.conversation_key, query, kind=kind, regex=regex, case=case)
856
+ try:
857
+ return q.find_occurrences_in_codex_conversation(
858
+ conn,
859
+ cref.conversation_key,
860
+ query,
861
+ kind=kind,
862
+ regex=regex,
863
+ case_sensitive=case,
864
+ limit=limit,
865
+ cursor=cursor,
866
+ direction=direction,
867
+ around=around,
868
+ )
869
+ except q.InvalidFindCursor:
870
+ return {"status": "invalid_find_cursor"}
871
+ except q.StaleFindCursor:
872
+ return {"status": "stale_find_cursor"}
741
873
  return _claude_find(
742
874
  conn, cref.native_key, cref.conversation_key, query,
743
875
  kind=kind, regex=regex, case=case)
@@ -51,7 +51,9 @@ def watch_step(files, seen, *, stat_fn=file_sig, ingest_fn, committed_sig_fn=Non
51
51
  cursor, in session_files, lags the new disk size). committed_sig_fn defaults
52
52
  to stat_fn for pure unit tests with no cache. A contended/declined/failed
53
53
  ingest leaves `seen` untouched so the next cycle retries (the 5s backstop is
54
- the floor)."""
54
+ the floor). Account-scoped callers may additionally expose
55
+ ``stats.targeted_visible``: the clean ingest still advances ``seen``, but a
56
+ false value suppresses the tail frame when only another account changed."""
55
57
  committed_sig_fn = committed_sig_fn or stat_fn
56
58
  changed = changed_paths(files, seen, stat_fn)
57
59
  if not changed:
@@ -64,4 +66,4 @@ def watch_step(files, seen, *, stat_fn=file_sig, ingest_fn, committed_sig_fn=Non
64
66
  sig = committed_sig_fn(p)
65
67
  if sig is not None:
66
68
  new_seen[p] = sig
67
- return new_seen, True
69
+ return new_seen, bool(getattr(stats, "targeted_visible", True))