cctally 1.90.1 → 1.92.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +74 -0
- package/README.md +2 -2
- package/bin/_cctally_cache.py +863 -74
- package/bin/_cctally_config.py +57 -0
- package/bin/_cctally_core.py +53 -8
- package/bin/_cctally_dashboard.py +146 -5
- package/bin/_cctally_dashboard_conversation.py +164 -18
- package/bin/_cctally_dashboard_envelope.py +69 -12
- package/bin/_cctally_dashboard_sources.py +27 -1
- package/bin/_cctally_db.py +372 -10
- package/bin/_cctally_doctor.py +18 -1
- package/bin/_cctally_journal.py +535 -13
- package/bin/_cctally_journal_repair.py +6 -0
- package/bin/_cctally_parser.py +6 -0
- package/bin/_cctally_quota.py +171 -55
- package/bin/_cctally_record.py +13 -1
- package/bin/_cctally_rederive.py +4 -0
- package/bin/_cctally_store.py +311 -6
- package/bin/_cctally_transcript.py +32 -2
- package/bin/_lib_cache_report.py +8 -3
- package/bin/_lib_cache_report_wire.py +8 -20
- package/bin/_lib_codex_conversation.py +959 -81
- package/bin/_lib_codex_conversation_query.py +2792 -167
- package/bin/_lib_codex_find_projection.py +370 -0
- package/bin/_lib_codex_harness_preamble.py +176 -0
- package/bin/_lib_codex_hooks.py +5 -3
- package/bin/_lib_codex_js_scan.py +254 -0
- package/bin/_lib_codex_landmarks.py +309 -0
- package/bin/_lib_codex_reasoning_headings.py +73 -0
- package/bin/_lib_codex_segments.py +259 -0
- package/bin/_lib_codex_title_clean.py +116 -0
- package/bin/_lib_conversation_dispatch.py +153 -21
- package/bin/_lib_conversation_watch.py +4 -2
- package/bin/_lib_dashboard_sources.py +33 -32
- package/bin/_lib_doctor.py +64 -0
- package/bin/_lib_quota_alert_axes.py +31 -34
- package/bin/_lib_stats_damage.py +523 -0
- package/bin/cctally +5 -0
- package/dashboard/static/assets/index-BEzzJtUd.js +97 -0
- package/dashboard/static/assets/index-DnWdv8um.css +1 -0
- package/dashboard/static/dashboard.html +2 -2
- package/package.json +9 -1
- package/dashboard/static/assets/index-Bar8-S1i.css +0 -1
- package/dashboard/static/assets/index-CRogVlEC.js +0 -92
|
@@ -12,20 +12,42 @@ the caller at its I/O boundary, never here (§5.4).
|
|
|
12
12
|
Public names imported verbatim by the S7 dispatch layer — do not rename:
|
|
13
13
|
``codex_normalization_authoritative``, ``codex_item_key``,
|
|
14
14
|
``get_codex_conversation``, ``get_codex_conversation_outline``,
|
|
15
|
-
``list_codex_conversations``, ``
|
|
15
|
+
``list_codex_conversations``, ``list_codex_conversation_facets``,
|
|
16
|
+
``search_codex_conversations``,
|
|
16
17
|
``CODEX_SEARCH_KINDS``.
|
|
17
18
|
"""
|
|
18
19
|
from __future__ import annotations
|
|
19
20
|
|
|
21
|
+
import base64
|
|
22
|
+
import binascii
|
|
23
|
+
import collections
|
|
24
|
+
import contextlib
|
|
20
25
|
import hashlib
|
|
21
26
|
import json
|
|
22
27
|
import os
|
|
23
28
|
import re
|
|
24
29
|
import sqlite3
|
|
30
|
+
import threading
|
|
31
|
+
from collections import deque
|
|
25
32
|
|
|
26
33
|
import _lib_codex_conversation as kern
|
|
34
|
+
import _lib_codex_landmarks as landmarks
|
|
35
|
+
import _lib_codex_segments as segkern
|
|
36
|
+
from _lib_codex_find_projection import (
|
|
37
|
+
ProjectedLeaf,
|
|
38
|
+
RenderLeaf,
|
|
39
|
+
iter_literal_ranges,
|
|
40
|
+
iter_regex_ranges,
|
|
41
|
+
literal_ranges,
|
|
42
|
+
project_markdown,
|
|
43
|
+
project_plain,
|
|
44
|
+
regex_ranges,
|
|
45
|
+
slice_range_to_leaves,
|
|
46
|
+
)
|
|
47
|
+
from _lib_codex_title_clean import clean_codex_title
|
|
27
48
|
from _lib_conversation import _strip_ansi
|
|
28
|
-
from _lib_conversation_query import
|
|
49
|
+
from _lib_conversation_query import (
|
|
50
|
+
_FULL_PAYLOAD_CEILING, _first_nonblank_line, _parse_outline_ts)
|
|
29
51
|
from _lib_pricing import _calculate_codex_entry_cost
|
|
30
52
|
|
|
31
53
|
# ── constants ────────────────────────────────────────────────────────────────
|
|
@@ -167,9 +189,30 @@ def codex_item_key(
|
|
|
167
189
|
line_offset, content_digest)`` — no population-relative ordinals, so
|
|
168
190
|
deleting an earlier duplicate or an out-of-order multi-file append never
|
|
169
191
|
moves an existing key, and a same-offset content replacement changes it.
|
|
192
|
+
|
|
193
|
+
Segments 1..N of a split turn use a third shape, ``(conversation_key, "seg",
|
|
194
|
+
fingerprint(source_path), line_offset, content_digest)``, computed from the
|
|
195
|
+
segment's anchor row (#463 S1). ``turn_id`` is deliberately NOT an input:
|
|
196
|
+
hashing the row alone is ordinal-free, so the key never changes while that
|
|
197
|
+
row's content stays at the same offset. Segment 0 does NOT take this shape —
|
|
198
|
+
it inherits its turn's ``"response"`` key unchanged, which is why every deep
|
|
199
|
+
link, permalink, bookmark, reading position and outline entry issued before
|
|
200
|
+
segmentation still resolves to the head of its turn by construction, with no
|
|
201
|
+
alias table and no migration.
|
|
202
|
+
|
|
203
|
+
The ``"seg"`` domain separator is what keeps a segment key from colliding
|
|
204
|
+
with the ``"row"`` key of the same anchor row.
|
|
170
205
|
"""
|
|
171
206
|
if klass == "response":
|
|
172
207
|
parts = ("turn", conversation_key or "", turn_id or "")
|
|
208
|
+
elif klass == "segment":
|
|
209
|
+
parts = (
|
|
210
|
+
"seg",
|
|
211
|
+
conversation_key or "",
|
|
212
|
+
_source_path_fingerprint(source_path),
|
|
213
|
+
"" if line_offset is None else str(line_offset),
|
|
214
|
+
content_digest or "",
|
|
215
|
+
)
|
|
173
216
|
else:
|
|
174
217
|
parts = (
|
|
175
218
|
"row",
|
|
@@ -209,13 +252,24 @@ def codex_block_key(
|
|
|
209
252
|
line_offset: int | None,
|
|
210
253
|
content_digest: str | None,
|
|
211
254
|
) -> str:
|
|
212
|
-
"""Opaque, ordinal-free
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
255
|
+
"""Opaque, ordinal-free anchor over a physical row's row-class identity (§3.4).
|
|
256
|
+
|
|
257
|
+
Carried by EVERY row-backed block since #463 S2. Stable identity and
|
|
258
|
+
payload-capability are DIFFERENT properties: a prose block has a key and no
|
|
259
|
+
retained payload, so a consumer must never infer payload availability from
|
|
260
|
+
the presence of a key. Payload readback is unchanged and gains no surface —
|
|
261
|
+
``_locate_payload_block`` still resolves only tool_call rows and the marker
|
|
262
|
+
and lifecycle event families, and refuses everything else.
|
|
263
|
+
|
|
264
|
+
Same domain-separated hash family as ``codex_item_key``'s row class —
|
|
265
|
+
``(conversation_key, fingerprint(source_path), line_offset,
|
|
266
|
+
content_digest)`` — with a DISTINCT domain, so a block key never collides
|
|
267
|
+
with an item key. Stable per block, unique per physical row: a same-offset
|
|
268
|
+
content replacement changes it (content_digest moves), an out-of-order
|
|
269
|
+
append elsewhere leaves it (no population-relative ordinals). Segment
|
|
270
|
+
boundaries never affect it, because nothing in the inputs is
|
|
271
|
+
population-relative.
|
|
272
|
+
"""
|
|
219
273
|
parts = (
|
|
220
274
|
conversation_key or "",
|
|
221
275
|
_source_path_fingerprint(source_path),
|
|
@@ -250,35 +304,368 @@ def _load_conversation_rows(conn: sqlite3.Connection, conversation_key: str) ->
|
|
|
250
304
|
]
|
|
251
305
|
|
|
252
306
|
|
|
253
|
-
|
|
307
|
+
# Narrow index columns (#463 S1, spec section 3, Phase A). Everything except
|
|
308
|
+
# ``text`` — and the event payloads are not touched at all. ``detail_json`` is
|
|
309
|
+
# REQUIRED, not optional: all three fold passes inside ``canonical_items`` parse
|
|
310
|
+
# it, and the reasoning title boundary comes from the stored projection, which
|
|
311
|
+
# ``search_thinking`` cannot supply (it is capped at 16,000 characters and stores
|
|
312
|
+
# ``summary + "\n" + body`` for a response item, so a title and a title-plus-body
|
|
313
|
+
# are indistinguishable there).
|
|
314
|
+
#
|
|
315
|
+
# ``length(CAST(detail_json AS BLOB))`` rather than ``length(detail_json)``:
|
|
316
|
+
# SQLite's ``length()`` on TEXT counts CHARACTERS, and the stored JSON is emitted
|
|
317
|
+
# with ``ensure_ascii=False``, so a character count understates a non-ASCII
|
|
318
|
+
# detail and would let a segment exceed its stated ceiling.
|
|
319
|
+
_NARROW_ROW_COLS = (
|
|
320
|
+
"source_path, line_offset, timestamp_utc, turn_id, call_id, kind, "
|
|
321
|
+
"event_type, record_family, model, content_digest, content_len, "
|
|
322
|
+
"detail_json, length(CAST(detail_json AS BLOB))"
|
|
323
|
+
)
|
|
324
|
+
|
|
325
|
+
# Bound on the ``line_offset`` list bound into one hydration query. SQLite's
|
|
326
|
+
# default host-parameter limit is 999 on older builds, and a full page can carry
|
|
327
|
+
# more positions than that, so the reads are chunked per source file.
|
|
328
|
+
_HYDRATE_CHUNK = 400
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def _load_conversation_index_rows(
|
|
254
332
|
conn: sqlite3.Connection, conversation_key: str,
|
|
333
|
+
) -> tuple[list, dict[tuple[str, int], int]]:
|
|
334
|
+
"""Phase A's narrow read: ``(rows, detail_bytes_by_position)``.
|
|
335
|
+
|
|
336
|
+
Rows come back as ordinary ``CodexNormalizedRow`` objects with ``text``,
|
|
337
|
+
``search_tool`` and ``search_thinking`` blanked, so ``pair_mirrors`` and
|
|
338
|
+
``canonical_items`` run unchanged — both key on ``turn_id``, ``kind``,
|
|
339
|
+
``content_digest``, ``content_len``, ``record_family`` and the physical
|
|
340
|
+
position, none of which lives in the excluded columns.
|
|
341
|
+
|
|
342
|
+
Excluding ``text`` defers the bulk: on the heaviest conversation in the
|
|
343
|
+
corpus ``content_len`` totals 84.6 MB against 7.1 MB of ``detail_json``.
|
|
344
|
+
"""
|
|
345
|
+
rows: list = []
|
|
346
|
+
detail_bytes: dict[tuple[str, int], int] = {}
|
|
347
|
+
for (source_path, line_offset, timestamp_utc, turn_id, call_id, kind,
|
|
348
|
+
event_type, record_family, model, content_digest, content_len,
|
|
349
|
+
detail_json, detail_len) in conn.execute(
|
|
350
|
+
"SELECT " + _NARROW_ROW_COLS + " FROM codex_conversation_messages "
|
|
351
|
+
"WHERE conversation_key = ? "
|
|
352
|
+
"ORDER BY timestamp_utc, source_path, line_offset",
|
|
353
|
+
(conversation_key,),
|
|
354
|
+
):
|
|
355
|
+
rows.append(kern.CodexNormalizedRow(
|
|
356
|
+
conversation_key=conversation_key, source_root_key="",
|
|
357
|
+
source_path=source_path, line_offset=line_offset,
|
|
358
|
+
timestamp_utc=timestamp_utc, turn_id=turn_id, call_id=call_id,
|
|
359
|
+
kind=kind, event_type=event_type, record_family=record_family,
|
|
360
|
+
model=model, text="", content_digest=content_digest,
|
|
361
|
+
content_len=content_len, detail_json=detail_json,
|
|
362
|
+
search_tool="", search_thinking=""))
|
|
363
|
+
detail_bytes[(source_path, line_offset)] = detail_len or 0
|
|
364
|
+
return rows, detail_bytes
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def _detail_bytes_of(rows) -> dict[tuple[str, int], int]:
|
|
368
|
+
"""``detail_json`` byte sizes for callers that already hold WIDE rows.
|
|
369
|
+
|
|
370
|
+
The same quantity ``_load_conversation_index_rows`` gets from
|
|
371
|
+
``length(CAST(detail_json AS BLOB))`` — BYTES, not characters, because the
|
|
372
|
+
stored JSON is emitted with ``ensure_ascii=False``.
|
|
373
|
+
"""
|
|
374
|
+
return {
|
|
375
|
+
(row.source_path, row.line_offset):
|
|
376
|
+
len((row.detail_json or "").encode("utf-8"))
|
|
377
|
+
for row in rows
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
def _chunk_positions(positions) -> dict[str, list[list[int]]]:
|
|
382
|
+
"""Group physical positions by source file, chunked for a bound IN clause."""
|
|
383
|
+
by_path: dict[str, list[int]] = {}
|
|
384
|
+
for source_path, line_offset in positions:
|
|
385
|
+
by_path.setdefault(source_path, []).append(line_offset)
|
|
386
|
+
return {
|
|
387
|
+
path: [offsets[i:i + _HYDRATE_CHUNK]
|
|
388
|
+
for i in range(0, len(offsets), _HYDRATE_CHUNK)]
|
|
389
|
+
for path, offsets in by_path.items()
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
def _load_rows_at_positions(
|
|
394
|
+
conn: sqlite3.Connection, conversation_key: str, positions,
|
|
395
|
+
) -> dict[tuple[str, int], object]:
|
|
396
|
+
"""Phase C's wide read, scoped to one page's physical positions."""
|
|
397
|
+
hydrated: dict[tuple[str, int], object] = {}
|
|
398
|
+
for path, chunks in _chunk_positions(positions).items():
|
|
399
|
+
for chunk in chunks:
|
|
400
|
+
marks = ",".join("?" for _ in chunk)
|
|
401
|
+
for row in conn.execute(
|
|
402
|
+
"SELECT " + _ROW_COLS + " FROM codex_conversation_messages "
|
|
403
|
+
f"WHERE conversation_key = ? AND source_path = ? "
|
|
404
|
+
f"AND line_offset IN ({marks})",
|
|
405
|
+
(conversation_key, path, *chunk),
|
|
406
|
+
):
|
|
407
|
+
built = kern.CodexNormalizedRow(*row)
|
|
408
|
+
hydrated[(built.source_path, built.line_offset)] = built
|
|
409
|
+
return hydrated
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def _iter_row_payloads(
|
|
413
|
+
conn: sqlite3.Connection, conversation_key: str, positions=None,
|
|
414
|
+
):
|
|
415
|
+
"""Yield ``(position, record_type, payload)`` for retained physical payloads.
|
|
416
|
+
|
|
417
|
+
ONE copy of the event-table read: the per-source-path chunking, the bound
|
|
418
|
+
``IN (…)`` construction, the defensive re-filter and the
|
|
419
|
+
``{"payload": …}`` unwrap. ``_load_row_payloads`` collects this into a dict
|
|
420
|
+
and ``_derive_outline_events`` consumes it one row at a time, which is the
|
|
421
|
+
only difference between them — duplicating the read to get the streaming
|
|
422
|
+
form would mean a later correction to the unwrap or the chunking has to be
|
|
423
|
+
made twice, and the second is easy to miss.
|
|
424
|
+
|
|
425
|
+
``positions`` scopes the read to a known set (#463 S1, Phase C). Passing
|
|
426
|
+
``None`` reads the whole conversation, which the export path still wants and
|
|
427
|
+
which #463 S4 §4.1 forbids on the outline route.
|
|
428
|
+
|
|
429
|
+
A row whose ``payload_json`` does not parse, or whose ``payload`` member is
|
|
430
|
+
not an object, yields nothing — the caller therefore sees no entry for that
|
|
431
|
+
position rather than an empty one.
|
|
432
|
+
"""
|
|
433
|
+
def _batches():
|
|
434
|
+
"""One cursor per bound chunk, opened only when its turn comes."""
|
|
435
|
+
if positions is None:
|
|
436
|
+
yield conn.execute(
|
|
437
|
+
"SELECT source_path,line_offset,record_type,payload_json "
|
|
438
|
+
"FROM codex_conversation_events WHERE conversation_key = ?",
|
|
439
|
+
(conversation_key,)), None
|
|
440
|
+
return
|
|
441
|
+
scope = set(positions)
|
|
442
|
+
for path, chunks in _chunk_positions(scope).items():
|
|
443
|
+
for chunk in chunks:
|
|
444
|
+
yield conn.execute(
|
|
445
|
+
"SELECT source_path,line_offset,record_type,payload_json "
|
|
446
|
+
"FROM codex_conversation_events "
|
|
447
|
+
"WHERE conversation_key = ? AND source_path = ? "
|
|
448
|
+
f"AND line_offset IN ({','.join('?' for _ in chunk)})",
|
|
449
|
+
(conversation_key, path, *chunk)), scope
|
|
450
|
+
|
|
451
|
+
for cursor, wanted in _batches():
|
|
452
|
+
for source_path, line_offset, record_type, payload_json in cursor:
|
|
453
|
+
position = (source_path, line_offset)
|
|
454
|
+
if wanted is not None and position not in wanted:
|
|
455
|
+
continue
|
|
456
|
+
try:
|
|
457
|
+
obj = json.loads(payload_json or "{}")
|
|
458
|
+
except (json.JSONDecodeError, TypeError):
|
|
459
|
+
continue
|
|
460
|
+
payload = obj.get("payload") if isinstance(obj, dict) else None
|
|
461
|
+
if isinstance(payload, dict):
|
|
462
|
+
yield position, record_type, payload
|
|
463
|
+
|
|
464
|
+
|
|
465
|
+
def _load_row_payloads(
|
|
466
|
+
conn: sqlite3.Connection, conversation_key: str, positions=None,
|
|
255
467
|
) -> dict[tuple[str, int], tuple[str | None, dict]]:
|
|
256
468
|
"""Retained physical payloads for query-time card shaping.
|
|
257
469
|
|
|
258
470
|
The retained payload remains the authoritative source used for full-payload
|
|
259
471
|
readback and a defensive read-time re-shape; contract v3 also persists the
|
|
260
472
|
same bounded card so replay-derived rollups and logical item counts converge.
|
|
473
|
+
|
|
474
|
+
``positions`` scopes the read to one page (#463 S1, Phase C). Passing None
|
|
475
|
+
keeps the whole-conversation behaviour, which the export path still wants.
|
|
261
476
|
"""
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
):
|
|
268
|
-
try:
|
|
269
|
-
obj = json.loads(payload_json or "{}")
|
|
270
|
-
except (json.JSONDecodeError, TypeError):
|
|
271
|
-
continue
|
|
272
|
-
payload = obj.get("payload") if isinstance(obj, dict) else None
|
|
273
|
-
if isinstance(payload, dict):
|
|
274
|
-
result[(source_path, line_offset)] = (record_type, payload)
|
|
275
|
-
return result
|
|
477
|
+
return {
|
|
478
|
+
position: (record_type, payload)
|
|
479
|
+
for position, record_type, payload
|
|
480
|
+
in _iter_row_payloads(conn, conversation_key, positions)
|
|
481
|
+
}
|
|
276
482
|
|
|
277
483
|
|
|
278
484
|
def _row_payload(row, payloads: dict) -> tuple[str | None, dict] | None:
|
|
279
485
|
return payloads.get((row.source_path, row.line_offset))
|
|
280
486
|
|
|
281
487
|
|
|
488
|
+
# The three lifecycle events that carry an outcome the §6.4 classification reads.
|
|
489
|
+
# A `tool_output` row is the fourth source and is selected by `kind`, not by
|
|
490
|
+
# event type, because it is a response item rather than an event.
|
|
491
|
+
_S4_OUTCOME_EVENTS = frozenset(
|
|
492
|
+
{"patch_apply_end", "web_search_end", "mcp_tool_call_end"})
|
|
493
|
+
|
|
494
|
+
|
|
495
|
+
# ── the outline derivation cache (#463 S4, D2's fallback) ────────────────────
|
|
496
|
+
#
|
|
497
|
+
# Task 11's re-measurement breached §4.7's first ceiling. On the heaviest
|
|
498
|
+
# production conversation the warm outline costs 332 ms, of which the event
|
|
499
|
+
# payload pass is 242 ms, against 229 ms for the detail page that opens beside
|
|
500
|
+
# it — so the outline HAD become the critical path on open, and the route is
|
|
501
|
+
# refetched on every live-tail growth push. §4.7 says a breach escalates to D2's
|
|
502
|
+
# fallback inside the session rather than deferring it, and this is that.
|
|
503
|
+
#
|
|
504
|
+
# **Why a position-keyed cache is sound.** A derivation entry is keyed by
|
|
505
|
+
# ``(source_path, line_offset)`` — a byte offset into an append-only rollout
|
|
506
|
+
# file. A line at a given offset is immutable, so the verdict for a position
|
|
507
|
+
# cannot change while the row exists, and a re-ingest rewrites the same bytes at
|
|
508
|
+
# the same offsets. The cache therefore EXTENDS on append rather than only
|
|
509
|
+
# hitting or missing: a growth push decodes the newly-in-scope positions and
|
|
510
|
+
# reuses every earlier one.
|
|
511
|
+
#
|
|
512
|
+
# **What the watermark is for.** Immutability covers append; it does not cover
|
|
513
|
+
# DELETION, and a deleted event row is exactly the case where the outline must
|
|
514
|
+
# stop reporting a verdict it can no longer support. The stored watermark is the
|
|
515
|
+
# cached prefix's ``(row count, max id)``, and reuse requires the count of rows
|
|
516
|
+
# still at ``id <= max id`` to equal it. Ids are ``AUTOINCREMENT`` and only
|
|
517
|
+
# increase, so no later insert can land inside that prefix — a preserved count
|
|
518
|
+
# therefore proves no row of the prefix was deleted. The check is one covering
|
|
519
|
+
# index read, measured at 0.18 ms on a conversation with 5,950 event rows.
|
|
520
|
+
#
|
|
521
|
+
# **Absence stays absence.** A position that was in scope and produced no entry
|
|
522
|
+
# — payload gone, unparseable, or a shape no decoder recognises — is recorded as
|
|
523
|
+
# covered and is not retried. That is deliberate: ``_outline_error_count`` reads
|
|
524
|
+
# absence as the third state "could not classify", and a payload does not appear
|
|
525
|
+
# later at an offset it was missing from.
|
|
526
|
+
#
|
|
527
|
+
# In-process only, so no cross-version staleness is possible: the binary that
|
|
528
|
+
# filled an entry is the binary that reads it.
|
|
529
|
+
_OUTLINE_DERIVATION_CACHE_MAX = 4
|
|
530
|
+
_outline_derivation_cache: "collections.OrderedDict[str, dict]" = (
|
|
531
|
+
collections.OrderedDict())
|
|
532
|
+
_outline_derivation_lock = threading.Lock()
|
|
533
|
+
|
|
534
|
+
|
|
535
|
+
def reset_outline_derivation_cache() -> None:
|
|
536
|
+
"""Drop every cached derivation. For tests and for measurement runs."""
|
|
537
|
+
with _outline_derivation_lock:
|
|
538
|
+
_outline_derivation_cache.clear()
|
|
539
|
+
|
|
540
|
+
|
|
541
|
+
def _outline_event_watermark(
|
|
542
|
+
conn: sqlite3.Connection, conversation_key: str, prefix_max_id: int | None,
|
|
543
|
+
) -> tuple[int, int | None, int]:
|
|
544
|
+
"""``(row count, max id, rows still at id <= prefix_max_id)`` in one read."""
|
|
545
|
+
count, max_id, prefix = conn.execute(
|
|
546
|
+
"SELECT COUNT(*), MAX(id), COALESCE(SUM(id <= ?), 0) "
|
|
547
|
+
"FROM codex_conversation_events WHERE conversation_key = ?",
|
|
548
|
+
(prefix_max_id if prefix_max_id is not None else -1, conversation_key),
|
|
549
|
+
).fetchone()
|
|
550
|
+
return count, max_id, prefix
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
def _derive_outline_events(
|
|
554
|
+
conn: sqlite3.Connection, conversation_key: str, rows: list,
|
|
555
|
+
) -> landmarks.EventDerivation:
|
|
556
|
+
"""The outline's read-time pass over the retained payloads (#463 S4, §4.1).
|
|
557
|
+
|
|
558
|
+
SCOPED and STREAMING, and Task 1 measured both halves of that on the
|
|
559
|
+
production store rather than assuming them.
|
|
560
|
+
|
|
561
|
+
*Scoped*: the position set comes from ``rows`` — the wide
|
|
562
|
+
``codex_conversation_messages`` read the outline has already performed — so
|
|
563
|
+
the pass never selects an event row no landmark can come from. That is the
|
|
564
|
+
idiom ``get_codex_conversation`` already uses two call sites above.
|
|
565
|
+
``positions=None`` would decode every event row of the conversation, which on
|
|
566
|
+
the heaviest one is 92.1 MB of JSON. Measured, scoping is worth 24% of that
|
|
567
|
+
read (109-122 ms and 183.1 MB of Python heap against 160 ms and 211.7 MB) —
|
|
568
|
+
real, but far less than the spec's framing implies, because 93.7% of that
|
|
569
|
+
conversation's payload bytes ARE the rows S4 needs. Do not treat scoping as
|
|
570
|
+
having solved the cost.
|
|
571
|
+
|
|
572
|
+
*Streaming*: each payload is decoded to the small fact the outline wants and
|
|
573
|
+
then dropped, rather than being retained in a dict the way
|
|
574
|
+
``_load_row_payloads`` retains it. Both go through the one
|
|
575
|
+
``_iter_row_payloads`` read; retaining is the only thing that differs.
|
|
576
|
+
Measured like for like — load AND decode,
|
|
577
|
+
to completion — the retaining form costs 254 ms and 182.9 MB of Python heap
|
|
578
|
+
on the heaviest conversation and this one costs 232 ms and 22.7 MB. That is
|
|
579
|
+
9% faster and 8x less memory, and peak RSS is in scope for §4.7's gate
|
|
580
|
+
precisely because this route is refetched on every live-tail growth push
|
|
581
|
+
rather than once per open.
|
|
582
|
+
|
|
583
|
+
Read-time only. Every decoder call takes ``for_storage=False`` and nothing
|
|
584
|
+
here is written back.
|
|
585
|
+
"""
|
|
586
|
+
outcomes: dict[tuple[str, int], object] = {}
|
|
587
|
+
headings_wanted: dict[tuple[str, int], object] = {}
|
|
588
|
+
for row in rows:
|
|
589
|
+
position = (row.source_path, row.line_offset)
|
|
590
|
+
if row.kind == "tool_output":
|
|
591
|
+
outcomes[position] = "output"
|
|
592
|
+
elif row.kind == "event" and row.event_type in _S4_OUTCOME_EVENTS:
|
|
593
|
+
outcomes[position] = row.event_type
|
|
594
|
+
elif row.kind == "reasoning":
|
|
595
|
+
# The same stored-detail gate `_reasoning_headings` applies, so the
|
|
596
|
+
# outline's headings are the set the reader route publishes and not
|
|
597
|
+
# a second, wider one.
|
|
598
|
+
detail = _parse_detail(row.detail_json)
|
|
599
|
+
if isinstance(detail, dict) and isinstance(detail.get("reasoning"), dict):
|
|
600
|
+
headings_wanted[position] = True
|
|
601
|
+
wanted = set(outcomes) | set(headings_wanted)
|
|
602
|
+
derivation = landmarks.EventDerivation()
|
|
603
|
+
if not wanted:
|
|
604
|
+
return derivation
|
|
605
|
+
|
|
606
|
+
with _outline_derivation_lock:
|
|
607
|
+
entry = _outline_derivation_cache.get(conversation_key)
|
|
608
|
+
count, max_id, prefix = _outline_event_watermark(
|
|
609
|
+
conn, conversation_key, entry["max_id"] if entry else None)
|
|
610
|
+
reusable = (entry is not None
|
|
611
|
+
and entry["count"] == prefix
|
|
612
|
+
and count >= entry["count"])
|
|
613
|
+
covered: set = set()
|
|
614
|
+
if reusable:
|
|
615
|
+
covered = entry["covered"]
|
|
616
|
+
derivation.errors_by_position.update(entry["derivation"].errors_by_position)
|
|
617
|
+
derivation.patch_files_by_position.update(
|
|
618
|
+
entry["derivation"].patch_files_by_position)
|
|
619
|
+
derivation.headings_by_position.update(
|
|
620
|
+
entry["derivation"].headings_by_position)
|
|
621
|
+
missing = wanted - covered
|
|
622
|
+
|
|
623
|
+
for position, _record_type, payload in (
|
|
624
|
+
_iter_row_payloads(conn, conversation_key, missing) if missing else ()):
|
|
625
|
+
kind = outcomes.get(position)
|
|
626
|
+
if kind == "output":
|
|
627
|
+
decoded = kern.decode_tool_output_card(payload, for_storage=False)
|
|
628
|
+
if decoded is not None:
|
|
629
|
+
derivation.errors_by_position[position] = (
|
|
630
|
+
landmarks.classify_tool_failure(
|
|
631
|
+
{"terminal_output": decoded[0]}))
|
|
632
|
+
elif kind == "patch_apply_end":
|
|
633
|
+
card = kern.decode_patch_event_card(payload, for_storage=False)
|
|
634
|
+
if card is not None:
|
|
635
|
+
derivation.errors_by_position[position] = (
|
|
636
|
+
landmarks.classify_tool_failure({"patch": card}))
|
|
637
|
+
# Counted from the UNBOUNDED raw `changes`, not off the card above,
|
|
638
|
+
# whose shared 16,000-character budget silently undercounts a large
|
|
639
|
+
# diff (§4.5).
|
|
640
|
+
derivation.patch_files_by_position[position] = (
|
|
641
|
+
landmarks.patch_file_touches(payload))
|
|
642
|
+
elif kind in _S4_OUTCOME_EVENTS:
|
|
643
|
+
card = kern.decode_secondary_event_card(payload)
|
|
644
|
+
if card is not None:
|
|
645
|
+
family = "web" if kind == "web_search_end" else "mcp"
|
|
646
|
+
derivation.errors_by_position[position] = (
|
|
647
|
+
landmarks.classify_tool_failure(
|
|
648
|
+
{family: {"completion": card}}))
|
|
649
|
+
if position in headings_wanted:
|
|
650
|
+
texts = landmarks.reasoning_heading_texts(payload)
|
|
651
|
+
if texts:
|
|
652
|
+
derivation.headings_by_position[position] = texts
|
|
653
|
+
|
|
654
|
+
# Stored AFTER the pass, so a raising derivation leaves the previous entry
|
|
655
|
+
# rather than a half-filled one. The value is the object just returned: it
|
|
656
|
+
# is never mutated again, because the next extension copies its three maps
|
|
657
|
+
# into a fresh `EventDerivation` above rather than adding to this one.
|
|
658
|
+
with _outline_derivation_lock:
|
|
659
|
+
_outline_derivation_cache[conversation_key] = {
|
|
660
|
+
"count": count, "max_id": max_id,
|
|
661
|
+
"covered": covered | wanted, "derivation": derivation,
|
|
662
|
+
}
|
|
663
|
+
_outline_derivation_cache.move_to_end(conversation_key)
|
|
664
|
+
while len(_outline_derivation_cache) > _OUTLINE_DERIVATION_CACHE_MAX:
|
|
665
|
+
_outline_derivation_cache.popitem(last=False)
|
|
666
|
+
return derivation
|
|
667
|
+
|
|
668
|
+
|
|
282
669
|
def _row_display(row) -> str:
|
|
283
670
|
"""The row's display/search text from whichever column carries it."""
|
|
284
671
|
return row.text or row.search_thinking or row.search_tool or ""
|
|
@@ -312,6 +699,50 @@ def _item_kind(item: dict) -> str:
|
|
|
312
699
|
return item["anchor_row"].kind # unturned: the row's own provider kind
|
|
313
700
|
|
|
314
701
|
|
|
702
|
+
def _item_model(item: dict) -> str | None:
|
|
703
|
+
"""The model a canonical tier-1 item states, or ``None`` (§4.2).
|
|
704
|
+
|
|
705
|
+
The anchor row first, then the item's own rows in order. Reading the anchor
|
|
706
|
+
row ALONE under-reported Codex model usage about sixfold: most Codex
|
|
707
|
+
response items anchor on a ``reasoning`` row, which carries no model, so a
|
|
708
|
+
conversation with 13 outline turns holding assistant rows and 82 assistant
|
|
709
|
+
rows in total rendered ``gpt-5.6-sol x2``, and 182 of 200 Codex
|
|
710
|
+
conversations in the production store reported a model total under a third
|
|
711
|
+
of their turn count. §4.2 defines the counting unit as the canonical tier-1
|
|
712
|
+
assistant TURN, mirroring the Claude side, whose histogram sums to exactly
|
|
713
|
+
``stats.turns.assistant``; the anchor row is one row of that turn.
|
|
714
|
+
"""
|
|
715
|
+
anchor = item["anchor_row"].model
|
|
716
|
+
if anchor:
|
|
717
|
+
return anchor
|
|
718
|
+
for row in item["rows"]:
|
|
719
|
+
if row.model:
|
|
720
|
+
return row.model
|
|
721
|
+
return None
|
|
722
|
+
|
|
723
|
+
|
|
724
|
+
def _outline_outcome_positions(rows: list) -> set:
|
|
725
|
+
"""Every position whose row could carry an outcome verdict this request."""
|
|
726
|
+
return {(row.source_path, row.line_offset) for row in rows
|
|
727
|
+
if row.kind == "tool_output"
|
|
728
|
+
or (row.kind == "event" and row.event_type in _S4_OUTCOME_EVENTS)}
|
|
729
|
+
|
|
730
|
+
|
|
731
|
+
def _outline_failing_calls(derivation, outcome_positions: set,
|
|
732
|
+
call_by_position: dict) -> set:
|
|
733
|
+
"""The calls this request found failing, charged to the call they fold into.
|
|
734
|
+
|
|
735
|
+
Intersected with ``outcome_positions`` because ``errors_by_position`` is
|
|
736
|
+
CACHED: the derivation is retained per conversation and EXTENDED on a
|
|
737
|
+
growth push, so the map can hold verdicts for positions the current request
|
|
738
|
+
never read, while every other consumer indexes it by a current position.
|
|
739
|
+
Defensive rather than a reproduction of an observed miscount.
|
|
740
|
+
"""
|
|
741
|
+
return {call_by_position.get(position, position)
|
|
742
|
+
for position in derivation.failing_positions()
|
|
743
|
+
if position in outcome_positions}
|
|
744
|
+
|
|
745
|
+
|
|
315
746
|
def _item_meta(item: dict) -> dict | None:
|
|
316
747
|
if item["klass"] != "meta":
|
|
317
748
|
return None
|
|
@@ -329,8 +760,194 @@ def _item_meta(item: dict) -> dict | None:
|
|
|
329
760
|
return meta
|
|
330
761
|
|
|
331
762
|
|
|
763
|
+
def _turn_scoped_call_owner_count(rows: list) -> dict[str, int]:
|
|
764
|
+
"""How many ``tool_call`` rows own each call id, counted over the WHOLE turn.
|
|
765
|
+
|
|
766
|
+
This must stay turn-scoped (#463 S1, spec section 1). Recomputing it over a
|
|
767
|
+
page-local segment would make a call id that appears twice in a turn but once
|
|
768
|
+
in the page look uniquely owned, and a ``tool_output`` would then fold into
|
|
769
|
+
the wrong call.
|
|
770
|
+
"""
|
|
771
|
+
counts: dict[str, int] = {}
|
|
772
|
+
for row in rows:
|
|
773
|
+
if row.kind == "tool_call" and row.call_id:
|
|
774
|
+
counts[row.call_id] = counts.get(row.call_id, 0) + 1
|
|
775
|
+
return counts
|
|
776
|
+
|
|
777
|
+
|
|
778
|
+
def _reasoning_headings(detail, payload, block_key: str):
|
|
779
|
+
"""The additive ``headings`` array for one reasoning block, or ``None``.
|
|
780
|
+
|
|
781
|
+
#463 S2 §2.3/§2.5. Read-time decomposition of the retained payload's
|
|
782
|
+
``summary`` entries into the individual authored headings, each addressed by
|
|
783
|
+
``<block_key>#<zero-based ordinal>``. The stored projection is NOT consulted
|
|
784
|
+
and NOT modified: it feeds ``_row_is_reasoning_title``, which is a
|
|
785
|
+
segmentation-boundary input.
|
|
786
|
+
|
|
787
|
+
Headings come from ``payload["summary"]`` ONLY. ``payload["content"]`` is the
|
|
788
|
+
body, which stays disclosure content and is never decomposed.
|
|
789
|
+
|
|
790
|
+
All-or-nothing. When the payload is absent, unreadable or malformed, this
|
|
791
|
+
returns ``None`` and the caller omits the field entirely, so the client falls
|
|
792
|
+
back to today's ``title``/``summary`` rendering. Decomposition never fails the
|
|
793
|
+
request and never partially populates.
|
|
794
|
+
|
|
795
|
+
#463 S4 — the summary parse itself moved to
|
|
796
|
+
``_lib_codex_landmarks.reasoning_heading_texts``, so the reader route here
|
|
797
|
+
and the outline's landmark derivation decompose by ONE rule. This function
|
|
798
|
+
keeps the two things the outline does not want: the stored-detail gate, and
|
|
799
|
+
the ``<block_key>#<ordinal>`` identity, which the outline mints from its own
|
|
800
|
+
block keys rather than from the reader's.
|
|
801
|
+
"""
|
|
802
|
+
if not isinstance(detail, dict) or not isinstance(detail.get("reasoning"), dict):
|
|
803
|
+
return None
|
|
804
|
+
headings = landmarks.reasoning_heading_texts(payload)
|
|
805
|
+
if not headings:
|
|
806
|
+
return None
|
|
807
|
+
return [{"key": f"{block_key}#{ordinal}", "text": text}
|
|
808
|
+
for ordinal, text in enumerate(headings)]
|
|
809
|
+
|
|
810
|
+
|
|
811
|
+
# ── the conversation-level session index (#463 S3, spec section 3.2) ─────────
|
|
812
|
+
#
|
|
813
|
+
# Page-local adaptation cannot decide whether a session label is unique across
|
|
814
|
+
# the conversation or whether an opener exists, because later pages adapt
|
|
815
|
+
# independently and live-tail can append. So the server publishes a bounded
|
|
816
|
+
# conversation-scoped index and the client never computes either fact itself.
|
|
817
|
+
#
|
|
818
|
+
# 870 sessions across 223 conversations, roughly four per conversation, so this
|
|
819
|
+
# cap is generous. A conversation past it publishes what fits and marks itself
|
|
820
|
+
# truncated rather than publishing a partial map that looks complete.
|
|
821
|
+
_SESSION_INDEX_MAX = 64
|
|
822
|
+
|
|
823
|
+
|
|
824
|
+
def _stored_write_stdin_session(detail) -> str | None:
|
|
825
|
+
"""The session id a `write_stdin` call names, from its STORED arguments.
|
|
826
|
+
|
|
827
|
+
Phase A must not load event payloads, and it does not need to: `detail.args`
|
|
828
|
+
is in the narrow index, and it is the provider's own argument JSON.
|
|
829
|
+
"""
|
|
830
|
+
if not isinstance(detail, dict) or detail.get("name") != "write_stdin":
|
|
831
|
+
return None
|
|
832
|
+
args = detail.get("args")
|
|
833
|
+
if not isinstance(args, str) or not args:
|
|
834
|
+
return None
|
|
835
|
+
try:
|
|
836
|
+
parsed = json.loads(args)
|
|
837
|
+
except (json.JSONDecodeError, TypeError, ValueError):
|
|
838
|
+
return None
|
|
839
|
+
raw = parsed.get("session_id") if isinstance(parsed, dict) else None
|
|
840
|
+
if isinstance(raw, bool) or not isinstance(raw, (str, int)):
|
|
841
|
+
return None
|
|
842
|
+
return str(raw)
|
|
843
|
+
|
|
844
|
+
|
|
845
|
+
def _stored_session_announcement(detail) -> str | None:
|
|
846
|
+
"""The session id a tool output ANNOUNCES, from its stored card.
|
|
847
|
+
|
|
848
|
+
`Process running with session ID <id>` is the evidence linking a shell
|
|
849
|
+
session to the call that opened it — 698 of 870 sessions, 80.2%. The line is
|
|
850
|
+
read through the same anchored preamble reader the card path uses rather than
|
|
851
|
+
by searching the text, so a user's own output cannot be mistaken for one.
|
|
852
|
+
|
|
853
|
+
Spec section 3.2 names `search_tool` as the column this comes from. It is
|
|
854
|
+
read from the stored card in `detail_json` instead, which carries the same
|
|
855
|
+
bytes at the head of its first part and IS in the narrow index —
|
|
856
|
+
`_load_conversation_index_rows` excludes `search_tool` along with the other
|
|
857
|
+
two bulk columns, and adding it back would undo S1's Phase A saving.
|
|
858
|
+
"""
|
|
859
|
+
card = detail.get("card") if isinstance(detail, dict) else None
|
|
860
|
+
if not isinstance(card, dict) or card.get("type") != "terminal_output":
|
|
861
|
+
return None
|
|
862
|
+
parts = card.get("parts")
|
|
863
|
+
if not isinstance(parts, list) or not parts:
|
|
864
|
+
return None
|
|
865
|
+
head = parts[0]
|
|
866
|
+
text = head.get("text") if isinstance(head, dict) else None
|
|
867
|
+
if not isinstance(text, str):
|
|
868
|
+
return None
|
|
869
|
+
parsed = kern.parse_harness_preamble(text)
|
|
870
|
+
return parsed[0]["session_announcement"] if parsed is not None else None
|
|
871
|
+
|
|
872
|
+
|
|
873
|
+
def _build_session_index(rows) -> tuple[dict, dict[str, str]]:
|
|
874
|
+
"""``(envelope, ordinal_by_provider_session_id)`` over the WHOLE conversation.
|
|
875
|
+
|
|
876
|
+
Ordinals are assigned in first-appearance order over the conversation's
|
|
877
|
+
physical row order, so they are stable across pages and live-tail appends and
|
|
878
|
+
no client-side uniqueness decision is made from a partial window.
|
|
879
|
+
|
|
880
|
+
The envelope's `sessions` map is keyed by the ordinal in decimal, which is
|
|
881
|
+
exactly what a `session_ref` card's `ref` carries, so the client's lookup is
|
|
882
|
+
direct. Nothing in the envelope is derived from the provider's own session
|
|
883
|
+
id — that token is removed rather than scrubbed (spec section 4.3).
|
|
884
|
+
"""
|
|
885
|
+
owners: dict[str, list] = {}
|
|
886
|
+
for row in rows:
|
|
887
|
+
if row.kind == "tool_call" and row.call_id:
|
|
888
|
+
owners.setdefault(row.call_id, []).append(row)
|
|
889
|
+
ordinals: dict[str, int] = {}
|
|
890
|
+
openers: dict[str, str | None] = {}
|
|
891
|
+
truncated = False
|
|
892
|
+
for row in rows:
|
|
893
|
+
session = None
|
|
894
|
+
opener_row = None
|
|
895
|
+
detail = _parse_detail(row.detail_json)
|
|
896
|
+
if row.kind == "tool_call":
|
|
897
|
+
session = _stored_write_stdin_session(detail)
|
|
898
|
+
elif row.kind == "tool_output":
|
|
899
|
+
session = _stored_session_announcement(detail)
|
|
900
|
+
if session is not None:
|
|
901
|
+
# The opener is the CALL that owns the announcing output, not the
|
|
902
|
+
# output row: a uniquely-owned output folds into its call and has
|
|
903
|
+
# no block of its own, so its key would name nothing on the page.
|
|
904
|
+
owning = owners.get(row.call_id or "", [])
|
|
905
|
+
opener_row = owning[0] if len(owning) == 1 else row
|
|
906
|
+
if session is None:
|
|
907
|
+
continue
|
|
908
|
+
if session not in ordinals:
|
|
909
|
+
if len(ordinals) >= _SESSION_INDEX_MAX:
|
|
910
|
+
truncated = True
|
|
911
|
+
continue
|
|
912
|
+
ordinals[session] = len(ordinals) + 1
|
|
913
|
+
openers[session] = None
|
|
914
|
+
if opener_row is not None and openers.get(session) is None:
|
|
915
|
+
openers[session] = _block_key_for_row(opener_row)
|
|
916
|
+
envelope = {
|
|
917
|
+
"sessions": {
|
|
918
|
+
str(ordinal): {"ordinal": ordinal,
|
|
919
|
+
"opener_block_key": openers.get(session)}
|
|
920
|
+
for session, ordinal in ordinals.items()
|
|
921
|
+
},
|
|
922
|
+
"truncated": truncated,
|
|
923
|
+
}
|
|
924
|
+
return envelope, {session: str(ordinal) for session, ordinal in ordinals.items()}
|
|
925
|
+
|
|
926
|
+
|
|
927
|
+
def _apply_session_ordinals(card, ordinals: dict[str, str]) -> None:
|
|
928
|
+
"""Replace every SHELL session reference with its conversation-local ordinal.
|
|
929
|
+
|
|
930
|
+
Fails closed: a reference the index does not know becomes ``None`` rather
|
|
931
|
+
than falling back to the provider's id. `cell` scope is left alone — a cell
|
|
932
|
+
id is a small per-conversation sandbox ordinal that identifies nothing
|
|
933
|
+
outside the sandbox, and it is never presented as a shell session.
|
|
934
|
+
"""
|
|
935
|
+
if not isinstance(card, dict):
|
|
936
|
+
return
|
|
937
|
+
if card.get("type") == "session_ref" and card.get("scope") == "shell":
|
|
938
|
+
card["ref"] = ordinals.get(card.get("ref"))
|
|
939
|
+
return
|
|
940
|
+
if card.get("type") == "program":
|
|
941
|
+
for entry in card.get("invocations") or []:
|
|
942
|
+
if (isinstance(entry, dict) and entry.get("kind") == "session"
|
|
943
|
+
and entry.get("scope") == "shell"):
|
|
944
|
+
entry["ref"] = ordinals.get(entry.get("ref"))
|
|
945
|
+
|
|
946
|
+
|
|
332
947
|
def _item_blocks_with_rows(
|
|
333
948
|
item: dict, payloads: dict | None = None, *, preserve_marker_text: bool = False,
|
|
949
|
+
call_owner_count: dict | None = None, decompose_headings: bool = False,
|
|
950
|
+
session_ordinals: dict[str, str] | None = None,
|
|
334
951
|
) -> list[list]:
|
|
335
952
|
"""Assemble an item's blocks (the historical ``_build_item_blocks`` behaviour)
|
|
336
953
|
AND expose each block's underlying rows, so the detail renderer and the payload
|
|
@@ -340,8 +957,17 @@ def _item_blocks_with_rows(
|
|
|
340
957
|
exactly one tool_call, and that call was already seen (call precedes output).
|
|
341
958
|
Physical order within the item is preserved.
|
|
342
959
|
|
|
343
|
-
|
|
344
|
-
|
|
960
|
+
EVERY block backed by a physical row carries an opaque ``block_key`` (§3.4)
|
|
961
|
+
since #463 S2 §1; before it, only ``tool_call`` and a few event families did.
|
|
962
|
+
The far smaller payload-readable set is marked separately by
|
|
963
|
+
``payload_which``, because a stable anchor and a retained payload are
|
|
964
|
+
different properties (§1.1).
|
|
965
|
+
|
|
966
|
+
``decompose_headings`` adds the additive ``detail.reasoning.headings`` array
|
|
967
|
+
(#463 S2 §2.5). It is OFF by default and off on the export path, because
|
|
968
|
+
``legacy_export`` loads only marker-bearing payloads and populating the field
|
|
969
|
+
there would force a whole-conversation payload read to fill something the
|
|
970
|
+
exporter never reads."""
|
|
345
971
|
rows = item["rows"]
|
|
346
972
|
payloads = payloads or {}
|
|
347
973
|
lifecycle_positions = {
|
|
@@ -349,10 +975,11 @@ def _item_blocks_with_rows(
|
|
|
349
975
|
}
|
|
350
976
|
row_order = {(row.source_path, row.line_offset): index
|
|
351
977
|
for index, row in enumerate(rows)}
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
978
|
+
# Turn-scoped when the caller supplies it (#463 S1 Phase C); otherwise
|
|
979
|
+
# computed over this item's own rows, which is the same thing for an
|
|
980
|
+
# unsegmented item.
|
|
981
|
+
if call_owner_count is None:
|
|
982
|
+
call_owner_count = _turn_scoped_call_owner_count(rows)
|
|
356
983
|
entries: list[list] = []
|
|
357
984
|
tool_entry_by_call: dict[str, int] = {}
|
|
358
985
|
for r in rows:
|
|
@@ -369,11 +996,28 @@ def _item_blocks_with_rows(
|
|
|
369
996
|
text = kern._join_content_texts(payload.get("content"))
|
|
370
997
|
elif retained[0] == "event_msg":
|
|
371
998
|
text = kern._stringify(payload.get("message"))
|
|
999
|
+
if r.kind == "assistant":
|
|
1000
|
+
# #463 S3 section 5.5. Read-time detection over the row's own stored
|
|
1001
|
+
# text, so it reaches every historical marker with no payload load.
|
|
1002
|
+
# It is written to `external_call` and never to `markers`, which
|
|
1003
|
+
# selects export payload hydration.
|
|
1004
|
+
external = kern._external_call_from_text(text)
|
|
1005
|
+
# Fail closed on the span: it is published as offsets into the very
|
|
1006
|
+
# string served as `block["text"]`, and a span that does not resolve
|
|
1007
|
+
# would make the client hide the wrong run of prose. `text` is the
|
|
1008
|
+
# same object the block below carries, including the
|
|
1009
|
+
# `preserve_marker_text` replacement, so the check is against what is
|
|
1010
|
+
# actually served rather than against what was read.
|
|
1011
|
+
if external is not None and kern.external_call_span_resolves(
|
|
1012
|
+
text, external):
|
|
1013
|
+
detail = dict(detail) if isinstance(detail, dict) else {}
|
|
1014
|
+
detail["external_call"] = external
|
|
372
1015
|
if r.kind == "tool_call" and isinstance(payload, dict):
|
|
373
1016
|
card = kern.decode_tool_call_card(payload)
|
|
374
1017
|
if card is None:
|
|
375
1018
|
card = kern.decode_secondary_tool_call_card(payload)
|
|
376
1019
|
if card is not None:
|
|
1020
|
+
_apply_session_ordinals(card, session_ordinals or {})
|
|
377
1021
|
detail = dict(detail) if isinstance(detail, dict) else {}
|
|
378
1022
|
detail["card"] = card
|
|
379
1023
|
output_card = (kern.decode_tool_output_card(payload)
|
|
@@ -411,17 +1055,26 @@ def _item_blocks_with_rows(
|
|
|
411
1055
|
owner_card["result"] = result
|
|
412
1056
|
owner[2] = r
|
|
413
1057
|
continue
|
|
1058
|
+
block_key = _block_key_for_row(r)
|
|
1059
|
+
if decompose_headings and r.kind == "reasoning":
|
|
1060
|
+
headings = _reasoning_headings(detail, payload, block_key)
|
|
1061
|
+
if headings is not None:
|
|
1062
|
+
detail = dict(detail)
|
|
1063
|
+
detail["reasoning"] = dict(detail["reasoning"])
|
|
1064
|
+
detail["reasoning"]["headings"] = headings
|
|
414
1065
|
block = {
|
|
415
1066
|
"kind": r.kind, "text": text, "detail": detail,
|
|
416
1067
|
"call_id": r.call_id, "timestamp_utc": r.timestamp_utc,
|
|
1068
|
+
# #463 S2 §1 — EVERY row-backed block carries the anchor, not only
|
|
1069
|
+
# tool_call. `payload_which` below still marks the far smaller set
|
|
1070
|
+
# that is payload-readable, because those are different properties.
|
|
1071
|
+
"block_key": block_key,
|
|
417
1072
|
}
|
|
418
1073
|
if (r.kind == "event" and r.event_type in {
|
|
419
1074
|
"web_search_end", "mcp_tool_call_end", "task_started", "task_complete"}
|
|
420
1075
|
or isinstance(stored_detail, dict) and stored_detail.get("markers")):
|
|
421
|
-
block["block_key"] = _block_key_for_row(r)
|
|
422
1076
|
block["payload_which"] = "event"
|
|
423
1077
|
if r.kind == "tool_call":
|
|
424
|
-
block["block_key"] = _block_key_for_row(r)
|
|
425
1078
|
if r.call_id and call_owner_count.get(r.call_id, 0) == 1:
|
|
426
1079
|
tool_entry_by_call[r.call_id] = len(entries)
|
|
427
1080
|
entries.append([block, r, None])
|
|
@@ -445,8 +1098,7 @@ def _item_blocks_with_rows(
|
|
|
445
1098
|
if not (event_row.kind == "event" and isinstance(event_card, dict)
|
|
446
1099
|
and event_card.get("source") == "patch_apply_end"):
|
|
447
1100
|
continue
|
|
448
|
-
event_key =
|
|
449
|
-
event_block["block_key"] = event_key
|
|
1101
|
+
event_key = event_block["block_key"]
|
|
450
1102
|
event_block["payload_which"] = "event"
|
|
451
1103
|
same_id = [
|
|
452
1104
|
index for index, owner_count in patch_calls
|
|
@@ -514,7 +1166,6 @@ def _item_blocks_with_rows(
|
|
|
514
1166
|
== "web_search_call")
|
|
515
1167
|
]
|
|
516
1168
|
if len(candidates) != 1:
|
|
517
|
-
event_block["block_key"] = _block_key_for_row(event_row)
|
|
518
1169
|
event_block["payload_which"] = "event"
|
|
519
1170
|
continue
|
|
520
1171
|
owner_index = candidates[0]
|
|
@@ -539,14 +1190,13 @@ def _item_blocks_with_rows(
|
|
|
539
1190
|
owner_card["call_status"] = owner_payload["status"]
|
|
540
1191
|
owner_detail["card"] = owner_card
|
|
541
1192
|
if not isinstance(owner_card, dict):
|
|
542
|
-
event_block["block_key"] = _block_key_for_row(event_row)
|
|
543
1193
|
event_block["payload_which"] = "event"
|
|
544
1194
|
continue
|
|
545
1195
|
owner_card["completion"] = {
|
|
546
1196
|
key: value for key, value in event_card.items()
|
|
547
1197
|
if key not in {"schema_version", "type", "source"}
|
|
548
1198
|
}
|
|
549
|
-
owner_card["completion"]["event_block_key"] =
|
|
1199
|
+
owner_card["completion"]["event_block_key"] = event_block["block_key"]
|
|
550
1200
|
matched_secondary.add(owner_index)
|
|
551
1201
|
suppress_secondary.add(event_index)
|
|
552
1202
|
if suppress_secondary:
|
|
@@ -557,13 +1207,18 @@ def _item_blocks_with_rows(
|
|
|
557
1207
|
|
|
558
1208
|
def _build_item_blocks(
|
|
559
1209
|
item: dict, payloads: dict | None = None, *, preserve_marker_text: bool = False,
|
|
1210
|
+
call_owner_count: dict | None = None, decompose_headings: bool = False,
|
|
1211
|
+
session_ordinals: dict[str, str] | None = None,
|
|
560
1212
|
) -> list[dict]:
|
|
561
1213
|
"""Assemble an item's blocks, folding each ``tool_output`` into its
|
|
562
1214
|
``tool_call`` block via ``call_id`` when that call_id has exactly one owner
|
|
563
1215
|
(§5.2). Physical order within the item is preserved. Thin projection of
|
|
564
1216
|
``_item_blocks_with_rows`` — the single source of truth for the folding rule."""
|
|
565
1217
|
return [entry[0] for entry in _item_blocks_with_rows(
|
|
566
|
-
item, payloads, preserve_marker_text=preserve_marker_text
|
|
1218
|
+
item, payloads, preserve_marker_text=preserve_marker_text,
|
|
1219
|
+
call_owner_count=call_owner_count,
|
|
1220
|
+
decompose_headings=decompose_headings,
|
|
1221
|
+
session_ordinals=session_ordinals)]
|
|
567
1222
|
|
|
568
1223
|
|
|
569
1224
|
def _item_lifecycle(item: dict) -> dict | None:
|
|
@@ -675,9 +1330,32 @@ def _attribute_costs(conn: sqlite3.Connection, conversation_key: str, effective_
|
|
|
675
1330
|
return turn_cost, turn_tokens, unattr_cost, unattr_tokens, total, conv_tokens
|
|
676
1331
|
|
|
677
1332
|
|
|
1333
|
+
def _conversation_totals(
|
|
1334
|
+
conn: sqlite3.Connection, conversation_key: str, effective_speed: str,
|
|
1335
|
+
) -> tuple[float, dict]:
|
|
1336
|
+
"""Lean priced and token totals over one conversation's accounting rows.
|
|
1337
|
+
|
|
1338
|
+
Unlike ``_attribute_costs``, this does not reconstruct the event-to-turn map:
|
|
1339
|
+
callers that need only conversation totals (outline, browse, child summaries)
|
|
1340
|
+
can sum the compact accounting rows directly. The row order and pricing
|
|
1341
|
+
primitive stay identical to the detail envelope's whole-conversation pass.
|
|
1342
|
+
"""
|
|
1343
|
+
total = 0.0
|
|
1344
|
+
tokens = _zero_tokens()
|
|
1345
|
+
for model, inp, cin, out, rout in conn.execute(
|
|
1346
|
+
"SELECT model, input_tokens, cached_input_tokens, output_tokens, "
|
|
1347
|
+
"reasoning_output_tokens FROM codex_session_entries WHERE conversation_key = ? "
|
|
1348
|
+
"ORDER BY source_path, line_offset",
|
|
1349
|
+
(conversation_key,),
|
|
1350
|
+
):
|
|
1351
|
+
total += _calculate_codex_entry_cost(
|
|
1352
|
+
model or "", inp or 0, cin or 0, out or 0, rout or 0, speed=effective_speed)
|
|
1353
|
+
_add_tokens(tokens, inp, out, cin, rout)
|
|
1354
|
+
return total, _tokens_union(tokens)
|
|
1355
|
+
|
|
1356
|
+
|
|
678
1357
|
def _conversation_total_cost(conn: sqlite3.Connection, conversation_key: str, effective_speed: str) -> float:
|
|
679
|
-
"""Lean priced total
|
|
680
|
-
child summaries) — same primitive as ``_attribute_costs`` (§5.4)."""
|
|
1358
|
+
"""Lean priced total for browse rows and child summaries (§5.4)."""
|
|
681
1359
|
total = 0.0
|
|
682
1360
|
for model, inp, cin, out, rout in conn.execute(
|
|
683
1361
|
"SELECT model, input_tokens, cached_input_tokens, output_tokens, "
|
|
@@ -760,9 +1438,19 @@ def _short_native(native: str | None) -> str:
|
|
|
760
1438
|
|
|
761
1439
|
def _display_chain(fields: dict) -> str:
|
|
762
1440
|
"""Read-time display fallback (§4.3): stored title → project_label → short
|
|
763
|
-
native-thread-id prefix.
|
|
764
|
-
|
|
765
|
-
|
|
1441
|
+
native-thread-id prefix.
|
|
1442
|
+
|
|
1443
|
+
#463 S4 §5 — the stored title is cleaned of recognized harness markup here.
|
|
1444
|
+
This is ONE of three read paths that need it, not the universal chokepoint
|
|
1445
|
+
the first draft assumed: the outline turn label is built independently from
|
|
1446
|
+
anchor-row text and the `kind=title` search path reads rollup titles
|
|
1447
|
+
directly, so both clean through the same helper rather than through this
|
|
1448
|
+
call. A construct that strips to nothing falls through the chain below on
|
|
1449
|
+
its own, which is what makes `strip` expressible at read time at all.
|
|
1450
|
+
"""
|
|
1451
|
+
return (clean_codex_title(fields.get("title"))
|
|
1452
|
+
or fields.get("project_label")
|
|
1453
|
+
or _short_native(fields.get("native_thread_id")) or "")
|
|
766
1454
|
|
|
767
1455
|
|
|
768
1456
|
def _conversation_display_title(conn: sqlite3.Connection, conversation_key: str, rows: list | None = None) -> str:
|
|
@@ -1005,42 +1693,395 @@ def codex_conversation_source_paths(
|
|
|
1005
1693
|
# ── detail assembly (§5.2 / §5.4 / §5.6) ──────────────────────────────────────
|
|
1006
1694
|
|
|
1007
1695
|
|
|
1008
|
-
def
|
|
1009
|
-
|
|
1010
|
-
|
|
1011
|
-
|
|
1012
|
-
|
|
1013
|
-
|
|
1014
|
-
|
|
1015
|
-
|
|
1016
|
-
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
|
|
1021
|
-
|
|
1022
|
-
|
|
1023
|
-
|
|
1024
|
-
|
|
1025
|
-
|
|
1026
|
-
|
|
1027
|
-
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
|
|
1031
|
-
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
|
|
1696
|
+
def _stale_codex_page(total: int) -> dict:
|
|
1697
|
+
"""The page a cursor resolving to nothing returns (#463 S1 / F4).
|
|
1698
|
+
|
|
1699
|
+
Mirrors the Claude kernel's ``_stale_empty_page`` contract: a stale or
|
|
1700
|
+
deleted cursor yields an EMPTY page, never a silent re-serve of the head or
|
|
1701
|
+
the tail. The old Codex kernel left the bound at the edge of the list, which
|
|
1702
|
+
is the second path by which ``before`` returned the wrong window.
|
|
1703
|
+
"""
|
|
1704
|
+
return {"total": total, "returned": 0, "before": None, "after": None,
|
|
1705
|
+
"has_before": False, "has_after": False}
|
|
1706
|
+
|
|
1707
|
+
|
|
1708
|
+
def _paginate_items(items: list[dict], *, after, before, tail: bool, limit: int,
|
|
1709
|
+
block_budget: int | None = None,
|
|
1710
|
+
byte_budget: int | None = None):
|
|
1711
|
+
"""Cut one page from the assembled item list (#463 S1 / F4).
|
|
1712
|
+
|
|
1713
|
+
``tail`` is the flag the HTTP layer parses out of ``?tail=1`` — which page
|
|
1714
|
+
of ``limit`` to cut, never how many items to return.
|
|
1715
|
+
|
|
1716
|
+
The four cursor branches mirror the Claude kernel's structure
|
|
1717
|
+
(``_lib_conversation_query.py`` :2372-2391) so the two can be read against
|
|
1718
|
+
each other, which is what stops them diverging again. TWO deliberate
|
|
1719
|
+
Codex-only differences, both pinned by ``tests/test_codex_pagination.py``:
|
|
1720
|
+
|
|
1721
|
+
* ``limit == 0`` means UNBOUNDED, and the export path depends on that
|
|
1722
|
+
sentinel (``get_codex_conversation_export`` passes ``limit=0``). Claude's
|
|
1723
|
+
default branch computes ``end = min(limit, N)``, which for zero yields an
|
|
1724
|
+
empty page — ported literally, every Codex export would contain no
|
|
1725
|
+
conversation items. The unbounded case is therefore explicit, ahead of
|
|
1726
|
+
the four branches.
|
|
1727
|
+
* Cursor resolution covers BOTH primary item keys and ``member_item_keys``
|
|
1728
|
+
aliases, so a cursor naming an item folded by a later contract version
|
|
1729
|
+
still resolves. Claude's ``_idx`` checks only its primary anchor id, and
|
|
1730
|
+
taking its resolver along with its branch arithmetic would make every
|
|
1731
|
+
folded-item cursor stale. The branch arithmetic is what is ported; the
|
|
1732
|
+
resolver is not.
|
|
1733
|
+
|
|
1734
|
+
The prior implementation computed ``lo``/``hi`` and then sliced, so a
|
|
1735
|
+
``before`` request returned ``items[0:hi][:limit]`` — the conversation's
|
|
1736
|
+
OPENING items — with ``has_before`` False.
|
|
1737
|
+
|
|
1738
|
+
TWO per-page budgets apply alongside ``limit``, and the first of them to be
|
|
1739
|
+
reached closes the page (spec section 2). They bound different costs and
|
|
1740
|
+
neither substitutes for the other:
|
|
1741
|
+
|
|
1742
|
+
* ``block_budget`` bounds DOM construction cost, which the F2 profile
|
|
1743
|
+
shows dominates mounting.
|
|
1744
|
+
* ``byte_budget`` bounds transfer and parse cost, which is a byte cost a
|
|
1745
|
+
block count does not express, because a Codex block is far heavier than
|
|
1746
|
+
a Claude block.
|
|
1747
|
+
|
|
1748
|
+
Neither is optional in production. The profiled response was
|
|
1749
|
+
``total: 78, returned: 78, has_after: false`` — 13.3 MB served in one page,
|
|
1750
|
+
because 78 is fewer than the requested 500 — so a change that capped items
|
|
1751
|
+
alone would not have bounded that conversation at all. And after
|
|
1752
|
+
segmentation that same conversation serves 1,713 blocks, BELOW
|
|
1753
|
+
``PAGE_BLOCK_BUDGET``, so the block bound never fires on it either and the
|
|
1754
|
+
response is still 13.24 MB: the byte budget is what actually closes it.
|
|
1755
|
+
|
|
1756
|
+
Both are deliberately NOT applied to the unbounded ``limit == 0`` export
|
|
1757
|
+
case, which must stay whole. The trim comes off whichever end is not
|
|
1758
|
+
anchored to the cursor, so a reverse page still ends where the caller asked
|
|
1759
|
+
it to, and a page never shrinks below one item however large that item is.
|
|
1760
|
+
"""
|
|
1761
|
+
index_by_key: dict[str, int] = {}
|
|
1762
|
+
for index, item in enumerate(items):
|
|
1763
|
+
index_by_key.setdefault(item["item_key"], index)
|
|
1764
|
+
for index, item in enumerate(items):
|
|
1765
|
+
for alias in item.get("member_item_keys", []):
|
|
1766
|
+
index_by_key.setdefault(alias, index)
|
|
1767
|
+
|
|
1768
|
+
n = len(items)
|
|
1769
|
+
|
|
1770
|
+
if not limit:
|
|
1771
|
+
start, end = 0, n
|
|
1772
|
+
elif tail:
|
|
1773
|
+
end = n
|
|
1774
|
+
start = max(0, n - limit)
|
|
1775
|
+
elif before is not None:
|
|
1776
|
+
b = index_by_key.get(before)
|
|
1777
|
+
if b is None:
|
|
1778
|
+
return [], _stale_codex_page(n)
|
|
1779
|
+
end = b
|
|
1780
|
+
start = max(0, end - limit)
|
|
1781
|
+
elif after is not None:
|
|
1782
|
+
a = index_by_key.get(after)
|
|
1783
|
+
if a is None:
|
|
1784
|
+
return [], _stale_codex_page(n)
|
|
1785
|
+
start = a + 1
|
|
1786
|
+
end = min(start + limit, n)
|
|
1787
|
+
else:
|
|
1788
|
+
start = 0
|
|
1789
|
+
end = min(limit, n)
|
|
1790
|
+
|
|
1791
|
+
if limit and end > start and (block_budget or byte_budget):
|
|
1792
|
+
blocks = sum(item.get("block_count", 0) for item in items[start:end])
|
|
1793
|
+
source = sum(item.get("source_bytes", 0) for item in items[start:end])
|
|
1794
|
+
|
|
1795
|
+
def _over() -> bool:
|
|
1796
|
+
return ((bool(block_budget) and blocks > block_budget)
|
|
1797
|
+
or (bool(byte_budget) and source > byte_budget))
|
|
1798
|
+
|
|
1799
|
+
anchored_at_end = tail or before is not None
|
|
1800
|
+
while end - start > 1 and _over():
|
|
1801
|
+
drop = items[start] if anchored_at_end else items[end - 1]
|
|
1802
|
+
blocks -= drop.get("block_count", 0)
|
|
1803
|
+
source -= drop.get("source_bytes", 0)
|
|
1804
|
+
if anchored_at_end:
|
|
1805
|
+
start += 1
|
|
1806
|
+
else:
|
|
1807
|
+
end -= 1
|
|
1808
|
+
|
|
1809
|
+
window = items[start:end]
|
|
1810
|
+
has_before = start > 0
|
|
1811
|
+
has_after = end < n
|
|
1035
1812
|
page = {
|
|
1036
|
-
"total":
|
|
1037
|
-
"before":
|
|
1038
|
-
"after":
|
|
1813
|
+
"total": n, "returned": len(window),
|
|
1814
|
+
"before": window[0]["item_key"] if (window and has_before) else None,
|
|
1815
|
+
"after": window[-1]["item_key"] if (window and has_after) else None,
|
|
1039
1816
|
"has_before": has_before, "has_after": has_after,
|
|
1040
1817
|
}
|
|
1041
1818
|
return window, page
|
|
1042
1819
|
|
|
1043
1820
|
|
|
1821
|
+
def _row_source_bytes(row, detail_bytes: dict) -> int:
|
|
1822
|
+
"""One row's source size: its content length plus its stored detail's BYTES.
|
|
1823
|
+
|
|
1824
|
+
Source bytes deliberately overstate wire bytes. The server clips block
|
|
1825
|
+
payloads, so the measured ratio is roughly six to eight times — a
|
|
1826
|
+
conversation holding 14.6 MB of ``content_len`` serves 1.8 MB, and one
|
|
1827
|
+
holding 84.6 MB serves 13.3 MB. Any byte threshold must therefore be stated
|
|
1828
|
+
in source bytes and derived from a wire target through that ratio; S1
|
|
1829
|
+
computes the figure and exposes it, and gates only on block count.
|
|
1830
|
+
"""
|
|
1831
|
+
return (row.content_len or 0) + detail_bytes.get(
|
|
1832
|
+
(row.source_path, row.line_offset), 0)
|
|
1833
|
+
|
|
1834
|
+
|
|
1835
|
+
_UNSET = object()
|
|
1836
|
+
|
|
1837
|
+
|
|
1838
|
+
def _row_is_reasoning_title(row, detail=_UNSET) -> bool:
|
|
1839
|
+
"""True when this row's stored reasoning projection produces a ``title``.
|
|
1840
|
+
|
|
1841
|
+
The first of the two semantic boundaries. Read from ``detail_json``, because
|
|
1842
|
+
``search_thinking`` stores ``summary + "\\n" + body`` for a response item and
|
|
1843
|
+
so cannot tell ``summary="**T**", body="x"`` (a title) from
|
|
1844
|
+
``summary="**T**\\nx", body=""`` (not one).
|
|
1845
|
+
|
|
1846
|
+
``detail`` lets a caller that has already parsed the row's ``detail_json``
|
|
1847
|
+
pass it in. Phase A runs over every row of the conversation, so parsing the
|
|
1848
|
+
same JSON twice per row is worth avoiding.
|
|
1849
|
+
"""
|
|
1850
|
+
if row.kind != "reasoning":
|
|
1851
|
+
return False
|
|
1852
|
+
if detail is _UNSET:
|
|
1853
|
+
detail = _parse_detail(row.detail_json)
|
|
1854
|
+
reasoning = detail.get("reasoning") if isinstance(detail, dict) else None
|
|
1855
|
+
return isinstance(reasoning, dict) and bool(reasoning.get("title"))
|
|
1856
|
+
|
|
1857
|
+
|
|
1858
|
+
def _fold_groups_for_item(item: dict, call_owner_count: dict,
|
|
1859
|
+
detail_bytes: dict) -> list:
|
|
1860
|
+
"""Derive one item's atomic fold groups (#463 S1, spec section 1).
|
|
1861
|
+
|
|
1862
|
+
A group is a row together with every later row in the item that could fold
|
|
1863
|
+
into it. Membership is decided from ``call_id`` and the STORED card in
|
|
1864
|
+
``detail_json``, deliberately without reading the retained event payloads —
|
|
1865
|
+
Phase A must not touch them. That makes the grouping a conservative SUPERSET
|
|
1866
|
+
of what the block builder will actually fold: a completion event whose card
|
|
1867
|
+
turns out not to fold stays grouped with its call anyway. A superset is the
|
|
1868
|
+
safe direction, because the only thing a group guarantees is that no boundary
|
|
1869
|
+
is drawn between a call and something that might fold into it.
|
|
1870
|
+
|
|
1871
|
+
``_item_blocks_with_rows`` performs THREE folds, not one, and grouping covers
|
|
1872
|
+
all three:
|
|
1873
|
+
|
|
1874
|
+
* the id-matched fold — a ``tool_output`` or completion event whose
|
|
1875
|
+
``call_id`` is owned by exactly one ``tool_call`` in the turn;
|
|
1876
|
+
* the bracketed native patch completion — a patch completion event may
|
|
1877
|
+
carry an INNER call id distinct from the outer custom-tool call, which
|
|
1878
|
+
the block builder folds by positional bracketing
|
|
1879
|
+
(``call_pos < event_pos < output_pos``) rather than by id. This is the
|
|
1880
|
+
common shape: 3,441 of the production corpus's 4,690 patch completion
|
|
1881
|
+
events carry a call id no ``tool_call`` in their turn owns. Such an event
|
|
1882
|
+
joins the most recent patch call whose output has not yet arrived;
|
|
1883
|
+
* the ``web_search_completion`` narrowing — that path filters its
|
|
1884
|
+
candidates by ``detail.name == "web_search_call"`` BEFORE requiring a
|
|
1885
|
+
unique candidate, and it bounds nothing about how many calls share the
|
|
1886
|
+
id, so it folds at ANY owner count. The registration below therefore
|
|
1887
|
+
imposes no owner ceiling either: it registers the web-search arm at
|
|
1888
|
+
``owners >= 1``. Naming a fixed count leaves the pair ungrouped at every
|
|
1889
|
+
other count — an ``== 1`` gate never covers a two-owner id, and an
|
|
1890
|
+
``== 2`` gate never covers a call id owned by three calls of which one
|
|
1891
|
+
is the web search.
|
|
1892
|
+
|
|
1893
|
+
A boundary that split any of the three would make the page-local builder emit
|
|
1894
|
+
a standalone event card where the whole-turn builder emits a folded
|
|
1895
|
+
``completion`` — the structural divergence spec section 1 forbids.
|
|
1896
|
+
|
|
1897
|
+
Block counts are overestimated for the same reason. A ``tool_output`` whose
|
|
1898
|
+
call id is uniquely owned provably folds and contributes nothing; every other
|
|
1899
|
+
grouped row is counted as its own block even though it may fold. Over-
|
|
1900
|
+
counting shrinks segments slightly, which keeps the budget an upper bound.
|
|
1901
|
+
|
|
1902
|
+
Each group also carries ``first_pos``/``last_pos``, its physical row range
|
|
1903
|
+
inside the item, which ``plan_segments`` uses to keep a segment contiguous.
|
|
1904
|
+
|
|
1905
|
+
Lifecycle rows are excluded entirely: they never produce a block, and they
|
|
1906
|
+
are carried on segment 0 instead, where ``_item_lifecycle`` renders them from
|
|
1907
|
+
the narrow row alone.
|
|
1908
|
+
"""
|
|
1909
|
+
lifecycle_positions = {
|
|
1910
|
+
(row.source_path, row.line_offset) for row in item.get("lifecycle_rows", [])
|
|
1911
|
+
}
|
|
1912
|
+
groups: list[dict] = []
|
|
1913
|
+
open_group_by_call: dict[str, dict] = {}
|
|
1914
|
+
open_patch_groups: list[dict] = []
|
|
1915
|
+
previous_kind = None
|
|
1916
|
+
position = 0
|
|
1917
|
+
for row in item["rows"]:
|
|
1918
|
+
if (row.source_path, row.line_offset) in lifecycle_positions:
|
|
1919
|
+
continue
|
|
1920
|
+
detail = _parse_detail(row.detail_json)
|
|
1921
|
+
card = detail.get("card") if isinstance(detail, dict) else None
|
|
1922
|
+
if not isinstance(card, dict):
|
|
1923
|
+
card = None
|
|
1924
|
+
owner = (open_group_by_call.get(row.call_id)
|
|
1925
|
+
if row.call_id and row.kind in {"tool_output", "event"} else None)
|
|
1926
|
+
if (owner is None and row.kind == "event" and card is not None
|
|
1927
|
+
and card.get("source") == "patch_apply_end" and open_patch_groups):
|
|
1928
|
+
# The most RECENT still-open patch call is the one the block
|
|
1929
|
+
# builder's bracket resolves to in the single-patch case; taking an
|
|
1930
|
+
# older one would leave the true owner ungrouped.
|
|
1931
|
+
owner = open_patch_groups.pop()
|
|
1932
|
+
if owner is not None:
|
|
1933
|
+
owner["rows"].append(row)
|
|
1934
|
+
owner["last_pos"] = position
|
|
1935
|
+
owner["source_bytes"] += _row_source_bytes(row, detail_bytes)
|
|
1936
|
+
folds = (row.kind == "tool_output"
|
|
1937
|
+
and call_owner_count.get(row.call_id, 0) == 1)
|
|
1938
|
+
if not folds:
|
|
1939
|
+
owner["block_count"] += 1
|
|
1940
|
+
# Identity, not equality. ``in`` and ``list.remove`` compare with
|
|
1941
|
+
# ``==``, so they would match — and delete — the FIRST group whose
|
|
1942
|
+
# dict merely compares equal to this one. That is correct today only
|
|
1943
|
+
# because two distinct groups can never hold equal contents, which is
|
|
1944
|
+
# an accident of the data rather than a property of this loop.
|
|
1945
|
+
if row.kind == "tool_output" and any(g is owner for g in open_patch_groups):
|
|
1946
|
+
open_patch_groups[:] = [g for g in open_patch_groups if g is not owner]
|
|
1947
|
+
position += 1
|
|
1948
|
+
continue
|
|
1949
|
+
group = {
|
|
1950
|
+
"rows": [row],
|
|
1951
|
+
"block_count": 1,
|
|
1952
|
+
"source_bytes": _row_source_bytes(row, detail_bytes),
|
|
1953
|
+
"is_title_boundary": _row_is_reasoning_title(row, detail),
|
|
1954
|
+
"is_tool_transition": (row.kind == "tool_call"
|
|
1955
|
+
and previous_kind in {"assistant", "reasoning"}),
|
|
1956
|
+
"first_pos": position,
|
|
1957
|
+
"last_pos": position,
|
|
1958
|
+
}
|
|
1959
|
+
groups.append(group)
|
|
1960
|
+
if row.kind == "tool_call" and row.call_id:
|
|
1961
|
+
owners = call_owner_count.get(row.call_id, 0)
|
|
1962
|
+
name = detail.get("name") if isinstance(detail, dict) else None
|
|
1963
|
+
# The web-search arm must not impose an owner ceiling the block
|
|
1964
|
+
# builder does not. `_pair_web_search_completions` filters the
|
|
1965
|
+
# candidates by `detail.name == "web_search_call"` and then requires a
|
|
1966
|
+
# unique survivor, with no bound on how many calls share the id — so a
|
|
1967
|
+
# call id owned by three calls of which exactly one is a web search
|
|
1968
|
+
# folds there while `owners == 2` refused to group it here, and a
|
|
1969
|
+
# segment boundary could fall between the call and its completion.
|
|
1970
|
+
# `owners >= 1` matches the builder and stays a conservative superset:
|
|
1971
|
+
# grouping only ever keeps rows together that the builder folds.
|
|
1972
|
+
if owners == 1 or (owners >= 1 and name == "web_search_call"):
|
|
1973
|
+
open_group_by_call.setdefault(row.call_id, group)
|
|
1974
|
+
if row.kind == "tool_call" and card is not None and card.get("type") == "patch":
|
|
1975
|
+
open_patch_groups.append(group)
|
|
1976
|
+
previous_kind = row.kind
|
|
1977
|
+
position += 1
|
|
1978
|
+
return [
|
|
1979
|
+
segkern.FoldGroup(
|
|
1980
|
+
rows=group["rows"], block_count=group["block_count"],
|
|
1981
|
+
source_bytes=group["source_bytes"],
|
|
1982
|
+
is_title_boundary=group["is_title_boundary"],
|
|
1983
|
+
is_tool_transition=group["is_tool_transition"],
|
|
1984
|
+
first_pos=group["first_pos"], last_pos=group["last_pos"])
|
|
1985
|
+
for group in groups
|
|
1986
|
+
]
|
|
1987
|
+
|
|
1988
|
+
|
|
1989
|
+
def _build_segment_index(
|
|
1990
|
+
conversation_key: str, items: list[dict], detail_bytes: dict, *,
|
|
1991
|
+
segmented: bool, block_budget: int | None = None,
|
|
1992
|
+
fold_groups: bool = False,
|
|
1993
|
+
) -> list[dict]:
|
|
1994
|
+
"""Phase A's output: an ordered index of segments, with no block content.
|
|
1995
|
+
|
|
1996
|
+
Each entry carries its keys, turn membership, ordinal, the physical rows it
|
|
1997
|
+
covers, its sizes, and the two structural facts Phase C requires and cannot
|
|
1998
|
+
recompute correctly on its own — the fold-group membership that makes a
|
|
1999
|
+
boundary legal, and the TURN-scoped ``call_owner_count``.
|
|
2000
|
+
|
|
2001
|
+
``segmented=False`` gives each item exactly one segment holding all of its
|
|
2002
|
+
groups, which is what the export path uses so its item grouping stays
|
|
2003
|
+
byte-identical.
|
|
2004
|
+
|
|
2005
|
+
``block_budget`` resolves to ``segkern.SEGMENT_BLOCK_BUDGET`` at CALL time
|
|
2006
|
+
when omitted, never as a default argument value: a default argument binds
|
|
2007
|
+
once at import, so a test that lowered the budget would silently keep the
|
|
2008
|
+
imported figure and pass vacuously.
|
|
2009
|
+
|
|
2010
|
+
``fold_groups`` publishes the fold-group membership as ``_fold_groups``. Only
|
|
2011
|
+
the outline reads it (#463 S4 — a ``tool_error`` landmark anchors on the call
|
|
2012
|
+
a failure folds into); the detail route S1 bounded and the search position
|
|
2013
|
+
map do not, and building it for them costs a tuple per row per request for a
|
|
2014
|
+
value nobody reads.
|
|
2015
|
+
"""
|
|
2016
|
+
index: list[dict] = []
|
|
2017
|
+
for item_index, item in enumerate(items):
|
|
2018
|
+
turn_key = _item_key_for_item(conversation_key, item)
|
|
2019
|
+
call_owner_count = _turn_scoped_call_owner_count(item["rows"])
|
|
2020
|
+
groups = _fold_groups_for_item(item, call_owner_count, detail_bytes)
|
|
2021
|
+
if not groups:
|
|
2022
|
+
segments = [segkern.Segment(
|
|
2023
|
+
ordinal=0, groups=[], block_count=0, source_bytes=0,
|
|
2024
|
+
anchor_row=item["anchor_row"])]
|
|
2025
|
+
elif segmented and item["klass"] == "response":
|
|
2026
|
+
segments = segkern.plan_segments(groups, block_budget=block_budget)
|
|
2027
|
+
else:
|
|
2028
|
+
segments = [segkern.Segment(
|
|
2029
|
+
ordinal=0, groups=groups,
|
|
2030
|
+
block_count=sum(group.block_count for group in groups),
|
|
2031
|
+
source_bytes=sum(group.source_bytes for group in groups),
|
|
2032
|
+
anchor_row=groups[0].rows[0])]
|
|
2033
|
+
for segment in segments:
|
|
2034
|
+
head = segment.ordinal == 0
|
|
2035
|
+
anchor = item["anchor_row"] if head else segment.anchor_row
|
|
2036
|
+
# PHYSICAL order, not group-flatten order. A folded row is appended
|
|
2037
|
+
# to an earlier group, so flattening the groups would move it next to
|
|
2038
|
+
# its call — and the patch-completion fold decides ambiguous cases by
|
|
2039
|
+
# positional bracketing (call < event < its output) over the item's
|
|
2040
|
+
# row order, which that reordering silently breaks.
|
|
2041
|
+
member = {(row.source_path, row.line_offset)
|
|
2042
|
+
for group in segment.groups for row in group.rows}
|
|
2043
|
+
segment_rows = [row for row in item["rows"]
|
|
2044
|
+
if (row.source_path, row.line_offset) in member]
|
|
2045
|
+
entry = {
|
|
2046
|
+
"item_key": turn_key if head else codex_item_key(
|
|
2047
|
+
conversation_key, klass="segment", turn_id=item["turn_id"],
|
|
2048
|
+
source_path=segment.anchor_row.source_path,
|
|
2049
|
+
line_offset=segment.anchor_row.line_offset,
|
|
2050
|
+
content_digest=segment.anchor_row.content_digest),
|
|
2051
|
+
"member_item_keys": (
|
|
2052
|
+
_member_item_keys(conversation_key, item) if head else []),
|
|
2053
|
+
"turn_item_key": turn_key,
|
|
2054
|
+
"segment_ordinal": segment.ordinal,
|
|
2055
|
+
"kind": _item_kind(item),
|
|
2056
|
+
"timestamp_utc": anchor.timestamp_utc,
|
|
2057
|
+
"model": anchor.model,
|
|
2058
|
+
"block_count": segment.block_count,
|
|
2059
|
+
"source_bytes": segment.source_bytes,
|
|
2060
|
+
# Phase C inputs — never serialized.
|
|
2061
|
+
"_item_index": item_index,
|
|
2062
|
+
"_klass": item["klass"],
|
|
2063
|
+
"_turn_id": item["turn_id"],
|
|
2064
|
+
"_anchor_row": anchor,
|
|
2065
|
+
"_rows": segment_rows,
|
|
2066
|
+
# #463 S4 — the fold-group membership, as physical positions.
|
|
2067
|
+
# `_fold_groups_for_item` computes it payload-free and this
|
|
2068
|
+
# index discarded it, so nothing downstream could say WHICH
|
|
2069
|
+
# `tool_call` a failing `tool_output` belongs to — only that the
|
|
2070
|
+
# segment contained one. A `tool_error` landmark anchors on the
|
|
2071
|
+
# call, so the membership has to survive Phase A.
|
|
2072
|
+
"_fold_groups": [
|
|
2073
|
+
[(row.source_path, row.line_offset) for row in group.rows]
|
|
2074
|
+
for group in segment.groups
|
|
2075
|
+
] if fold_groups else (),
|
|
2076
|
+
"_call_owner_count": call_owner_count,
|
|
2077
|
+
"_meta": _item_meta(item) if head else None,
|
|
2078
|
+
"_lifecycle": _item_lifecycle(item) if head else None,
|
|
2079
|
+
"_lifecycle_rows": item.get("lifecycle_rows", []) if head else [],
|
|
2080
|
+
}
|
|
2081
|
+
index.append(entry)
|
|
2082
|
+
return index
|
|
2083
|
+
|
|
2084
|
+
|
|
1044
2085
|
def get_codex_conversation(
|
|
1045
2086
|
conn: sqlite3.Connection,
|
|
1046
2087
|
conversation_key: str,
|
|
@@ -1055,35 +2096,51 @@ def get_codex_conversation(
|
|
|
1055
2096
|
"""Detail envelope (§5.6): status ``ok`` | ``normalization_pending`` |
|
|
1056
2097
|
``not_found``. ``ok`` carries canonical items (mirror-paired, tool-folded),
|
|
1057
2098
|
per-turn cost with an explicit unattributed bucket, threading, and a page
|
|
1058
|
-
over ``item_key``.
|
|
2099
|
+
over ``item_key``.
|
|
2100
|
+
|
|
2101
|
+
Assembly runs in three phases (#463 S1, finding F3). Before this change every
|
|
2102
|
+
step from row loading through block building processed the WHOLE
|
|
2103
|
+
conversation, and ``_paginate_items`` ran last, so pagination reduced
|
|
2104
|
+
serialization and transfer but bounded no work.
|
|
2105
|
+
|
|
2106
|
+
* **Phase A** reads every row narrowly — everything except ``text``, and no
|
|
2107
|
+
event payloads at all — pairs mirrors, groups canonical items, derives
|
|
2108
|
+
fold groups and segment boundaries, and emits a segment index.
|
|
2109
|
+
* **Phase B** paginates that index. It is arithmetic over a narrow list.
|
|
2110
|
+
* **Phase C** hydrates ONLY the requested page: the wide ``text`` read and
|
|
2111
|
+
the events-table payload scan are scoped to the page's physical
|
|
2112
|
+
positions, and blocks are built per segment.
|
|
2113
|
+
|
|
2114
|
+
What stays proportional to the conversation is the narrow index pass, because
|
|
2115
|
+
segment boundaries and the ``has_before``/``has_after`` flags are global
|
|
2116
|
+
facts. What becomes proportional to the page is everything expensive.
|
|
2117
|
+
|
|
2118
|
+
Cost attribution deliberately stays whole-conversation: ``_attribute_costs``
|
|
2119
|
+
reconciles per-turn costs against an unattributed bucket and the envelope
|
|
2120
|
+
reports conversation-level totals, so scoping it to a page would change the
|
|
2121
|
+
reported total. It reads ``codex_session_entries``, not the large message
|
|
2122
|
+
table.
|
|
2123
|
+
|
|
2124
|
+
``page.total`` is now a count of SEGMENTS rather than of items.
|
|
2125
|
+
"""
|
|
1059
2126
|
if not codex_normalization_authoritative(conn):
|
|
1060
2127
|
return {"status": "normalization_pending", "conversation_key": conversation_key,
|
|
1061
2128
|
"items": [], "children": []}
|
|
1062
|
-
|
|
2129
|
+
# ── Phase A: the narrow index pass ───────────────────────────────────────
|
|
2130
|
+
rows, detail_bytes = _load_conversation_index_rows(conn, conversation_key)
|
|
1063
2131
|
if not rows:
|
|
1064
2132
|
return {"status": "not_found", "conversation_key": conversation_key}
|
|
1065
2133
|
kept, _suppressed = kern.pair_mirrors(rows)
|
|
1066
2134
|
items = kern.canonical_items(
|
|
1067
2135
|
kept, fold_patch_completions=not legacy_export)
|
|
1068
|
-
# Detail/API callers receive the exact card display projection from retained
|
|
1069
|
-
# provider payloads. Export deliberately renders the byte-frozen legacy text
|
|
1070
|
-
# while retaining the same additive card metadata.
|
|
1071
|
-
payloads = _load_row_payloads(conn, conversation_key)
|
|
1072
|
-
if legacy_export:
|
|
1073
|
-
marker_positions = {
|
|
1074
|
-
(row.source_path, row.line_offset)
|
|
1075
|
-
for row in rows
|
|
1076
|
-
if isinstance(_parse_detail(row.detail_json), dict)
|
|
1077
|
-
and bool(_parse_detail(row.detail_json).get("markers"))
|
|
1078
|
-
}
|
|
1079
|
-
payloads = {
|
|
1080
|
-
position: retained for position, retained in payloads.items()
|
|
1081
|
-
if position in marker_positions
|
|
1082
|
-
}
|
|
1083
2136
|
turn_cost, turn_tokens, unattr_cost, unattr_tokens, total, conv_tokens = _attribute_costs(
|
|
1084
2137
|
conn, conversation_key, effective_speed)
|
|
1085
2138
|
# Carrier item per turn: prefer the response item, else the first item of the
|
|
1086
2139
|
# turn — so every priced turn's cost lands on exactly one item (§5.4 reconcile).
|
|
2140
|
+
# Segmentation must not move the carrier, so the selection stays keyed on the
|
|
2141
|
+
# ITEM index and the cost lands on that item's segment 0. Every other segment
|
|
2142
|
+
# carries null rather than zero, because a zero is indistinguishable from a
|
|
2143
|
+
# genuinely free turn.
|
|
1087
2144
|
carriers: dict[str, int] = {}
|
|
1088
2145
|
for idx, it in enumerate(items):
|
|
1089
2146
|
if it["klass"] == "response" and it["turn_id"] is not None and it["turn_id"] not in carriers:
|
|
@@ -1097,40 +2154,109 @@ def get_codex_conversation(
|
|
|
1097
2154
|
if turn not in carriers:
|
|
1098
2155
|
leftover_cost += cost
|
|
1099
2156
|
unattributed_cost = unattr_cost + leftover_cost
|
|
2157
|
+
# Segmentation is disabled under legacy_export, so item grouping there stays
|
|
2158
|
+
# byte-identical to what the export golden already pins.
|
|
2159
|
+
index = _build_segment_index(
|
|
2160
|
+
conversation_key, items, detail_bytes, segmented=not legacy_export)
|
|
2161
|
+
# #463 S3 section 3.2. Built here, in Phase A, from the narrow index that is
|
|
2162
|
+
# already loaded for the whole conversation: it needs no extra payload read
|
|
2163
|
+
# and no extra column, and it must be whole-conversation because ordinals and
|
|
2164
|
+
# opener presence are global facts a page cannot decide.
|
|
2165
|
+
session_index, session_ordinals = _build_session_index(rows)
|
|
2166
|
+
|
|
2167
|
+
# ── Phase B: paginate the index ──────────────────────────────────────────
|
|
2168
|
+
page_index, page = _paginate_items(
|
|
2169
|
+
index, after=after, before=before, tail=tail, limit=limit,
|
|
2170
|
+
block_budget=None if legacy_export else segkern.PAGE_BLOCK_BUDGET,
|
|
2171
|
+
byte_budget=None if legacy_export else segkern.PAGE_SOURCE_BYTE_BUDGET)
|
|
2172
|
+
|
|
2173
|
+
# ── Phase C: hydrate only the page ───────────────────────────────────────
|
|
2174
|
+
page_positions = {
|
|
2175
|
+
(row.source_path, row.line_offset)
|
|
2176
|
+
for entry in page_index for row in entry["_rows"]
|
|
2177
|
+
}
|
|
2178
|
+
hydrated = _load_rows_at_positions(conn, conversation_key, page_positions)
|
|
2179
|
+
if len(hydrated) != len(page_positions):
|
|
2180
|
+
# A miss here is a bug, not a degradation. The narrow index pass and this
|
|
2181
|
+
# wide read select from the SAME table on the same conversation key, so a
|
|
2182
|
+
# position that appears in one and not the other means the two reads
|
|
2183
|
+
# disagree. Falling back to the narrow row would silently render an empty
|
|
2184
|
+
# block, because the narrow row carries no ``text``.
|
|
2185
|
+
missing = sorted(page_positions - set(hydrated))[:5]
|
|
2186
|
+
raise RuntimeError(
|
|
2187
|
+
f"codex detail hydration missed {len(page_positions) - len(hydrated)} "
|
|
2188
|
+
f"of {len(page_positions)} page rows for {conversation_key}; "
|
|
2189
|
+
f"first missing positions: {missing}")
|
|
2190
|
+
# Detail/API callers receive the exact card display projection from retained
|
|
2191
|
+
# provider payloads. Export deliberately renders the byte-frozen legacy text
|
|
2192
|
+
# while retaining the same additive card metadata, so it scopes the payload
|
|
2193
|
+
# read to marker-bearing rows across the WHOLE conversation rather than to
|
|
2194
|
+
# the page — a different set from page_positions, hence a different name.
|
|
2195
|
+
if legacy_export:
|
|
2196
|
+
marker_positions = {
|
|
2197
|
+
(row.source_path, row.line_offset)
|
|
2198
|
+
for row in rows
|
|
2199
|
+
if isinstance(_parse_detail(row.detail_json), dict)
|
|
2200
|
+
and bool(_parse_detail(row.detail_json).get("markers"))
|
|
2201
|
+
}
|
|
2202
|
+
payloads = _load_row_payloads(conn, conversation_key, marker_positions)
|
|
2203
|
+
else:
|
|
2204
|
+
payloads = _load_row_payloads(conn, conversation_key, page_positions)
|
|
1100
2205
|
built: list[dict] = []
|
|
1101
|
-
for
|
|
1102
|
-
|
|
1103
|
-
|
|
1104
|
-
|
|
1105
|
-
|
|
2206
|
+
for entry in page_index:
|
|
2207
|
+
page_rows = [
|
|
2208
|
+
hydrated[(row.source_path, row.line_offset)]
|
|
2209
|
+
for row in entry["_rows"]
|
|
2210
|
+
]
|
|
2211
|
+
page_item = {
|
|
2212
|
+
"klass": entry["_klass"], "rows": page_rows,
|
|
2213
|
+
"turn_id": entry["_turn_id"], "anchor_row": entry["_anchor_row"],
|
|
2214
|
+
}
|
|
2215
|
+
item_index = entry["_item_index"]
|
|
2216
|
+
turn = entry["_turn_id"]
|
|
2217
|
+
cost = tokens = None
|
|
2218
|
+
if (entry["segment_ordinal"] == 0 and turn is not None
|
|
2219
|
+
and carriers.get(turn) == item_index and turn in turn_cost):
|
|
1106
2220
|
cost = turn_cost[turn]
|
|
1107
2221
|
tokens = _tokens_union(turn_tokens[turn])
|
|
1108
2222
|
item = {
|
|
1109
|
-
"item_key":
|
|
1110
|
-
"member_item_keys":
|
|
1111
|
-
"
|
|
1112
|
-
"
|
|
1113
|
-
"
|
|
2223
|
+
"item_key": entry["item_key"],
|
|
2224
|
+
"member_item_keys": entry["member_item_keys"],
|
|
2225
|
+
"turn_item_key": entry["turn_item_key"],
|
|
2226
|
+
"segment_ordinal": entry["segment_ordinal"],
|
|
2227
|
+
"kind": entry["kind"],
|
|
2228
|
+
"timestamp_utc": entry["timestamp_utc"],
|
|
2229
|
+
"model": entry["model"],
|
|
1114
2230
|
"blocks": _build_item_blocks(
|
|
1115
|
-
|
|
2231
|
+
page_item, payloads, preserve_marker_text=legacy_export,
|
|
2232
|
+
call_owner_count=entry["_call_owner_count"],
|
|
2233
|
+
# #463 S2 §2.5 — never under legacy_export: that path loads only
|
|
2234
|
+
# marker-bearing payloads, so populating `headings` there would
|
|
2235
|
+
# force a whole-conversation payload read for a field the
|
|
2236
|
+
# exporter never reads.
|
|
2237
|
+
decompose_headings=not legacy_export,
|
|
2238
|
+
session_ordinals=session_ordinals),
|
|
1116
2239
|
"cost_usd": cost,
|
|
1117
2240
|
"tokens": tokens,
|
|
1118
2241
|
}
|
|
1119
|
-
|
|
1120
|
-
|
|
1121
|
-
|
|
1122
|
-
|
|
1123
|
-
if lifecycle is not None:
|
|
1124
|
-
item["lifecycle"] = lifecycle
|
|
2242
|
+
if entry["_meta"] is not None:
|
|
2243
|
+
item.update(entry["_meta"])
|
|
2244
|
+
if entry["_lifecycle"] is not None:
|
|
2245
|
+
item["lifecycle"] = entry["_lifecycle"]
|
|
1125
2246
|
built.append(item)
|
|
1126
2247
|
_attach_spawn_child_links(conn, conversation_key, built)
|
|
1127
|
-
page_items
|
|
2248
|
+
page_items = built
|
|
1128
2249
|
return {
|
|
1129
2250
|
"status": "ok",
|
|
1130
2251
|
"conversation_key": conversation_key,
|
|
1131
|
-
|
|
2252
|
+
# NOT the Phase A rows: those carry no ``text``, and the live-recompute
|
|
2253
|
+
# fallback inside _rollup_fields derives the title from it. Passing None
|
|
2254
|
+
# keeps the stored fast path unchanged and lets the rare no-rollup case
|
|
2255
|
+
# do its own wide read rather than titling the conversation "".
|
|
2256
|
+
"title": _conversation_display_title(conn, conversation_key),
|
|
1132
2257
|
"items": page_items,
|
|
1133
2258
|
"page": page,
|
|
2259
|
+
"session_index": session_index,
|
|
1134
2260
|
"children": _children_of(conn, conversation_key, effective_speed),
|
|
1135
2261
|
"parent": _parent_of(conn, conversation_key),
|
|
1136
2262
|
"total_cost_usd": total,
|
|
@@ -1142,15 +2268,299 @@ def get_codex_conversation(
|
|
|
1142
2268
|
# ── outline assembly (§5.6) ───────────────────────────────────────────────────
|
|
1143
2269
|
|
|
1144
2270
|
|
|
1145
|
-
def
|
|
1146
|
-
|
|
1147
|
-
|
|
1148
|
-
|
|
1149
|
-
|
|
1150
|
-
|
|
1151
|
-
|
|
1152
|
-
|
|
1153
|
-
|
|
2271
|
+
def _tool_call_name(row) -> str | None:
|
|
2272
|
+
"""The tool a ``tool_call`` row invoked, from its STORED detail.
|
|
2273
|
+
|
|
2274
|
+
Ingest writes ``{"name": payload["name"] or <record type>, …}`` for every
|
|
2275
|
+
tool call, so this needs no payload and stays outside the scoped pass.
|
|
2276
|
+
"""
|
|
2277
|
+
detail = _parse_detail(row.detail_json)
|
|
2278
|
+
name = detail.get("name") if isinstance(detail, dict) else None
|
|
2279
|
+
return name if isinstance(name, str) and name else None
|
|
2280
|
+
|
|
2281
|
+
|
|
2282
|
+
def _conversation_duration_seconds(rows) -> int | None:
|
|
2283
|
+
"""Wall span of the conversation, as MIN to MAX over row timestamps (§4.2).
|
|
2284
|
+
|
|
2285
|
+
Never last minus first. §3.4 records that ``timestamp_utc`` is monotone
|
|
2286
|
+
within a turn only — item and segment emission is physical order, not
|
|
2287
|
+
timestamp order — and Task 1 found five decreases across turns in the
|
|
2288
|
+
corpus, on which the naive form returns a negative duration.
|
|
2289
|
+
|
|
2290
|
+
The outline's own caller cannot exhibit that today, because
|
|
2291
|
+
``_load_conversation_rows`` reads ``ORDER BY timestamp_utc, source_path,
|
|
2292
|
+
line_offset`` and the two forms therefore coincide on the rows it passes.
|
|
2293
|
+
The rule is stated for the next caller, which is much more likely to hand
|
|
2294
|
+
over an item-anchor list.
|
|
2295
|
+
"""
|
|
2296
|
+
stamps = [row.timestamp_utc for row in rows if row.timestamp_utc]
|
|
2297
|
+
if not stamps:
|
|
2298
|
+
return None
|
|
2299
|
+
first = _parse_outline_ts(min(stamps))
|
|
2300
|
+
last = _parse_outline_ts(max(stamps))
|
|
2301
|
+
if first is None or last is None:
|
|
2302
|
+
return None
|
|
2303
|
+
return int((last - first).total_seconds())
|
|
2304
|
+
|
|
2305
|
+
|
|
2306
|
+
def _outline_error_count(
|
|
2307
|
+
failed_calls: set, outcome_positions: set,
|
|
2308
|
+
derivation: landmarks.EventDerivation,
|
|
2309
|
+
) -> int | None:
|
|
2310
|
+
"""How many calls failed, or ``None`` when that cannot be answered (D3).
|
|
2311
|
+
|
|
2312
|
+
Three states, because 0 and null are different claims. A conversation with
|
|
2313
|
+
no outcome-bearing row at all reports 0: nothing failed because nothing ran,
|
|
2314
|
+
and that is determinable. A conversation whose outcome rows produced no
|
|
2315
|
+
verdict at all — every retained payload gone, unparseable, or of a shape no
|
|
2316
|
+
decoder recognises — reports null, because the stored projection answers
|
|
2317
|
+
nothing here (Task 1 measured stored ``is_error`` true for 0 of 63,150
|
|
2318
|
+
production ``tool_output`` rows) and 0 would assert an absence nobody proved.
|
|
2319
|
+
|
|
2320
|
+
A PARTIAL read reports what it found rather than declining. The alternative
|
|
2321
|
+
would suppress real failures the pass did see because of one unreadable
|
|
2322
|
+
neighbour, which is the worse error of the two.
|
|
2323
|
+
"""
|
|
2324
|
+
if not outcome_positions:
|
|
2325
|
+
return 0
|
|
2326
|
+
if not (outcome_positions & set(derivation.errors_by_position)):
|
|
2327
|
+
return None
|
|
2328
|
+
return len(failed_calls)
|
|
2329
|
+
|
|
2330
|
+
|
|
2331
|
+
# The Codex tool names that open the `plan` landmark family. Codex's decoded
|
|
2332
|
+
# plan card is named `update_plan`, and both existing CLIENT plan predicates
|
|
2333
|
+
# recognise only Claude's `ExitPlanMode` and `AskUserQuestion` — so publishing
|
|
2334
|
+
# raw Codex tool names into tier-1 `tools` would NOT have made the plan jump
|
|
2335
|
+
# work, it would have been a silent no-op (§3.2). The mapping is explicit here
|
|
2336
|
+
# and the outline target derivation reads landmark KINDS rather than inferring
|
|
2337
|
+
# from names.
|
|
2338
|
+
_S4_PLAN_TOOLS = frozenset({"update_plan"})
|
|
2339
|
+
|
|
2340
|
+
|
|
2341
|
+
def _landmark_label(row) -> str:
|
|
2342
|
+
"""What a landmark row says. Never a raw provider identifier (§8).
|
|
2343
|
+
|
|
2344
|
+
§3.6 enumerates exactly TWO sources for a landmark label — reasoning heading
|
|
2345
|
+
text, or a tool name — and rests the decision not to scrub these labels on
|
|
2346
|
+
that enumeration. So a row that is neither a named ``tool_call`` nor a typed
|
|
2347
|
+
``event`` falls back to its own KIND, which is normalizer vocabulary, and
|
|
2348
|
+
never to the row's stored text.
|
|
2349
|
+
|
|
2350
|
+
That branch is reachable: a failing ``tool_output`` whose ``call_id`` is
|
|
2351
|
+
owned by two or more ``tool_call`` rows in its turn does not fold, becomes
|
|
2352
|
+
its own group head, and enters ``failed_calls`` directly. Its ``text`` column
|
|
2353
|
+
is the harness preamble that ``decode_tool_output_card(for_storage=False)``
|
|
2354
|
+
exists to remove, and ``test_s3_no_raw_session_id_reaches_any_served_route``
|
|
2355
|
+
documents that preamble as carrying the provider ``session_id``.
|
|
2356
|
+
"""
|
|
2357
|
+
if row.kind == "tool_call":
|
|
2358
|
+
name = _tool_call_name(row)
|
|
2359
|
+
if name is not None:
|
|
2360
|
+
return name
|
|
2361
|
+
elif row.kind == "event" and row.event_type:
|
|
2362
|
+
return row.event_type
|
|
2363
|
+
return row.kind or ""
|
|
2364
|
+
|
|
2365
|
+
|
|
2366
|
+
def _clean_outline_label(text: str) -> str:
|
|
2367
|
+
"""An outline turn label, cleaned, and never cleaned away to nothing (§5.1).
|
|
2368
|
+
|
|
2369
|
+
Unlike ``_display_chain``, which §5.3 leans on as "already a fallback chain"
|
|
2370
|
+
when it justifies the ``strip`` disposition, this path has no chain: it
|
|
2371
|
+
cleans the anchor row's first non-blank line and publishes the result. Two
|
|
2372
|
+
allowlisted grammars can consume the whole string — ``<recommended_plugins>``
|
|
2373
|
+
(6 of the census's 438 titles) and ``<command-name>`` with no sibling tag —
|
|
2374
|
+
and the client's ``cleanQualifiedTitle(turn.label) ?? turn.label`` passes an
|
|
2375
|
+
empty string straight through, so the reader would get a row with no text at
|
|
2376
|
+
all. The uncleaned line is the pre-S4 label, which is legible.
|
|
2377
|
+
"""
|
|
2378
|
+
cleaned = clean_codex_title(text)
|
|
2379
|
+
return cleaned if cleaned.strip() else text
|
|
2380
|
+
|
|
2381
|
+
|
|
2382
|
+
def _build_landmarks(
|
|
2383
|
+
index: list[dict], derivation: landmarks.EventDerivation,
|
|
2384
|
+
failed_calls: set,
|
|
2385
|
+
) -> list[dict]:
|
|
2386
|
+
"""Tier 2 — the landmarks a jump can reach (§3.2).
|
|
2387
|
+
|
|
2388
|
+
Three kinds and deliberately NOT one entry per tool call: a 523-call turn
|
|
2389
|
+
would contribute 523 rows, which is noise rather than navigation.
|
|
2390
|
+
|
|
2391
|
+
Emission is PHYSICAL order — the segment index in order, and each segment's
|
|
2392
|
+
rows in order — because §3.4 records that ``timestamp_utc`` is monotone
|
|
2393
|
+
within a turn only and no consumer sorts by it.
|
|
2394
|
+
|
|
2395
|
+
``landmark_key`` is always COMPOUND — ``<block_key>#<discriminator>`` — and
|
|
2396
|
+
unique across every kind. A reasoning heading discriminates by its zero-based
|
|
2397
|
+
ordinal, which is the identity the reader route already mints for the same
|
|
2398
|
+
heading, because one block yields several headings and ``block_key`` alone
|
|
2399
|
+
would collide. Every other kind discriminates by the kind itself.
|
|
2400
|
+
|
|
2401
|
+
That is what lets one ``tool_call`` block carry BOTH a ``tool_error`` and a
|
|
2402
|
+
``plan`` landmark. It has to: §3.2 gives the plan kind one entry per plan
|
|
2403
|
+
call, and a failed plan call is one, so filing it only as the error made the
|
|
2404
|
+
jump cluster's plan family report zero — asserting no plan activity in a
|
|
2405
|
+
conversation that has some, which is the claim the spec's own "0 is a claim,
|
|
2406
|
+
hiding is not" rule forbids. The error is emitted first, because it is the
|
|
2407
|
+
more urgent of the two claims about the same call.
|
|
2408
|
+
|
|
2409
|
+
A block carrying ``detail.external_call`` produces no landmark of any kind
|
|
2410
|
+
(§3.2). That holds by construction rather than by a filter here: the marker
|
|
2411
|
+
is published on ``assistant`` blocks only, and no kind below comes from an
|
|
2412
|
+
assistant row. ``test_external_call_block_produces_no_landmark`` pins it.
|
|
2413
|
+
"""
|
|
2414
|
+
out: list[dict] = []
|
|
2415
|
+
for entry in index:
|
|
2416
|
+
for row in entry["_rows"]:
|
|
2417
|
+
position = (row.source_path, row.line_offset)
|
|
2418
|
+
block_key = _block_key_for_row(row)
|
|
2419
|
+
common = {
|
|
2420
|
+
"block_key": block_key,
|
|
2421
|
+
"item_key": entry["item_key"],
|
|
2422
|
+
"parent_item_key": entry["turn_item_key"],
|
|
2423
|
+
"timestamp_utc": row.timestamp_utc,
|
|
2424
|
+
}
|
|
2425
|
+
if row.kind == "reasoning":
|
|
2426
|
+
for ordinal, text in enumerate(
|
|
2427
|
+
derivation.headings_by_position.get(position, ())):
|
|
2428
|
+
out.append({"landmark_key": f"{block_key}#{ordinal}",
|
|
2429
|
+
"kind": "reasoning", "label": text, **common})
|
|
2430
|
+
continue
|
|
2431
|
+
if position in failed_calls:
|
|
2432
|
+
out.append({"landmark_key": f"{block_key}#tool_error",
|
|
2433
|
+
"kind": "tool_error",
|
|
2434
|
+
"label": _landmark_label(row), **common})
|
|
2435
|
+
if (row.kind == "tool_call"
|
|
2436
|
+
and _tool_call_name(row) in _S4_PLAN_TOOLS):
|
|
2437
|
+
out.append({"landmark_key": f"{block_key}#plan", "kind": "plan",
|
|
2438
|
+
"label": _landmark_label(row), **common})
|
|
2439
|
+
return out
|
|
2440
|
+
|
|
2441
|
+
|
|
2442
|
+
# The literal ingest writes into `codex_conversation_file_touches.tool`, kept so
|
|
2443
|
+
# the outline wire field and the file-search projection mean the same thing.
|
|
2444
|
+
# Every touch S4 derives comes from a `patch_apply_end`, which is the completion
|
|
2445
|
+
# of an `apply_patch` call.
|
|
2446
|
+
_PATCH_TOUCH_TOOL = "apply_patch"
|
|
2447
|
+
|
|
2448
|
+
|
|
2449
|
+
def _conversation_files(
|
|
2450
|
+
segment_index: list[dict], derivation: landmarks.EventDerivation,
|
|
2451
|
+
) -> list[dict]:
|
|
2452
|
+
"""The whole-conversation file list, DERIVED read-time (§1.2, §4.3).
|
|
2453
|
+
|
|
2454
|
+
The stored ``codex_conversation_file_touches`` table is the source for
|
|
2455
|
+
cross-conversation ``kind=files`` search after #489 repaired dict-shaped
|
|
2456
|
+
ingest and backfilled retained history. It is deliberately not the outline
|
|
2457
|
+
source: this payload pass has the richer evidence the outline contract needs.
|
|
2458
|
+
|
|
2459
|
+
Deriving it instead buys three things the table could not have supplied: a
|
|
2460
|
+
real segment anchor per touch, so a file row jumps to its change rather than
|
|
2461
|
+
to the top of a turn; the true per-file count; and first-touch DOCUMENT
|
|
2462
|
+
order, which is what ``OutlineFile`` has always promised while the SQL
|
|
2463
|
+
ordered alphabetically by path.
|
|
2464
|
+
|
|
2465
|
+
``added``/``removed`` are summed over the touches, and go ``None`` as soon as
|
|
2466
|
+
ONE touch of that file cannot be counted. Summing only the countable touches
|
|
2467
|
+
would publish a number for a file that changed more — a file edited once with
|
|
2468
|
+
a real diff and then moved by a count-free ``update`` would report the first
|
|
2469
|
+
figure — and nothing in ``touches[]`` marks such a total as partial. §4.5 is
|
|
2470
|
+
explicit that an undeterminable count is null and the badge renders nothing
|
|
2471
|
+
rather than an understated number; that rule has to reach the aggregate, not
|
|
2472
|
+
only the individual touch. The per-touch counts themselves come from the
|
|
2473
|
+
UNBOUNDED raw ``changes`` entry (§4.5) — see ``landmarks.patch_file_touches``.
|
|
2474
|
+
"""
|
|
2475
|
+
files: dict[str, dict] = {}
|
|
2476
|
+
undetermined: dict[str, set[str]] = {}
|
|
2477
|
+
for entry in segment_index:
|
|
2478
|
+
for row in entry["_rows"]:
|
|
2479
|
+
position = (row.source_path, row.line_offset)
|
|
2480
|
+
for touch in derivation.patch_files_by_position.get(position, ()):
|
|
2481
|
+
record = files.get(touch["path"])
|
|
2482
|
+
if record is None:
|
|
2483
|
+
record = files[touch["path"]] = {
|
|
2484
|
+
"file_path": touch["path"], "tool": _PATCH_TOUCH_TOOL,
|
|
2485
|
+
"count": 0, "touches": [],
|
|
2486
|
+
"added": None, "removed": None}
|
|
2487
|
+
record["count"] += 1
|
|
2488
|
+
record["touches"].append({
|
|
2489
|
+
"item_key": entry["item_key"],
|
|
2490
|
+
"timestamp_utc": row.timestamp_utc,
|
|
2491
|
+
# The raw change KIND — `add`/`delete`/`update` from the dict
|
|
2492
|
+
# shape, `modified` from the list one — never the tool name.
|
|
2493
|
+
"op": touch["op"],
|
|
2494
|
+
})
|
|
2495
|
+
for field in ("added", "removed"):
|
|
2496
|
+
if touch[field] is None:
|
|
2497
|
+
undetermined.setdefault(touch["path"], set()).add(field)
|
|
2498
|
+
else:
|
|
2499
|
+
record[field] = (record[field] or 0) + touch[field]
|
|
2500
|
+
for path, fields in undetermined.items():
|
|
2501
|
+
for field in fields:
|
|
2502
|
+
files[path][field] = None
|
|
2503
|
+
return list(files.values())
|
|
2504
|
+
|
|
2505
|
+
|
|
2506
|
+
# Connections whose read snapshot THIS module opened, by identity. A
|
|
2507
|
+
# ``sqlite3.Connection`` supports neither attribute assignment nor a weak
|
|
2508
|
+
# reference, so ownership cannot be recorded on the object; ``id`` is unique
|
|
2509
|
+
# among live objects and the connection is alive for the whole ``with`` body, so
|
|
2510
|
+
# the token cannot be confused with another connection's while it is registered.
|
|
2511
|
+
_OWNED_READ_SNAPSHOTS: set[int] = set()
|
|
2512
|
+
|
|
2513
|
+
|
|
2514
|
+
@contextlib.contextmanager
|
|
2515
|
+
def _read_snapshot(conn: sqlite3.Connection):
|
|
2516
|
+
"""One consistent read snapshot across a multi-query envelope (#463 S4 §4.1).
|
|
2517
|
+
|
|
2518
|
+
The outline route uses one connection but opened no explicit read
|
|
2519
|
+
transaction, so its several queries each took their own snapshot. Concurrent
|
|
2520
|
+
APPEND is benign there — extra raw event rows have no normalized mapping yet
|
|
2521
|
+
— but a concurrent delete or truncation between the wide message read and the
|
|
2522
|
+
payload read can expose a message row whose payload is already gone, and the
|
|
2523
|
+
derivation would then report an absence that never existed.
|
|
2524
|
+
|
|
2525
|
+
A deferred ``BEGIN`` takes the snapshot on the first read and holds it for
|
|
2526
|
+
every later one. It is released with ``rollback``, which is the honest end of
|
|
2527
|
+
a transaction that wrote nothing.
|
|
2528
|
+
|
|
2529
|
+
**A transaction this module did not open is refused, not inherited.**
|
|
2530
|
+
``conn.in_transaction`` is true for an outer WRITE transaction exactly as it
|
|
2531
|
+
is for an outer read snapshot, and Python's ``sqlite3`` exposes no
|
|
2532
|
+
``txn_state``, so the two cannot be told apart here. Only one of them is safe
|
|
2533
|
+
to borrow: inside a write, the envelope would read that writer's uncommitted
|
|
2534
|
+
and possibly half-applied state — a message row whose events are already
|
|
2535
|
+
deleted — with no snapshot of its own and no way to notice. Treating both
|
|
2536
|
+
alike is silent; refusing is not. A caller that wants several envelopes on
|
|
2537
|
+
one snapshot opens it through this same helper, which nests without issuing
|
|
2538
|
+
the second ``BEGIN`` SQLite would refuse.
|
|
2539
|
+
|
|
2540
|
+
The caller sweep behind that decision, pinned by
|
|
2541
|
+
``test_every_outline_caller_arrives_outside_a_transaction``: the three call
|
|
2542
|
+
paths into ``get_codex_conversation_outline`` are
|
|
2543
|
+
``_lib_conversation_dispatch.neutral_outline`` (the dashboard route, on a
|
|
2544
|
+
connection ``open_conversations_db`` returns fresh per request and closes
|
|
2545
|
+
after), ``bin/build-codex-reader-fixtures.py``, and the tests. None holds a
|
|
2546
|
+
transaction at the call.
|
|
2547
|
+
"""
|
|
2548
|
+
token = id(conn)
|
|
2549
|
+
if token in _OWNED_READ_SNAPSHOTS:
|
|
2550
|
+
yield
|
|
2551
|
+
return
|
|
2552
|
+
if conn.in_transaction:
|
|
2553
|
+
raise RuntimeError(
|
|
2554
|
+
"this envelope needs its own read snapshot, and the connection is "
|
|
2555
|
+
"already inside a transaction it did not open; wrap the outer "
|
|
2556
|
+
"scope in _read_snapshot instead")
|
|
2557
|
+
conn.execute("BEGIN")
|
|
2558
|
+
_OWNED_READ_SNAPSHOTS.add(token)
|
|
2559
|
+
try:
|
|
2560
|
+
yield
|
|
2561
|
+
finally:
|
|
2562
|
+
_OWNED_READ_SNAPSHOTS.discard(token)
|
|
2563
|
+
conn.rollback()
|
|
1154
2564
|
|
|
1155
2565
|
|
|
1156
2566
|
def get_codex_conversation_outline(
|
|
@@ -1158,44 +2568,164 @@ def get_codex_conversation_outline(
|
|
|
1158
2568
|
) -> dict:
|
|
1159
2569
|
"""Outline envelope (§5.6): one ``turns[]`` entry per canonical item (label
|
|
1160
2570
|
via the shared first-non-blank-line helper), plus stats, file touches, and
|
|
1161
|
-
child summaries.
|
|
2571
|
+
child summaries.
|
|
2572
|
+
|
|
2573
|
+
The outline stays TURN-granular: ``turns[].item_key`` remains a turn key,
|
|
2574
|
+
which is still valid because it is segment 0's key.
|
|
2575
|
+
|
|
2576
|
+
Turn-granular keys alone are not sufficient, though (#463 S1). On a cold
|
|
2577
|
+
jump ``loadToTarget`` resolves the target through the outline and does
|
|
2578
|
+
nothing when the identifier is absent, and outline membership carried only
|
|
2579
|
+
folded-item aliases — no segment keys at all — so a deep link into any
|
|
2580
|
+
segment but a turn's first would silently fail to navigate. Each turn
|
|
2581
|
+
therefore also carries ``segment_item_keys``, where entry ``i`` is the key of
|
|
2582
|
+
segment ``i``.
|
|
2583
|
+
|
|
2584
|
+
That channel is deliberately DISTINCT from ``member_item_keys``. Putting
|
|
2585
|
+
segment keys there would make ``loadToTarget``'s "is it already loaded" test
|
|
2586
|
+
report true for a segment that has not been fetched, so the drain would never
|
|
2587
|
+
run and the jump would land nowhere. Membership for navigation and membership
|
|
2588
|
+
for "this item subsumes that key" are different relations.
|
|
2589
|
+
|
|
2590
|
+
#463 S4 — the route now makes TWO conversation reads under one snapshot: the
|
|
2591
|
+
wide message read below, and a scoped read-time pass over the retained event
|
|
2592
|
+
payloads (``_derive_outline_events``) whose position set comes from that
|
|
2593
|
+
first read. The pass is what gives the outline a failure verdict per call,
|
|
2594
|
+
the authored reasoning headings, and the per-file patch touches; §1.2 records
|
|
2595
|
+
why the stored ``codex_conversation_file_touches`` search projection is not
|
|
2596
|
+
the OUTLINE source, and Task 1 measured that the stored card carries 0 of the
|
|
2597
|
+
corpus's 896 tool failures, so read-time is not a preference here.
|
|
2598
|
+
"""
|
|
2599
|
+
with _read_snapshot(conn):
|
|
2600
|
+
return _outline_envelope(
|
|
2601
|
+
conn, conversation_key, effective_speed=effective_speed)
|
|
2602
|
+
|
|
2603
|
+
|
|
2604
|
+
def _outline_envelope(
|
|
2605
|
+
conn: sqlite3.Connection, conversation_key: str, *, effective_speed: str
|
|
2606
|
+
) -> dict:
|
|
1162
2607
|
if not codex_normalization_authoritative(conn):
|
|
1163
2608
|
return {"status": "normalization_pending", "conversation_key": conversation_key,
|
|
1164
2609
|
"turns": [], "files": [], "children": []}
|
|
2610
|
+
# Deliberately the WIDE read: an outline label is the first non-blank line of
|
|
2611
|
+
# its anchor row's display text, which the narrow index pass does not carry.
|
|
2612
|
+
# The outline is not the route F3 bounds, and it stays turn-granular.
|
|
1165
2613
|
rows = _load_conversation_rows(conn, conversation_key)
|
|
1166
2614
|
if not rows:
|
|
1167
2615
|
return {"status": "not_found", "conversation_key": conversation_key}
|
|
2616
|
+
detail_bytes = _detail_bytes_of(rows)
|
|
1168
2617
|
kept, _suppressed = kern.pair_mirrors(rows)
|
|
1169
2618
|
items = kern.canonical_items(kept)
|
|
2619
|
+
segment_keys: dict[int, list[str]] = {}
|
|
2620
|
+
fold_groups: list[list[tuple[str, int]]] = []
|
|
2621
|
+
segment_index = _build_segment_index(
|
|
2622
|
+
conversation_key, items, detail_bytes, segmented=True, fold_groups=True)
|
|
2623
|
+
for entry in segment_index:
|
|
2624
|
+
segment_keys.setdefault(entry["_item_index"], []).append(entry["item_key"])
|
|
2625
|
+
fold_groups.extend(entry["_fold_groups"])
|
|
2626
|
+
# `rows` here is the wide read directly above, and that is what makes the
|
|
2627
|
+
# payload pass SCOPED rather than a second whole-conversation decode (§4.1).
|
|
2628
|
+
derivation = _derive_outline_events(conn, conversation_key, rows)
|
|
2629
|
+
call_by_position = landmarks.fold_owner_by_position(fold_groups)
|
|
2630
|
+
outcome_positions = _outline_outcome_positions(rows)
|
|
2631
|
+
# A failing outcome row is charged to the `tool_call` it folds into, so the
|
|
2632
|
+
# same failure cannot be counted twice when a call and its output both carry
|
|
2633
|
+
# one, and so a turn's `tools` entry can say WHICH call failed.
|
|
2634
|
+
failed_calls = _outline_failing_calls(
|
|
2635
|
+
derivation, outcome_positions, call_by_position)
|
|
1170
2636
|
turns: list[dict] = []
|
|
1171
2637
|
kind_totals: dict[str, int] = {}
|
|
1172
|
-
|
|
2638
|
+
tool_counts: dict[str, int] = {}
|
|
2639
|
+
models: dict[str, int] = {}
|
|
2640
|
+
# Keyed on the ITEM index, which is what _build_segment_index records. Using
|
|
2641
|
+
# ``len(turns)`` would be correct only for as long as this loop appends a
|
|
2642
|
+
# turn for every item without exception; a later ``continue`` would misalign
|
|
2643
|
+
# every subsequent turn's segment keys, and the plausible-looking
|
|
2644
|
+
# ``[item_key]`` fallback would hide it by returning a well-formed answer.
|
|
2645
|
+
for index, it in enumerate(items):
|
|
1173
2646
|
meta = _item_meta(it)
|
|
1174
2647
|
anchor_text = _row_display(it["anchor_row"])
|
|
1175
2648
|
if meta is not None:
|
|
1176
2649
|
label = _META_LABEL_TEXT.get(meta["meta_label"], "Harness context")
|
|
1177
2650
|
else:
|
|
1178
|
-
|
|
2651
|
+
# Built from anchor-row TEXT, which is why it does not reach
|
|
2652
|
+
# `_display_chain` and has to clean through the shared helper here
|
|
2653
|
+
# (§5.1). A label with no recognized markup passes through byte for
|
|
2654
|
+
# byte, so this cannot move an ordinary prose label.
|
|
2655
|
+
label = _clean_outline_label(
|
|
2656
|
+
_first_nonblank_line(_strip_ansi(anchor_text))) if anchor_text else ""
|
|
1179
2657
|
kinds: dict[str, int] = {}
|
|
2658
|
+
tools: list[dict] = []
|
|
2659
|
+
tool_slot: dict[str | None, int] = {}
|
|
2660
|
+
tool_call_count = 0
|
|
2661
|
+
first_failure_name: str | None = None
|
|
2662
|
+
thinking: list[str] = []
|
|
1180
2663
|
for r in it["rows"]:
|
|
1181
2664
|
kinds[r.kind] = kinds.get(r.kind, 0) + 1
|
|
1182
2665
|
kind_totals[r.kind] = kind_totals.get(r.kind, 0) + 1
|
|
2666
|
+
position = (r.source_path, r.line_offset)
|
|
2667
|
+
if r.kind == "tool_call":
|
|
2668
|
+
tool_call_count += 1
|
|
2669
|
+
name = _tool_call_name(r)
|
|
2670
|
+
failed = position in failed_calls
|
|
2671
|
+
if failed and first_failure_name is None:
|
|
2672
|
+
first_failure_name = name
|
|
2673
|
+
if name is not None:
|
|
2674
|
+
tool_counts[name] = tool_counts.get(name, 0) + 1
|
|
2675
|
+
slot = tool_slot.get(name)
|
|
2676
|
+
if slot is None:
|
|
2677
|
+
tool_slot[name] = len(tools)
|
|
2678
|
+
tools.append({"name": name, "is_error": failed})
|
|
2679
|
+
elif failed:
|
|
2680
|
+
tools[slot]["is_error"] = True
|
|
2681
|
+
elif r.kind == "reasoning":
|
|
2682
|
+
thinking.extend(derivation.headings_by_position.get(position, ()))
|
|
2683
|
+
item_key = _item_key_for_item(conversation_key, it)
|
|
1183
2684
|
turn = {
|
|
1184
|
-
"item_key":
|
|
2685
|
+
"item_key": item_key,
|
|
1185
2686
|
"member_item_keys": _member_item_keys(conversation_key, it),
|
|
2687
|
+
"segment_item_keys": segment_keys.get(index, [item_key]),
|
|
1186
2688
|
"label": label,
|
|
1187
2689
|
"timestamp_utc": it["anchor_row"].timestamp_utc,
|
|
1188
2690
|
"kinds": kinds,
|
|
1189
2691
|
}
|
|
2692
|
+
# Additive, and only where there is something to say: a turn with no
|
|
2693
|
+
# calls publishes neither an empty array nor a zero count, matching how
|
|
2694
|
+
# the Claude outline omits `tools` and `thinking`.
|
|
2695
|
+
if tools:
|
|
2696
|
+
turn["tools"] = tools
|
|
2697
|
+
turn["tool_call_count"] = tool_call_count
|
|
2698
|
+
turn["first_failure_name"] = first_failure_name
|
|
2699
|
+
if thinking:
|
|
2700
|
+
turn["thinking"] = thinking
|
|
2701
|
+
item_model = _item_model(it) if _item_kind(it) == "assistant" else None
|
|
2702
|
+
if item_model:
|
|
2703
|
+
turn["model"] = item_model
|
|
2704
|
+
models[item_model] = models.get(item_model, 0) + 1
|
|
1190
2705
|
if meta is not None:
|
|
1191
2706
|
turn.update(meta)
|
|
1192
2707
|
turns.append(turn)
|
|
2708
|
+
total_cost, tokens = _conversation_totals(
|
|
2709
|
+
conn, conversation_key, effective_speed)
|
|
1193
2710
|
return {
|
|
1194
2711
|
"status": "ok",
|
|
1195
2712
|
"conversation_key": conversation_key,
|
|
1196
2713
|
"turns": turns,
|
|
1197
|
-
|
|
1198
|
-
|
|
2714
|
+
# Tier 2, deliberately a SEPARATE array (§3.3): `adaptQualifiedOutline`
|
|
2715
|
+
# derives `stats.turns.{human,assistant,tool_result,meta}` by filtering
|
|
2716
|
+
# `turns[]` on kind, so putting landmarks there would inflate counts
|
|
2717
|
+
# meant to describe the conversation's structure.
|
|
2718
|
+
"landmarks": _build_landmarks(
|
|
2719
|
+
segment_index, derivation, failed_calls),
|
|
2720
|
+
"stats": {
|
|
2721
|
+
"items": len(items), "kinds": kind_totals,
|
|
2722
|
+
"cost_usd": total_cost, "tokens": tokens,
|
|
2723
|
+
"tool_counts": tool_counts, "models": models,
|
|
2724
|
+
"duration_seconds": _conversation_duration_seconds(rows),
|
|
2725
|
+
"error_count": _outline_error_count(
|
|
2726
|
+
failed_calls, outcome_positions, derivation),
|
|
2727
|
+
},
|
|
2728
|
+
"files": _conversation_files(segment_index, derivation),
|
|
1199
2729
|
"children": _children_of(conn, conversation_key, effective_speed),
|
|
1200
2730
|
}
|
|
1201
2731
|
|
|
@@ -1209,6 +2739,16 @@ def _is_fork(fields: dict) -> bool:
|
|
|
1209
2739
|
|
|
1210
2740
|
|
|
1211
2741
|
def _browse_row(conn: sqlite3.Connection, conversation_key: str, effective_speed: str, fields: dict) -> dict:
|
|
2742
|
+
return _browse_row_from_fields(
|
|
2743
|
+
conversation_key, fields,
|
|
2744
|
+
cost_usd=_conversation_total_cost(conn, conversation_key, effective_speed),
|
|
2745
|
+
parent=_parent_of(conn, conversation_key),
|
|
2746
|
+
)
|
|
2747
|
+
|
|
2748
|
+
|
|
2749
|
+
def _browse_row_from_fields(
|
|
2750
|
+
conversation_key: str, fields: dict, *, cost_usd: float, parent,
|
|
2751
|
+
) -> dict:
|
|
1212
2752
|
return {
|
|
1213
2753
|
"conversation_key": conversation_key,
|
|
1214
2754
|
"title": _display_chain(fields),
|
|
@@ -1217,9 +2757,9 @@ def _browse_row(conn: sqlite3.Connection, conversation_key: str, effective_speed
|
|
|
1217
2757
|
"started_utc": fields["started"],
|
|
1218
2758
|
"last_activity_utc": fields["last"],
|
|
1219
2759
|
"count": fields["item_count"],
|
|
1220
|
-
"cost_usd":
|
|
2760
|
+
"cost_usd": cost_usd,
|
|
1221
2761
|
"models": list(fields["models"]),
|
|
1222
|
-
"parent":
|
|
2762
|
+
"parent": parent,
|
|
1223
2763
|
"is_fork": _is_fork(fields),
|
|
1224
2764
|
}
|
|
1225
2765
|
|
|
@@ -1264,6 +2804,200 @@ def _paginate_rows(rows: list[dict], *, cursor: str | None, limit: int):
|
|
|
1264
2804
|
return window, page
|
|
1265
2805
|
|
|
1266
2806
|
|
|
2807
|
+
def _stored_rollups_present(conn: sqlite3.Connection) -> bool:
|
|
2808
|
+
"""Whether the authoritative stored-rollup branch has been materialized.
|
|
2809
|
+
|
|
2810
|
+
Normal writes update normalized rows and their rollups in one transaction.
|
|
2811
|
+
The only supported no-rollup state is the pre-first-recompute window, where
|
|
2812
|
+
the live branch keeps the rail available. This constant-time probe avoids
|
|
2813
|
+
reintroducing the whole-message-table DISTINCT scan on every cold browse.
|
|
2814
|
+
"""
|
|
2815
|
+
return conn.execute(
|
|
2816
|
+
"SELECT 1 FROM codex_conversation_rollups LIMIT 1").fetchone() is not None
|
|
2817
|
+
|
|
2818
|
+
|
|
2819
|
+
def _live_browse_fields(conn: sqlite3.Connection) -> list[tuple[str, dict]]:
|
|
2820
|
+
"""Live-recompute fallback used only while no stored rollups exist."""
|
|
2821
|
+
out = []
|
|
2822
|
+
for (conversation_key,) in conn.execute(
|
|
2823
|
+
"SELECT DISTINCT conversation_key FROM codex_conversation_messages"):
|
|
2824
|
+
fields = _rollup_fields(conn, conversation_key)
|
|
2825
|
+
if fields is not None:
|
|
2826
|
+
out.append((conversation_key, fields))
|
|
2827
|
+
return out
|
|
2828
|
+
|
|
2829
|
+
|
|
2830
|
+
def _facets_from_fields(fields_rows: list[tuple[str, dict]]) -> dict:
|
|
2831
|
+
return _browse_facets([
|
|
2832
|
+
{
|
|
2833
|
+
"project_key": fields["project_key"],
|
|
2834
|
+
"project_label": fields["project_label"],
|
|
2835
|
+
"models": list(fields["models"]),
|
|
2836
|
+
}
|
|
2837
|
+
for _conversation_key, fields in fields_rows
|
|
2838
|
+
])
|
|
2839
|
+
|
|
2840
|
+
|
|
2841
|
+
def _stored_browse_facets(conn: sqlite3.Connection) -> dict:
|
|
2842
|
+
fields_rows = []
|
|
2843
|
+
for conversation_key, project_key, project_label, models_json in conn.execute(
|
|
2844
|
+
"SELECT conversation_key, project_key, project_label, models_json "
|
|
2845
|
+
"FROM codex_conversation_rollups"
|
|
2846
|
+
):
|
|
2847
|
+
try:
|
|
2848
|
+
parsed = json.loads(models_json) if models_json else []
|
|
2849
|
+
models = parsed if isinstance(parsed, list) else []
|
|
2850
|
+
except (TypeError, json.JSONDecodeError):
|
|
2851
|
+
models = []
|
|
2852
|
+
fields_rows.append((conversation_key, {
|
|
2853
|
+
"project_key": project_key,
|
|
2854
|
+
"project_label": project_label,
|
|
2855
|
+
"models": models,
|
|
2856
|
+
}))
|
|
2857
|
+
return _facets_from_fields(fields_rows)
|
|
2858
|
+
|
|
2859
|
+
|
|
2860
|
+
def _stored_filter_sql(alias: str, project_key: str | None, model: str | None):
|
|
2861
|
+
clauses = []
|
|
2862
|
+
params = []
|
|
2863
|
+
if project_key is not None:
|
|
2864
|
+
clauses.append(f"{alias}.project_key = ?")
|
|
2865
|
+
params.append(project_key)
|
|
2866
|
+
if model is not None:
|
|
2867
|
+
# models_json is the writer's canonical JSON array. Searching for the
|
|
2868
|
+
# complete JSON string literal is exact and does not require JSON1.
|
|
2869
|
+
clauses.append(f"instr(COALESCE({alias}.models_json, ''), ?) > 0")
|
|
2870
|
+
params.append(json.dumps(model))
|
|
2871
|
+
return (" AND ".join(clauses) if clauses else "1"), params
|
|
2872
|
+
|
|
2873
|
+
|
|
2874
|
+
def _page_costs(
|
|
2875
|
+
conn: sqlite3.Connection, conversation_keys: list[str], effective_speed: str,
|
|
2876
|
+
) -> dict[str, float]:
|
|
2877
|
+
if not conversation_keys:
|
|
2878
|
+
return {}
|
|
2879
|
+
placeholders = ",".join("?" for _ in conversation_keys)
|
|
2880
|
+
totals = {key: 0.0 for key in conversation_keys}
|
|
2881
|
+
for ck, model, inp, cin, out, rout in conn.execute(
|
|
2882
|
+
"SELECT conversation_key, model, input_tokens, cached_input_tokens, "
|
|
2883
|
+
"output_tokens, reasoning_output_tokens FROM codex_session_entries "
|
|
2884
|
+
f"WHERE conversation_key IN ({placeholders}) "
|
|
2885
|
+
"ORDER BY conversation_key, id",
|
|
2886
|
+
conversation_keys,
|
|
2887
|
+
):
|
|
2888
|
+
totals[ck] += _calculate_codex_entry_cost(
|
|
2889
|
+
model or "", inp or 0, cin or 0, out or 0, rout or 0,
|
|
2890
|
+
speed=effective_speed)
|
|
2891
|
+
return totals
|
|
2892
|
+
|
|
2893
|
+
|
|
2894
|
+
def _stored_browse_page(
|
|
2895
|
+
conn: sqlite3.Connection, *, effective_speed: str,
|
|
2896
|
+
project_key: str | None, model: str | None, limit: int,
|
|
2897
|
+
cursor: str | None,
|
|
2898
|
+
):
|
|
2899
|
+
where_sql, filter_params = _stored_filter_sql("r", project_key, model)
|
|
2900
|
+
total = conn.execute(
|
|
2901
|
+
f"SELECT COUNT(*) FROM codex_conversation_rollups r WHERE {where_sql}",
|
|
2902
|
+
filter_params,
|
|
2903
|
+
).fetchone()[0]
|
|
2904
|
+
|
|
2905
|
+
cursor_row = None
|
|
2906
|
+
if cursor is not None:
|
|
2907
|
+
cursor_where, cursor_params = _stored_filter_sql("c", project_key, model)
|
|
2908
|
+
cursor_row = conn.execute(
|
|
2909
|
+
"SELECT COALESCE(c.last_activity_utc, ''), c.conversation_key "
|
|
2910
|
+
"FROM codex_conversation_rollups c "
|
|
2911
|
+
f"WHERE c.conversation_key = ? AND {cursor_where}",
|
|
2912
|
+
[cursor, *cursor_params],
|
|
2913
|
+
).fetchone()
|
|
2914
|
+
|
|
2915
|
+
page_where = [where_sql]
|
|
2916
|
+
page_params = list(filter_params)
|
|
2917
|
+
if cursor_row is not None:
|
|
2918
|
+
cursor_last, cursor_key = cursor_row
|
|
2919
|
+
page_where.append(
|
|
2920
|
+
"(COALESCE(r.last_activity_utc, '') < ? OR "
|
|
2921
|
+
"(COALESCE(r.last_activity_utc, '') = ? AND r.conversation_key < ?))")
|
|
2922
|
+
page_params.extend((cursor_last, cursor_last, cursor_key))
|
|
2923
|
+
|
|
2924
|
+
sql = (
|
|
2925
|
+
"SELECT r.conversation_key, r.item_count, r.started_utc, "
|
|
2926
|
+
"r.last_activity_utc, r.project_key, r.project_label, r.models_json, "
|
|
2927
|
+
"r.title, r.parent_thread_id, r.source_root_key, t.native_thread_id, "
|
|
2928
|
+
"pt.conversation_key, pr.title, pr.project_label, pt.native_thread_id "
|
|
2929
|
+
"FROM codex_conversation_rollups r "
|
|
2930
|
+
"LEFT JOIN codex_conversation_threads t "
|
|
2931
|
+
"ON t.conversation_key = r.conversation_key "
|
|
2932
|
+
"LEFT JOIN codex_conversation_threads pt "
|
|
2933
|
+
"ON pt.source_root_key = r.source_root_key "
|
|
2934
|
+
"AND pt.native_thread_id = r.parent_thread_id "
|
|
2935
|
+
"AND pt.conversation_key != r.conversation_key "
|
|
2936
|
+
"LEFT JOIN codex_conversation_rollups pr "
|
|
2937
|
+
"ON pr.conversation_key = pt.conversation_key "
|
|
2938
|
+
f"WHERE {' AND '.join(page_where)} "
|
|
2939
|
+
"ORDER BY COALESCE(r.last_activity_utc, '') DESC, r.conversation_key DESC"
|
|
2940
|
+
)
|
|
2941
|
+
if limit:
|
|
2942
|
+
sql += " LIMIT ?"
|
|
2943
|
+
page_params.append(limit + 1)
|
|
2944
|
+
raw_rows = list(conn.execute(sql, page_params))
|
|
2945
|
+
has_more = bool(limit and len(raw_rows) > limit)
|
|
2946
|
+
if has_more:
|
|
2947
|
+
raw_rows = raw_rows[:limit]
|
|
2948
|
+
|
|
2949
|
+
keys = [row[0] for row in raw_rows]
|
|
2950
|
+
costs = _page_costs(conn, keys, effective_speed)
|
|
2951
|
+
rows = []
|
|
2952
|
+
for row in raw_rows:
|
|
2953
|
+
(conversation_key, item_count, started, last, row_project_key,
|
|
2954
|
+
project_label, models_json, title, parent_thread_id, source_root_key,
|
|
2955
|
+
native_thread_id, parent_key, parent_title, parent_project_label,
|
|
2956
|
+
parent_native_thread_id) = row
|
|
2957
|
+
try:
|
|
2958
|
+
parsed = json.loads(models_json) if models_json else []
|
|
2959
|
+
models = parsed if isinstance(parsed, list) else []
|
|
2960
|
+
except (TypeError, json.JSONDecodeError):
|
|
2961
|
+
models = []
|
|
2962
|
+
fields = {
|
|
2963
|
+
"item_count": item_count,
|
|
2964
|
+
"started": started,
|
|
2965
|
+
"last": last,
|
|
2966
|
+
"project_key": row_project_key,
|
|
2967
|
+
"project_label": project_label,
|
|
2968
|
+
"models": models,
|
|
2969
|
+
"title": title,
|
|
2970
|
+
"parent_thread_id": parent_thread_id,
|
|
2971
|
+
"source_root_key": source_root_key,
|
|
2972
|
+
"native_thread_id": native_thread_id,
|
|
2973
|
+
}
|
|
2974
|
+
parent = None
|
|
2975
|
+
if parent_key is not None:
|
|
2976
|
+
parent = {
|
|
2977
|
+
"conversation_key": parent_key,
|
|
2978
|
+
"title": _display_chain({
|
|
2979
|
+
"title": parent_title,
|
|
2980
|
+
"project_label": parent_project_label,
|
|
2981
|
+
"native_thread_id": parent_native_thread_id,
|
|
2982
|
+
}),
|
|
2983
|
+
}
|
|
2984
|
+
rows.append(_browse_row_from_fields(
|
|
2985
|
+
conversation_key, fields, cost_usd=costs.get(conversation_key, 0.0),
|
|
2986
|
+
parent=parent))
|
|
2987
|
+
next_cursor = rows[-1]["conversation_key"] if (rows and has_more) else None
|
|
2988
|
+
return rows, {"total": total, "returned": len(rows), "cursor": next_cursor}
|
|
2989
|
+
|
|
2990
|
+
|
|
2991
|
+
def list_codex_conversation_facets(conn: sqlite3.Connection) -> dict:
|
|
2992
|
+
"""Facet-only browse projection; never builds or prices a discarded page."""
|
|
2993
|
+
if not codex_normalization_authoritative(conn):
|
|
2994
|
+
return {"status": "normalization_pending",
|
|
2995
|
+
"facets": {"projects": [], "models": []}}
|
|
2996
|
+
facets = (_stored_browse_facets(conn) if _stored_rollups_present(conn)
|
|
2997
|
+
else _facets_from_fields(_live_browse_fields(conn)))
|
|
2998
|
+
return {"status": "ok", "facets": facets}
|
|
2999
|
+
|
|
3000
|
+
|
|
1267
3001
|
def list_codex_conversations(
|
|
1268
3002
|
conn: sqlite3.Connection,
|
|
1269
3003
|
*,
|
|
@@ -1282,20 +3016,22 @@ def list_codex_conversations(
|
|
|
1282
3016
|
if not codex_normalization_authoritative(conn):
|
|
1283
3017
|
return {"status": "normalization_pending", "rows": [],
|
|
1284
3018
|
"facets": {"projects": [], "models": []}, "page": {"total": 0}}
|
|
1285
|
-
|
|
1286
|
-
|
|
1287
|
-
|
|
1288
|
-
|
|
1289
|
-
|
|
1290
|
-
|
|
1291
|
-
|
|
1292
|
-
|
|
1293
|
-
|
|
1294
|
-
|
|
1295
|
-
|
|
1296
|
-
if (project_key is None or row["project_key"] == project_key)
|
|
1297
|
-
and (model is None or model in (row["models"] or []))
|
|
3019
|
+
if _stored_rollups_present(conn):
|
|
3020
|
+
facets = _stored_browse_facets(conn)
|
|
3021
|
+
page_rows, page = _stored_browse_page(
|
|
3022
|
+
conn, effective_speed=effective_speed, project_key=project_key,
|
|
3023
|
+
model=model, limit=limit, cursor=cursor)
|
|
3024
|
+
return {"status": "ok", "rows": page_rows, "facets": facets, "page": page}
|
|
3025
|
+
|
|
3026
|
+
fields_rows = _live_browse_fields(conn)
|
|
3027
|
+
rows = [
|
|
3028
|
+
_browse_row(conn, conversation_key, effective_speed, fields)
|
|
3029
|
+
for conversation_key, fields in fields_rows
|
|
1298
3030
|
]
|
|
3031
|
+
facets = _facets_from_fields(fields_rows)
|
|
3032
|
+
filtered = [row for row in rows
|
|
3033
|
+
if (project_key is None or row["project_key"] == project_key)
|
|
3034
|
+
and (model is None or model in (row["models"] or []))]
|
|
1299
3035
|
filtered.sort(key=_recent_sort_key, reverse=True)
|
|
1300
3036
|
page_rows, page = _paginate_rows(filtered, cursor=cursor, limit=limit)
|
|
1301
3037
|
return {"status": "ok", "rows": page_rows, "facets": facets, "page": page}
|
|
@@ -1321,27 +3057,445 @@ def _search_mode(conn: sqlite3.Connection) -> str:
|
|
|
1321
3057
|
return "fts"
|
|
1322
3058
|
|
|
1323
3059
|
|
|
1324
|
-
def
|
|
1325
|
-
|
|
1326
|
-
|
|
1327
|
-
|
|
1328
|
-
|
|
1329
|
-
|
|
3060
|
+
def _pos_to_item_key_and_order(
|
|
3061
|
+
conn: sqlite3.Connection, conversation_key: str,
|
|
3062
|
+
) -> tuple[dict, list[str]]:
|
|
3063
|
+
"""``_pos_to_item_key``'s map, plus the segment keys in detail document order.
|
|
3064
|
+
|
|
3065
|
+
Any caller that turns matched POSITIONS back into an ordered anchor list needs
|
|
3066
|
+
both halves, and it must take the order from the SAME segment index the map
|
|
3067
|
+
came from. Rebuilding the order from ``kern.canonical_items`` instead yields
|
|
3068
|
+
turn keys, which agree with the map only for segment 0 — so every hit past the
|
|
3069
|
+
first segment of a turn silently disappears.
|
|
3070
|
+
|
|
3071
|
+
Suppressed mirror members fold to their canonical partner's key, so both
|
|
3072
|
+
members of a pair share one key and can never double-count.
|
|
3073
|
+
|
|
3074
|
+
Resolving to the turn rather than to the segment is the defect that most
|
|
3075
|
+
nearly shipped: search and find derive their anchors here, so a find hit
|
|
3076
|
+
would jump to the head of a turn instead of to the matching content.
|
|
3077
|
+
|
|
3078
|
+
The narrow index read is enough — the map needs positions and keys, not
|
|
3079
|
+
``text``.
|
|
3080
|
+
"""
|
|
3081
|
+
rows, detail_bytes = _load_conversation_index_rows(conn, conversation_key)
|
|
1330
3082
|
partners = kern.pair_mirror_partners(rows)
|
|
1331
3083
|
kept, _suppressed = kern.pair_mirrors(rows)
|
|
1332
3084
|
items = kern.canonical_items(kept)
|
|
1333
3085
|
pos_map: dict[tuple, str] = {}
|
|
1334
|
-
|
|
1335
|
-
|
|
1336
|
-
|
|
1337
|
-
|
|
3086
|
+
order: list[str] = []
|
|
3087
|
+
for entry in _build_segment_index(
|
|
3088
|
+
conversation_key, items, detail_bytes, segmented=True):
|
|
3089
|
+
order.append(entry["item_key"])
|
|
3090
|
+
for r in entry["_rows"]:
|
|
3091
|
+
pos_map[(r.source_path, r.line_offset)] = entry["item_key"]
|
|
3092
|
+
# Lifecycle rows produce no block and so belong to no fold group, but
|
|
3093
|
+
# they are still physical rows a search hit can name. They resolve to
|
|
3094
|
+
# their turn's head segment.
|
|
3095
|
+
for r in entry["_lifecycle_rows"]:
|
|
3096
|
+
pos_map.setdefault((r.source_path, r.line_offset), entry["item_key"])
|
|
1338
3097
|
for sup_idx, canon_idx in partners.items():
|
|
1339
3098
|
sup = rows[sup_idx]
|
|
1340
3099
|
canon = rows[canon_idx]
|
|
1341
3100
|
canon_key = pos_map.get((canon.source_path, canon.line_offset))
|
|
1342
3101
|
if canon_key is not None:
|
|
1343
3102
|
pos_map[(sup.source_path, sup.line_offset)] = canon_key
|
|
1344
|
-
return pos_map
|
|
3103
|
+
return pos_map, order
|
|
3104
|
+
|
|
3105
|
+
|
|
3106
|
+
def _pos_to_item_key(conn: sqlite3.Connection, conversation_key: str) -> dict:
|
|
3107
|
+
"""Map every physical row ``(source_path, line_offset)`` of a conversation to
|
|
3108
|
+
the ``item_key`` of the SEGMENT that contains it (§6.2, #463 S1)."""
|
|
3109
|
+
return _pos_to_item_key_and_order(conn, conversation_key)[0]
|
|
3110
|
+
|
|
3111
|
+
|
|
3112
|
+
# ── #482 visible render-leaf projection ─────────────────────────────────────
|
|
3113
|
+
|
|
3114
|
+
CODEX_FIND_PROJECTION_VERSION = 1
|
|
3115
|
+
_COMPLETION_EVENT_TYPES = {
|
|
3116
|
+
"patch_apply_end",
|
|
3117
|
+
"web_search_end",
|
|
3118
|
+
"mcp_tool_call_end",
|
|
3119
|
+
}
|
|
3120
|
+
|
|
3121
|
+
|
|
3122
|
+
def _find_surface(row) -> str | None:
|
|
3123
|
+
if row.kind in {"user", "assistant", "reasoning"}:
|
|
3124
|
+
return "body"
|
|
3125
|
+
if row.kind == "tool_call":
|
|
3126
|
+
return "call"
|
|
3127
|
+
if row.kind == "tool_output":
|
|
3128
|
+
return "output"
|
|
3129
|
+
if row.kind == "event" and row.event_type in _COMPLETION_EVENT_TYPES:
|
|
3130
|
+
return "completion"
|
|
3131
|
+
return None
|
|
3132
|
+
|
|
3133
|
+
|
|
3134
|
+
def _project_plain_leaves(leaves: list[RenderLeaf]):
|
|
3135
|
+
"""Project structured-card leaves with a non-searchable visual boundary.
|
|
3136
|
+
|
|
3137
|
+
Native card fields render in separate block/inline containers. A newline
|
|
3138
|
+
between fields prevents a match from crossing that visual boundary while
|
|
3139
|
+
keeping each leaf's offsets local to the exact string its React component
|
|
3140
|
+
receives.
|
|
3141
|
+
"""
|
|
3142
|
+
text_parts: list[str] = []
|
|
3143
|
+
projected: list[ProjectedLeaf] = []
|
|
3144
|
+
cursor = 0
|
|
3145
|
+
for leaf in leaves:
|
|
3146
|
+
if not leaf.text:
|
|
3147
|
+
continue
|
|
3148
|
+
if text_parts:
|
|
3149
|
+
text_parts.append("\n")
|
|
3150
|
+
cursor += 1
|
|
3151
|
+
start = cursor
|
|
3152
|
+
text_parts.append(leaf.text)
|
|
3153
|
+
cursor += len(leaf.text)
|
|
3154
|
+
projected.append(ProjectedLeaf(leaf.key, start, cursor))
|
|
3155
|
+
return "".join(text_parts), tuple(projected)
|
|
3156
|
+
|
|
3157
|
+
|
|
3158
|
+
def _project_markdown_fields(fields: list[tuple[str, str]]):
|
|
3159
|
+
text_parts: list[str] = []
|
|
3160
|
+
leaves: list[ProjectedLeaf] = []
|
|
3161
|
+
cursor = 0
|
|
3162
|
+
for field, source in fields:
|
|
3163
|
+
if not source:
|
|
3164
|
+
continue
|
|
3165
|
+
if text_parts:
|
|
3166
|
+
text_parts.append("\n")
|
|
3167
|
+
cursor += 1
|
|
3168
|
+
projected_text, projected_leaves = project_markdown(source)
|
|
3169
|
+
text_parts.append(projected_text)
|
|
3170
|
+
leaves.extend(
|
|
3171
|
+
ProjectedLeaf(f"{field}/{leaf.key}", cursor + leaf.start, cursor + leaf.end)
|
|
3172
|
+
for leaf in projected_leaves
|
|
3173
|
+
)
|
|
3174
|
+
cursor += len(projected_text)
|
|
3175
|
+
return "".join(text_parts), tuple(leaves)
|
|
3176
|
+
|
|
3177
|
+
|
|
3178
|
+
def _patch_diff_leaves(files) -> list[RenderLeaf]:
|
|
3179
|
+
leaves: list[RenderLeaf] = []
|
|
3180
|
+
for file_index, file in enumerate(files or []):
|
|
3181
|
+
if not isinstance(file, dict):
|
|
3182
|
+
continue
|
|
3183
|
+
for field in ("path", "move_path"):
|
|
3184
|
+
value = file.get(field)
|
|
3185
|
+
if isinstance(value, str) and value:
|
|
3186
|
+
leaves.append(RenderLeaf(f"files.{file_index}.{field}", value))
|
|
3187
|
+
diff = file.get("unified_diff")
|
|
3188
|
+
if not isinstance(diff, str):
|
|
3189
|
+
continue
|
|
3190
|
+
hunk_index = -1
|
|
3191
|
+
row_index = 0
|
|
3192
|
+
for line in diff.replace("\r\n", "\n").replace("\r", "\n").split("\n"):
|
|
3193
|
+
if line.startswith("@@"):
|
|
3194
|
+
hunk_index += 1
|
|
3195
|
+
row_index = 0
|
|
3196
|
+
continue
|
|
3197
|
+
if hunk_index < 0 or not line or line.startswith(("--- ", "+++ ", "\\")):
|
|
3198
|
+
continue
|
|
3199
|
+
if line[0] not in {"+", "-", " "}:
|
|
3200
|
+
continue
|
|
3201
|
+
leaves.append(RenderLeaf(
|
|
3202
|
+
f"files.{file_index}.diff.{hunk_index}.{row_index}", line[1:]))
|
|
3203
|
+
row_index += 1
|
|
3204
|
+
return leaves
|
|
3205
|
+
|
|
3206
|
+
|
|
3207
|
+
def _json_card_text(value) -> str:
|
|
3208
|
+
return value if isinstance(value, str) else json.dumps(
|
|
3209
|
+
value, ensure_ascii=False, indent=2, separators=(",", ": "))
|
|
3210
|
+
|
|
3211
|
+
|
|
3212
|
+
def _project_completion_payload(payload: dict):
|
|
3213
|
+
patch = kern.decode_patch_event_card(payload)
|
|
3214
|
+
if patch is not None:
|
|
3215
|
+
leaves = _patch_diff_leaves(patch.get("files"))
|
|
3216
|
+
for field in ("stdout", "stderr"):
|
|
3217
|
+
value = patch.get(field)
|
|
3218
|
+
if isinstance(value, str) and value:
|
|
3219
|
+
leaves.append(RenderLeaf(field, _strip_ansi(value)))
|
|
3220
|
+
return _project_plain_leaves(leaves) if leaves else None
|
|
3221
|
+
|
|
3222
|
+
completion = kern.decode_secondary_event_card(payload)
|
|
3223
|
+
if completion is None:
|
|
3224
|
+
return None
|
|
3225
|
+
if completion.get("type") == "web_search_completion":
|
|
3226
|
+
leaves: list[RenderLeaf] = []
|
|
3227
|
+
for index, result in enumerate(completion.get("results") or []):
|
|
3228
|
+
if not isinstance(result, dict):
|
|
3229
|
+
continue
|
|
3230
|
+
for field in ("title", "domain", "snippet", "ref_id"):
|
|
3231
|
+
value = result.get(field)
|
|
3232
|
+
if isinstance(value, str) and value:
|
|
3233
|
+
leaves.append(RenderLeaf(f"results.{index}.{field}", value))
|
|
3234
|
+
error = completion.get("error")
|
|
3235
|
+
if error is not None:
|
|
3236
|
+
leaves.append(RenderLeaf("error", _json_card_text(error)))
|
|
3237
|
+
return _project_plain_leaves(leaves) if leaves else None
|
|
3238
|
+
if completion.get("type") == "mcp_completion":
|
|
3239
|
+
leaves = [
|
|
3240
|
+
RenderLeaf("arguments", _json_card_text(completion.get("arguments"))),
|
|
3241
|
+
RenderLeaf("result", _json_card_text(completion.get("result"))),
|
|
3242
|
+
]
|
|
3243
|
+
return _project_plain_leaves(leaves)
|
|
3244
|
+
return None
|
|
3245
|
+
|
|
3246
|
+
|
|
3247
|
+
def _project_find_row(row, *, payload: dict | None = None, block: dict | None = None):
|
|
3248
|
+
if row.kind == "event" and row.event_type in _COMPLETION_EVENT_TYPES and payload:
|
|
3249
|
+
completion = _project_completion_payload(payload)
|
|
3250
|
+
if completion is not None:
|
|
3251
|
+
return completion
|
|
3252
|
+
text = _row_display(row)
|
|
3253
|
+
if not text:
|
|
3254
|
+
return None
|
|
3255
|
+
if row.kind == "reasoning" and isinstance(block, dict):
|
|
3256
|
+
detail = block.get("detail")
|
|
3257
|
+
reasoning = detail.get("reasoning") if isinstance(detail, dict) else None
|
|
3258
|
+
if isinstance(reasoning, dict):
|
|
3259
|
+
visible_headings = block.get("_find_visible_headings")
|
|
3260
|
+
if isinstance(visible_headings, list):
|
|
3261
|
+
leaves = [
|
|
3262
|
+
RenderLeaf(leaf_key, text)
|
|
3263
|
+
for leaf_key, text in visible_headings
|
|
3264
|
+
if isinstance(leaf_key, str) and isinstance(text, str) and text
|
|
3265
|
+
]
|
|
3266
|
+
if leaves:
|
|
3267
|
+
return _project_plain_leaves(leaves)
|
|
3268
|
+
if reasoning.get("body") is None:
|
|
3269
|
+
return None
|
|
3270
|
+
fields = [
|
|
3271
|
+
(field, reasoning[field])
|
|
3272
|
+
for field in ("title", "summary", "body")
|
|
3273
|
+
if isinstance(reasoning.get(field), str) and reasoning[field]
|
|
3274
|
+
]
|
|
3275
|
+
if fields:
|
|
3276
|
+
return _project_markdown_fields(fields)
|
|
3277
|
+
if row.kind in {"user", "assistant", "reasoning"}:
|
|
3278
|
+
return project_markdown(text)
|
|
3279
|
+
if row.kind == "tool_output" and isinstance(block, dict):
|
|
3280
|
+
detail = block.get("detail")
|
|
3281
|
+
card = detail.get("card") if isinstance(detail, dict) else None
|
|
3282
|
+
if isinstance(card, dict) and card.get("type") == "terminal":
|
|
3283
|
+
output = card.get("output")
|
|
3284
|
+
parts = output.get("parts") if isinstance(output, dict) else None
|
|
3285
|
+
if isinstance(parts, list):
|
|
3286
|
+
stdout = "".join(
|
|
3287
|
+
part.get("text", "") for part in parts
|
|
3288
|
+
if isinstance(part, dict) and part.get("type") == "text"
|
|
3289
|
+
and part.get("stream") != "stderr"
|
|
3290
|
+
)
|
|
3291
|
+
stderr = "".join(
|
|
3292
|
+
part.get("text", "") for part in parts
|
|
3293
|
+
if isinstance(part, dict) and part.get("type") == "text"
|
|
3294
|
+
and part.get("stream") == "stderr"
|
|
3295
|
+
)
|
|
3296
|
+
leaves = []
|
|
3297
|
+
if stdout:
|
|
3298
|
+
leaves.append(RenderLeaf("stdout", _strip_ansi(stdout)))
|
|
3299
|
+
if stderr:
|
|
3300
|
+
leaves.append(RenderLeaf("stderr", _strip_ansi(stderr)))
|
|
3301
|
+
leaves.extend(
|
|
3302
|
+
RenderLeaf(f"raw.{index}", part["text"])
|
|
3303
|
+
for index, part in enumerate(parts)
|
|
3304
|
+
if isinstance(part, dict) and part.get("type") == "raw"
|
|
3305
|
+
and isinstance(part.get("text"), str) and part["text"]
|
|
3306
|
+
)
|
|
3307
|
+
if leaves:
|
|
3308
|
+
return _project_plain_leaves(leaves)
|
|
3309
|
+
if row.kind == "tool_call" and isinstance(block, dict):
|
|
3310
|
+
detail = block.get("detail")
|
|
3311
|
+
detail = detail if isinstance(detail, dict) else {}
|
|
3312
|
+
card = detail.get("card")
|
|
3313
|
+
if isinstance(card, dict):
|
|
3314
|
+
if card.get("type") == "patch":
|
|
3315
|
+
return None
|
|
3316
|
+
if card.get("type") == "web_search" and isinstance(card.get("query"), str):
|
|
3317
|
+
return project_plain((RenderLeaf("query", card["query"]),))
|
|
3318
|
+
if card.get("type") == "mcp":
|
|
3319
|
+
return None
|
|
3320
|
+
if card.get("type") == "terminal":
|
|
3321
|
+
commands = [
|
|
3322
|
+
RenderLeaf(f"commands.{index}", entry["command"])
|
|
3323
|
+
for index, entry in enumerate(card.get("commands") or [])
|
|
3324
|
+
if isinstance(entry, dict) and isinstance(entry.get("command"), str)
|
|
3325
|
+
]
|
|
3326
|
+
if commands:
|
|
3327
|
+
return _project_plain_leaves(commands)
|
|
3328
|
+
args = detail.get("args")
|
|
3329
|
+
if isinstance(args, str) and args:
|
|
3330
|
+
return project_plain((RenderLeaf("t0", args),))
|
|
3331
|
+
return project_plain((RenderLeaf("t0", text),))
|
|
3332
|
+
|
|
3333
|
+
|
|
3334
|
+
def materialize_codex_find_projection(
|
|
3335
|
+
conn: sqlite3.Connection,
|
|
3336
|
+
conversation_keys,
|
|
3337
|
+
) -> None:
|
|
3338
|
+
"""Replace #482 projection rows for the affected conversations.
|
|
3339
|
+
|
|
3340
|
+
The existing item/block builder is the only authority for native folds.
|
|
3341
|
+
Every searchable physical row keeps its own block key; a folded output or
|
|
3342
|
+
completion separately records the visual call block that owns it.
|
|
3343
|
+
"""
|
|
3344
|
+
keys = sorted({key for key in conversation_keys if key})
|
|
3345
|
+
if not keys:
|
|
3346
|
+
return
|
|
3347
|
+
for conversation_key in keys:
|
|
3348
|
+
conn.execute(
|
|
3349
|
+
"DELETE FROM codex_find_projection WHERE conversation_key=?",
|
|
3350
|
+
(conversation_key,),
|
|
3351
|
+
)
|
|
3352
|
+
rows = [
|
|
3353
|
+
kern.CodexNormalizedRow(*row)
|
|
3354
|
+
for row in conn.execute(
|
|
3355
|
+
"SELECT " + _ROW_COLS + " FROM codex_conversation_messages "
|
|
3356
|
+
"WHERE conversation_key=? "
|
|
3357
|
+
"ORDER BY timestamp_utc,source_path,line_offset",
|
|
3358
|
+
(conversation_key,),
|
|
3359
|
+
)
|
|
3360
|
+
]
|
|
3361
|
+
if not rows:
|
|
3362
|
+
continue
|
|
3363
|
+
kept, _suppressed = kern.pair_mirrors(rows)
|
|
3364
|
+
items = kern.canonical_items(kept)
|
|
3365
|
+
payloads = _load_row_payloads(conn, conversation_key)
|
|
3366
|
+
pos_to_item = _pos_to_item_key(conn, conversation_key)
|
|
3367
|
+
row_ids = {
|
|
3368
|
+
(source_path, line_offset): message_id
|
|
3369
|
+
for message_id, source_path, line_offset in conn.execute(
|
|
3370
|
+
"SELECT id,source_path,line_offset "
|
|
3371
|
+
"FROM codex_conversation_messages WHERE conversation_key=?",
|
|
3372
|
+
(conversation_key,),
|
|
3373
|
+
)
|
|
3374
|
+
}
|
|
3375
|
+
render_order = 0
|
|
3376
|
+
seen: set[tuple[str, int]] = set()
|
|
3377
|
+
seen_reasoning_by_turn: dict[str, set[str]] = {}
|
|
3378
|
+
|
|
3379
|
+
def store(row, *, container_block_key: str, block: dict | None = None) -> None:
|
|
3380
|
+
nonlocal render_order
|
|
3381
|
+
position = (row.source_path, row.line_offset)
|
|
3382
|
+
if position in seen:
|
|
3383
|
+
return
|
|
3384
|
+
surface = _find_surface(row)
|
|
3385
|
+
retained = _row_payload(row, payloads)
|
|
3386
|
+
payload = retained[1] if retained is not None else None
|
|
3387
|
+
projected = _project_find_row(row, payload=payload, block=block)
|
|
3388
|
+
message_id = row_ids.get(position)
|
|
3389
|
+
if surface is None or projected is None or message_id is None:
|
|
3390
|
+
return
|
|
3391
|
+
text, leaves = projected
|
|
3392
|
+
if not text:
|
|
3393
|
+
return
|
|
3394
|
+
physical_block_key = _block_key_for_row(row)
|
|
3395
|
+
item_key = pos_to_item.get(position)
|
|
3396
|
+
if item_key is None:
|
|
3397
|
+
return
|
|
3398
|
+
disclosure = (
|
|
3399
|
+
[container_block_key]
|
|
3400
|
+
if row.kind == "reasoning" or surface != "body"
|
|
3401
|
+
else []
|
|
3402
|
+
)
|
|
3403
|
+
conn.execute(
|
|
3404
|
+
"INSERT INTO codex_find_projection "
|
|
3405
|
+
"(message_id,conversation_key,item_key,block_key,"
|
|
3406
|
+
"container_block_key,surface,render_order,projected_text,"
|
|
3407
|
+
"leaves_json,disclosure_json,projection_version) "
|
|
3408
|
+
"VALUES (?,?,?,?,?,?,?,?,?,?,?)",
|
|
3409
|
+
(
|
|
3410
|
+
message_id,
|
|
3411
|
+
conversation_key,
|
|
3412
|
+
item_key,
|
|
3413
|
+
physical_block_key,
|
|
3414
|
+
container_block_key,
|
|
3415
|
+
surface,
|
|
3416
|
+
render_order,
|
|
3417
|
+
text,
|
|
3418
|
+
json.dumps(
|
|
3419
|
+
[
|
|
3420
|
+
{"key": leaf.key, "start": leaf.start, "end": leaf.end}
|
|
3421
|
+
for leaf in leaves
|
|
3422
|
+
],
|
|
3423
|
+
sort_keys=True,
|
|
3424
|
+
separators=(",", ":"),
|
|
3425
|
+
),
|
|
3426
|
+
json.dumps(disclosure, separators=(",", ":")),
|
|
3427
|
+
CODEX_FIND_PROJECTION_VERSION,
|
|
3428
|
+
),
|
|
3429
|
+
)
|
|
3430
|
+
seen.add(position)
|
|
3431
|
+
render_order += 1
|
|
3432
|
+
|
|
3433
|
+
for item in items:
|
|
3434
|
+
built_entries = _item_blocks_with_rows(
|
|
3435
|
+
item, payloads, decompose_headings=True,
|
|
3436
|
+
)
|
|
3437
|
+
completion_owner: dict[str, str] = {}
|
|
3438
|
+
for candidate_block, _candidate_primary, _candidate_output in built_entries:
|
|
3439
|
+
candidate_container = (
|
|
3440
|
+
candidate_block.get("block_key")
|
|
3441
|
+
or _block_key_for_row(_candidate_primary)
|
|
3442
|
+
)
|
|
3443
|
+
candidate_detail = candidate_block.get("detail")
|
|
3444
|
+
candidate_card = (
|
|
3445
|
+
candidate_detail.get("card")
|
|
3446
|
+
if isinstance(candidate_detail, dict) else None
|
|
3447
|
+
)
|
|
3448
|
+
completion = (
|
|
3449
|
+
candidate_card.get("completion")
|
|
3450
|
+
if isinstance(candidate_card, dict) else None
|
|
3451
|
+
)
|
|
3452
|
+
event_key = (
|
|
3453
|
+
completion.get("event_block_key")
|
|
3454
|
+
if isinstance(completion, dict) else None
|
|
3455
|
+
)
|
|
3456
|
+
if isinstance(event_key, str):
|
|
3457
|
+
completion_owner[event_key] = candidate_container
|
|
3458
|
+
|
|
3459
|
+
for block, primary, output in built_entries:
|
|
3460
|
+
container = block.get("block_key") or _block_key_for_row(primary)
|
|
3461
|
+
container = completion_owner.get(_block_key_for_row(primary), container)
|
|
3462
|
+
if primary.kind == "reasoning":
|
|
3463
|
+
detail = block.get("detail")
|
|
3464
|
+
reasoning = (
|
|
3465
|
+
detail.get("reasoning") if isinstance(detail, dict) else None
|
|
3466
|
+
)
|
|
3467
|
+
headings = (
|
|
3468
|
+
reasoning.get("headings")
|
|
3469
|
+
if isinstance(reasoning, dict) else None
|
|
3470
|
+
)
|
|
3471
|
+
if isinstance(headings, list):
|
|
3472
|
+
turn_key = primary.turn_id or item.get("turn_id") or ""
|
|
3473
|
+
prior = seen_reasoning_by_turn.setdefault(turn_key, set())
|
|
3474
|
+
visible = []
|
|
3475
|
+
for heading_index, heading in enumerate(headings):
|
|
3476
|
+
text = heading.get("text") if isinstance(heading, dict) else None
|
|
3477
|
+
if not isinstance(text, str) or text in prior:
|
|
3478
|
+
continue
|
|
3479
|
+
prior.add(text)
|
|
3480
|
+
visible.append((f"headings.{heading_index}", text))
|
|
3481
|
+
block["_find_visible_headings"] = visible
|
|
3482
|
+
store(primary, container_block_key=container, block=block)
|
|
3483
|
+
if output is not None:
|
|
3484
|
+
store(output, container_block_key=container, block=block)
|
|
3485
|
+
# Completion folds that intentionally produce no standalone block
|
|
3486
|
+
# still own a searchable physical surface and point at their call.
|
|
3487
|
+
for row in item["rows"]:
|
|
3488
|
+
if row.event_type not in _COMPLETION_EVENT_TYPES:
|
|
3489
|
+
continue
|
|
3490
|
+
physical_key = _block_key_for_row(row)
|
|
3491
|
+
container = completion_owner.get(physical_key, physical_key)
|
|
3492
|
+
store(row, container_block_key=container)
|
|
3493
|
+
|
|
3494
|
+
conn.execute(
|
|
3495
|
+
"INSERT INTO cache_meta(key,value) VALUES"
|
|
3496
|
+
"('codex_find_projection_generation','1') "
|
|
3497
|
+
"ON CONFLICT(key) DO UPDATE SET value=CAST(value AS INTEGER)+1"
|
|
3498
|
+
)
|
|
1345
3499
|
|
|
1346
3500
|
|
|
1347
3501
|
def _fts_query(query: str, column: str | None) -> str:
|
|
@@ -1407,7 +3561,60 @@ def _excerpt(text: str | None) -> str:
|
|
|
1407
3561
|
return collapsed[:200]
|
|
1408
3562
|
|
|
1409
3563
|
|
|
1410
|
-
def
|
|
3564
|
+
def _search_display_text(text: str | None) -> str:
|
|
3565
|
+
"""Readable search projection for retained structured content arrays.
|
|
3566
|
+
|
|
3567
|
+
Tool outputs must retain their provider JSON in ``search_tool`` so every
|
|
3568
|
+
leaf stays searchable. The rail, however, needs the same ordered text
|
|
3569
|
+
leaves a reader sees—not the serialized wrapper. Unknown JSON and future
|
|
3570
|
+
shapes fall back byte-for-byte to the retained string.
|
|
3571
|
+
"""
|
|
3572
|
+
if not text:
|
|
3573
|
+
return ""
|
|
3574
|
+
raw = str(text)
|
|
3575
|
+
if not raw.lstrip().startswith("["):
|
|
3576
|
+
return raw
|
|
3577
|
+
try:
|
|
3578
|
+
parsed = json.loads(raw)
|
|
3579
|
+
except (TypeError, json.JSONDecodeError):
|
|
3580
|
+
# Search columns are capped. A large content array can therefore end
|
|
3581
|
+
# mid-string and cease to be valid JSON even though one or more leading
|
|
3582
|
+
# text parts are complete. `_canonical_json` sorts object keys, so a
|
|
3583
|
+
# text-bearing part starts as `{\"text\":...}`. Decode only those
|
|
3584
|
+
# complete JSON string literals; never regex-unescape provider bytes.
|
|
3585
|
+
parts = []
|
|
3586
|
+
for match in re.finditer(r'(?:\A\[\{|,\{)"text":', raw):
|
|
3587
|
+
try:
|
|
3588
|
+
value, _end = json.JSONDecoder().raw_decode(raw, match.end())
|
|
3589
|
+
except (TypeError, json.JSONDecodeError):
|
|
3590
|
+
continue
|
|
3591
|
+
if isinstance(value, str) and value:
|
|
3592
|
+
parts.append(value)
|
|
3593
|
+
return "\n".join(parts) + ("\n…" if parts else "") or raw
|
|
3594
|
+
joined = kern._join_content_texts(parsed)
|
|
3595
|
+
return joined if joined else raw
|
|
3596
|
+
|
|
3597
|
+
|
|
3598
|
+
def _search_excerpt(text: str | None, query: str, width: int = 200) -> str:
|
|
3599
|
+
"""Whitespace-collapsed, match-centred excerpt from readable search text."""
|
|
3600
|
+
collapsed = " ".join(_search_display_text(text).split())
|
|
3601
|
+
if not collapsed:
|
|
3602
|
+
return ""
|
|
3603
|
+
needle = " ".join(query.split())
|
|
3604
|
+
found = collapsed.casefold().find(needle.casefold()) if needle else -1
|
|
3605
|
+
if found < 0 or len(collapsed) <= width:
|
|
3606
|
+
return collapsed[:width]
|
|
3607
|
+
start = max(0, found - (width // 3))
|
|
3608
|
+
end = min(len(collapsed), start + width)
|
|
3609
|
+
start = max(0, end - width)
|
|
3610
|
+
excerpt = collapsed[start:end]
|
|
3611
|
+
return (("… " if start else "") + excerpt
|
|
3612
|
+
+ (" …" if end < len(collapsed) else ""))
|
|
3613
|
+
|
|
3614
|
+
|
|
3615
|
+
def _collapse_message_hits(
|
|
3616
|
+
conn: sqlite3.Connection, matched_rows: list, query: str,
|
|
3617
|
+
) -> list[dict]:
|
|
1411
3618
|
"""Collapse matched physical rows to canonical ``item_key`` BEFORE totals /
|
|
1412
3619
|
badges (§6.2) — both members of a mirror pair map to one item_key, so mirror
|
|
1413
3620
|
rows never double-count (turned or unturned)."""
|
|
@@ -1430,7 +3637,7 @@ def _collapse_message_hits(conn: sqlite3.Connection, matched_rows: list) -> list
|
|
|
1430
3637
|
"last_activity_utc": last_act, "project_label": project_label})
|
|
1431
3638
|
hit["_badges"].add(_badge_for_kind(kind))
|
|
1432
3639
|
if hit["snippet"] is None:
|
|
1433
|
-
hit["snippet"] =
|
|
3640
|
+
hit["snippet"] = _search_excerpt(disp, query)
|
|
1434
3641
|
return [
|
|
1435
3642
|
{"conversation_key": h["conversation_key"], "item_key": h["item_key"],
|
|
1436
3643
|
"title": h["title"], "snippet": h["snippet"], "badges": sorted(h["_badges"]),
|
|
@@ -1441,22 +3648,31 @@ def _collapse_message_hits(conn: sqlite3.Connection, matched_rows: list) -> list
|
|
|
1441
3648
|
|
|
1442
3649
|
def _search_title(conn: sqlite3.Connection, query: str) -> list[dict]:
|
|
1443
3650
|
"""Title search over the rollup table — identical LIKE semantics in both FTS
|
|
1444
|
-
and LIKE modes (§6.2). Conversation-level hits (no item anchor).
|
|
3651
|
+
and LIKE modes (§6.2). Conversation-level hits (no item anchor).
|
|
3652
|
+
|
|
3653
|
+
#463 S4 §5.1 — the third read path that needs cleaning, and the one that is
|
|
3654
|
+
user-facing on the CLI: `cctally transcript search --source codex
|
|
3655
|
+
--kind title` prints this `snippet` and emits this `title` in its JSON. The
|
|
3656
|
+
MATCH still runs against the stored value, so a query that names markup
|
|
3657
|
+
still finds its conversation; only what is shown is cleaned.
|
|
3658
|
+
"""
|
|
1445
3659
|
like = f"%{query}%"
|
|
1446
3660
|
hits = []
|
|
1447
3661
|
for ck, title, last_act, project_label in conn.execute(
|
|
1448
3662
|
"SELECT conversation_key, title, last_activity_utc, project_label "
|
|
1449
3663
|
"FROM codex_conversation_rollups WHERE title LIKE ?", (like,)):
|
|
3664
|
+
cleaned = clean_codex_title(title)
|
|
1450
3665
|
hits.append(
|
|
1451
|
-
{"conversation_key": ck, "item_key": None, "title":
|
|
1452
|
-
"snippet": _excerpt(
|
|
3666
|
+
{"conversation_key": ck, "item_key": None, "title": cleaned,
|
|
3667
|
+
"snippet": _excerpt(cleaned), "badges": ["title"],
|
|
1453
3668
|
"last_activity_utc": last_act, "project_label": project_label})
|
|
1454
3669
|
return hits
|
|
1455
3670
|
|
|
1456
3671
|
|
|
1457
3672
|
def _search_files(conn: sqlite3.Connection, query: str) -> list[dict]:
|
|
1458
3673
|
"""File-touch search — matches file paths, collapsed to the owning message's
|
|
1459
|
-
canonical item_key (§6.2).
|
|
3674
|
+
canonical item_key (§6.2). ``message_id`` is an application-level link, so
|
|
3675
|
+
an orphan is skipped and cannot suppress valid rows."""
|
|
1460
3676
|
like = f"%{query}%"
|
|
1461
3677
|
pos_cache: dict[str, dict] = {}
|
|
1462
3678
|
fields_cache: dict[str, tuple] = {}
|
|
@@ -1531,7 +3747,7 @@ def search_codex_conversations(
|
|
|
1531
3747
|
hits = _search_files(conn, query)
|
|
1532
3748
|
else:
|
|
1533
3749
|
hits = _collapse_message_hits(
|
|
1534
|
-
conn, _matched_message_rows(conn, query, kind, mode))
|
|
3750
|
+
conn, _matched_message_rows(conn, query, kind, mode), query)
|
|
1535
3751
|
hits.sort(key=lambda h: (h["conversation_key"], h["item_key"] or ""))
|
|
1536
3752
|
total = len(hits)
|
|
1537
3753
|
page_hits, page = _paginate_hits(hits, cursor=cursor, limit=limit)
|
|
@@ -1543,6 +3759,402 @@ def search_codex_conversations(
|
|
|
1543
3759
|
|
|
1544
3760
|
# ── in-conversation find (§3.1) ───────────────────────────────────────────────
|
|
1545
3761
|
|
|
3762
|
+
_CODEX_EXACT_FIND_SCHEMA_VERSION = 2
|
|
3763
|
+
_CODEX_EXACT_FIND_DEFAULT_LIMIT = 100
|
|
3764
|
+
_CODEX_EXACT_FIND_MAX_LIMIT = 200
|
|
3765
|
+
_CODEX_EXACT_FIND_CURSOR_PREFIX = "ofc1."
|
|
3766
|
+
_CODEX_EXACT_FIND_QUERY_DOMAIN = b"cctally-codex-find-query-v1\0"
|
|
3767
|
+
_CODEX_EXACT_FIND_OCCURRENCE_DOMAIN = b"cctally-codex-find-occurrence-v1\0"
|
|
3768
|
+
|
|
3769
|
+
|
|
3770
|
+
class InvalidFindCursor(ValueError):
|
|
3771
|
+
"""The external exact-find cursor is malformed."""
|
|
3772
|
+
|
|
3773
|
+
|
|
3774
|
+
class StaleFindCursor(ValueError):
|
|
3775
|
+
"""The exact-find cursor belongs to another query or projection generation."""
|
|
3776
|
+
|
|
3777
|
+
|
|
3778
|
+
def _exact_find_query_id(
|
|
3779
|
+
query: str, *, regex: bool, case_sensitive: bool, kind: str
|
|
3780
|
+
) -> str:
|
|
3781
|
+
payload = json.dumps(
|
|
3782
|
+
{
|
|
3783
|
+
"case": case_sensitive,
|
|
3784
|
+
"kind": kind,
|
|
3785
|
+
"projection": CODEX_FIND_PROJECTION_VERSION,
|
|
3786
|
+
"query": query,
|
|
3787
|
+
"regex": regex,
|
|
3788
|
+
},
|
|
3789
|
+
ensure_ascii=False,
|
|
3790
|
+
sort_keys=True,
|
|
3791
|
+
separators=(",", ":"),
|
|
3792
|
+
).encode("utf-8")
|
|
3793
|
+
return hashlib.sha256(_CODEX_EXACT_FIND_QUERY_DOMAIN + payload).hexdigest()
|
|
3794
|
+
|
|
3795
|
+
|
|
3796
|
+
def _exact_find_occurrence_id(
|
|
3797
|
+
query_id: str,
|
|
3798
|
+
*,
|
|
3799
|
+
block_key: str,
|
|
3800
|
+
surface: str,
|
|
3801
|
+
ordinal: int,
|
|
3802
|
+
start: int,
|
|
3803
|
+
end: int,
|
|
3804
|
+
) -> str:
|
|
3805
|
+
payload = json.dumps(
|
|
3806
|
+
[
|
|
3807
|
+
CODEX_FIND_PROJECTION_VERSION,
|
|
3808
|
+
query_id,
|
|
3809
|
+
block_key,
|
|
3810
|
+
surface,
|
|
3811
|
+
ordinal,
|
|
3812
|
+
start,
|
|
3813
|
+
end,
|
|
3814
|
+
],
|
|
3815
|
+
ensure_ascii=False,
|
|
3816
|
+
separators=(",", ":"),
|
|
3817
|
+
).encode("utf-8")
|
|
3818
|
+
digest = hashlib.sha256(_CODEX_EXACT_FIND_OCCURRENCE_DOMAIN + payload).digest()
|
|
3819
|
+
return "o1." + base64.urlsafe_b64encode(digest).decode("ascii").rstrip("=")
|
|
3820
|
+
|
|
3821
|
+
|
|
3822
|
+
def _encode_exact_find_cursor(
|
|
3823
|
+
*,
|
|
3824
|
+
query_id: str,
|
|
3825
|
+
generation: int,
|
|
3826
|
+
start_index: int,
|
|
3827
|
+
direction: str,
|
|
3828
|
+
boundary: tuple[int, int, str, int],
|
|
3829
|
+
) -> str:
|
|
3830
|
+
payload = json.dumps(
|
|
3831
|
+
{
|
|
3832
|
+
"b": list(boundary),
|
|
3833
|
+
"d": direction,
|
|
3834
|
+
"g": generation,
|
|
3835
|
+
"i": start_index,
|
|
3836
|
+
"q": query_id,
|
|
3837
|
+
"v": CODEX_FIND_PROJECTION_VERSION,
|
|
3838
|
+
},
|
|
3839
|
+
sort_keys=True,
|
|
3840
|
+
separators=(",", ":"),
|
|
3841
|
+
).encode("utf-8")
|
|
3842
|
+
return _CODEX_EXACT_FIND_CURSOR_PREFIX + base64.urlsafe_b64encode(
|
|
3843
|
+
payload
|
|
3844
|
+
).decode("ascii").rstrip("=")
|
|
3845
|
+
|
|
3846
|
+
|
|
3847
|
+
def _decode_exact_find_cursor(cursor: str) -> dict[str, object]:
|
|
3848
|
+
if not isinstance(cursor, str) or not cursor.startswith(
|
|
3849
|
+
_CODEX_EXACT_FIND_CURSOR_PREFIX
|
|
3850
|
+
):
|
|
3851
|
+
raise InvalidFindCursor(cursor)
|
|
3852
|
+
encoded = cursor[len(_CODEX_EXACT_FIND_CURSOR_PREFIX):]
|
|
3853
|
+
try:
|
|
3854
|
+
raw = base64.urlsafe_b64decode(encoded + "=" * (-len(encoded) % 4))
|
|
3855
|
+
canonical = base64.urlsafe_b64encode(raw).decode("ascii").rstrip("=")
|
|
3856
|
+
if canonical != encoded:
|
|
3857
|
+
raise InvalidFindCursor(cursor)
|
|
3858
|
+
payload = json.loads(raw.decode("utf-8"))
|
|
3859
|
+
except (binascii.Error, ValueError, TypeError, UnicodeDecodeError, json.JSONDecodeError):
|
|
3860
|
+
raise InvalidFindCursor(cursor) from None
|
|
3861
|
+
if not isinstance(payload, dict) or set(payload) != {"b", "d", "g", "i", "q", "v"}:
|
|
3862
|
+
raise InvalidFindCursor(cursor)
|
|
3863
|
+
boundary = payload.get("b")
|
|
3864
|
+
if (
|
|
3865
|
+
payload.get("d") not in {"next", "previous"}
|
|
3866
|
+
or type(payload.get("g")) is not int
|
|
3867
|
+
or type(payload.get("i")) is not int
|
|
3868
|
+
or payload["i"] < 0
|
|
3869
|
+
or not isinstance(payload.get("q"), str)
|
|
3870
|
+
or payload.get("v") != CODEX_FIND_PROJECTION_VERSION
|
|
3871
|
+
or not isinstance(boundary, list)
|
|
3872
|
+
or len(boundary) != 4
|
|
3873
|
+
or type(boundary[0]) is not int
|
|
3874
|
+
or type(boundary[1]) is not int
|
|
3875
|
+
or not isinstance(boundary[2], str)
|
|
3876
|
+
or type(boundary[3]) is not int
|
|
3877
|
+
):
|
|
3878
|
+
raise InvalidFindCursor(cursor)
|
|
3879
|
+
return payload
|
|
3880
|
+
|
|
3881
|
+
|
|
3882
|
+
def _exact_find_base(
|
|
3883
|
+
query_id: str,
|
|
3884
|
+
*,
|
|
3885
|
+
status: str,
|
|
3886
|
+
regex: bool,
|
|
3887
|
+
kind: str,
|
|
3888
|
+
) -> dict[str, object]:
|
|
3889
|
+
return {
|
|
3890
|
+
"schema_version": _CODEX_EXACT_FIND_SCHEMA_VERSION,
|
|
3891
|
+
"semantics": "occurrence",
|
|
3892
|
+
"status": status,
|
|
3893
|
+
"query_id": query_id,
|
|
3894
|
+
"selection_stale": False,
|
|
3895
|
+
"mode": "regex" if regex else "literal",
|
|
3896
|
+
"kind": kind,
|
|
3897
|
+
"search_depth": "full",
|
|
3898
|
+
}
|
|
3899
|
+
|
|
3900
|
+
|
|
3901
|
+
def find_occurrences_in_codex_conversation(
|
|
3902
|
+
conn: sqlite3.Connection,
|
|
3903
|
+
conversation_key: str,
|
|
3904
|
+
query: str,
|
|
3905
|
+
*,
|
|
3906
|
+
regex: bool,
|
|
3907
|
+
case_sensitive: bool,
|
|
3908
|
+
kind: str,
|
|
3909
|
+
limit: int = _CODEX_EXACT_FIND_DEFAULT_LIMIT,
|
|
3910
|
+
cursor: str | None = None,
|
|
3911
|
+
direction: str = "next",
|
|
3912
|
+
around: str | None = None,
|
|
3913
|
+
) -> dict[str, object]:
|
|
3914
|
+
"""Return occurrence-exact matches over the materialized visible projection.
|
|
3915
|
+
|
|
3916
|
+
Matching never crosses a physical projection surface. Coordinates are
|
|
3917
|
+
Unicode-scalar offsets into stable render leaves, while paging cursors are
|
|
3918
|
+
bound to both query semantics and the current projection generation.
|
|
3919
|
+
"""
|
|
3920
|
+
if kind not in CODEX_FIND_KINDS:
|
|
3921
|
+
raise ValueError(f"unknown kind: {kind}")
|
|
3922
|
+
if not isinstance(limit, int) or not 1 <= limit <= _CODEX_EXACT_FIND_MAX_LIMIT:
|
|
3923
|
+
raise ValueError("find limit must be between 1 and 200")
|
|
3924
|
+
if direction not in {"next", "previous"}:
|
|
3925
|
+
raise ValueError("find direction must be next or previous")
|
|
3926
|
+
if cursor is not None and around is not None:
|
|
3927
|
+
raise ValueError("find cursor and around are mutually exclusive")
|
|
3928
|
+
q = (query or "").strip()
|
|
3929
|
+
query_id = _exact_find_query_id(
|
|
3930
|
+
q, regex=regex, case_sensitive=case_sensitive, kind=kind
|
|
3931
|
+
)
|
|
3932
|
+
exists = conn.execute(
|
|
3933
|
+
"SELECT 1 FROM codex_conversation_messages WHERE conversation_key=? LIMIT 1",
|
|
3934
|
+
(conversation_key,),
|
|
3935
|
+
).fetchone()
|
|
3936
|
+
if exists is None:
|
|
3937
|
+
return {"status": "not_found", "conversation_key": conversation_key}
|
|
3938
|
+
complete = conn.execute(
|
|
3939
|
+
"SELECT 1 FROM cache_meta WHERE "
|
|
3940
|
+
"key='codex_find_projection_complete_version' AND value=?",
|
|
3941
|
+
(str(CODEX_FIND_PROJECTION_VERSION),),
|
|
3942
|
+
).fetchone()
|
|
3943
|
+
base = _exact_find_base(query_id, status="ready", regex=regex, kind=kind)
|
|
3944
|
+
empty_page = {
|
|
3945
|
+
"start_index": 0,
|
|
3946
|
+
"previous_cursor": None,
|
|
3947
|
+
"next_cursor": None,
|
|
3948
|
+
"occurrences": [],
|
|
3949
|
+
}
|
|
3950
|
+
if complete is None:
|
|
3951
|
+
return {**base, "status": "indexing", "page": empty_page}
|
|
3952
|
+
generation_row = conn.execute(
|
|
3953
|
+
"SELECT value FROM cache_meta WHERE key='codex_find_projection_generation'"
|
|
3954
|
+
).fetchone()
|
|
3955
|
+
try:
|
|
3956
|
+
generation = int(generation_row[0]) if generation_row else 0
|
|
3957
|
+
except (TypeError, ValueError):
|
|
3958
|
+
generation = 0
|
|
3959
|
+
|
|
3960
|
+
decoded_cursor = None
|
|
3961
|
+
if cursor is not None:
|
|
3962
|
+
decoded_cursor = _decode_exact_find_cursor(cursor)
|
|
3963
|
+
if (
|
|
3964
|
+
decoded_cursor["q"] != query_id
|
|
3965
|
+
or decoded_cursor["g"] != generation
|
|
3966
|
+
or decoded_cursor["d"] != direction
|
|
3967
|
+
):
|
|
3968
|
+
raise StaleFindCursor(cursor)
|
|
3969
|
+
|
|
3970
|
+
if not q or (regex and len(q) > _CODEX_FIND_REGEX_MAX_LEN):
|
|
3971
|
+
return {**base, "total": 0, "page": empty_page}
|
|
3972
|
+
pattern = re.compile(q, 0 if case_sensitive else re.IGNORECASE) if regex else None
|
|
3973
|
+
kind_predicate = {
|
|
3974
|
+
"all": "1=1",
|
|
3975
|
+
"prompts": "m.kind='user'",
|
|
3976
|
+
"assistant": "m.kind='assistant'",
|
|
3977
|
+
"tools": "p.surface IN ('call','output','completion')",
|
|
3978
|
+
"thinking": "m.kind='reasoning'",
|
|
3979
|
+
}[kind]
|
|
3980
|
+
rows = conn.execute(
|
|
3981
|
+
"SELECT p.message_id,p.item_key,p.block_key,p.container_block_key,"
|
|
3982
|
+
"p.surface,p.render_order,"
|
|
3983
|
+
"p.projected_text,p.leaves_json,p.disclosure_json,m.kind "
|
|
3984
|
+
"FROM codex_find_projection p "
|
|
3985
|
+
"JOIN codex_conversation_messages m ON m.id=p.message_id "
|
|
3986
|
+
"WHERE p.conversation_key=? AND p.projection_version=? AND "
|
|
3987
|
+
+ kind_predicate
|
|
3988
|
+
+ " ORDER BY p.render_order,p.message_id,p.surface",
|
|
3989
|
+
(conversation_key, CODEX_FIND_PROJECTION_VERSION),
|
|
3990
|
+
)
|
|
3991
|
+
requested: list[tuple[dict[str, object], tuple[int, int, str, int]]] = []
|
|
3992
|
+
around_page: list[tuple[dict[str, object], tuple[int, int, str, int]]] = []
|
|
3993
|
+
head: list[tuple[dict[str, object], tuple[int, int, str, int]]] = []
|
|
3994
|
+
tail: deque[tuple[dict[str, object], tuple[int, int, str, int]]] = deque(
|
|
3995
|
+
maxlen=limit
|
|
3996
|
+
)
|
|
3997
|
+
head_next = None
|
|
3998
|
+
requested_next = None
|
|
3999
|
+
around_next = None
|
|
4000
|
+
around_index = None
|
|
4001
|
+
cursor_valid = decoded_cursor is None
|
|
4002
|
+
cursor_index = int(decoded_cursor["i"]) if decoded_cursor is not None else None
|
|
4003
|
+
requested_start = None
|
|
4004
|
+
requested_end = None
|
|
4005
|
+
if decoded_cursor is not None:
|
|
4006
|
+
if direction == "previous":
|
|
4007
|
+
requested_end = cursor_index
|
|
4008
|
+
requested_start = max(0, cursor_index - limit)
|
|
4009
|
+
else:
|
|
4010
|
+
requested_start = cursor_index
|
|
4011
|
+
requested_end = cursor_index + limit
|
|
4012
|
+
total = 0
|
|
4013
|
+
for (
|
|
4014
|
+
message_id,
|
|
4015
|
+
item_key,
|
|
4016
|
+
block_key,
|
|
4017
|
+
container_block_key,
|
|
4018
|
+
surface,
|
|
4019
|
+
render_order,
|
|
4020
|
+
text,
|
|
4021
|
+
leaves_json,
|
|
4022
|
+
disclosure_json,
|
|
4023
|
+
row_kind,
|
|
4024
|
+
) in rows:
|
|
4025
|
+
try:
|
|
4026
|
+
leaves = tuple(ProjectedLeaf(**leaf) for leaf in json.loads(leaves_json))
|
|
4027
|
+
disclosure = json.loads(disclosure_json)
|
|
4028
|
+
except (TypeError, ValueError, json.JSONDecodeError):
|
|
4029
|
+
continue
|
|
4030
|
+
ranges = (
|
|
4031
|
+
iter_regex_ranges(text, pattern)
|
|
4032
|
+
if pattern is not None
|
|
4033
|
+
else iter_literal_ranges(text, q, case_sensitive=case_sensitive)
|
|
4034
|
+
)
|
|
4035
|
+
for ordinal, match in enumerate(ranges):
|
|
4036
|
+
fragments = slice_range_to_leaves(match, leaves)
|
|
4037
|
+
if not fragments:
|
|
4038
|
+
continue
|
|
4039
|
+
occurrence_id = _exact_find_occurrence_id(
|
|
4040
|
+
query_id,
|
|
4041
|
+
block_key=block_key,
|
|
4042
|
+
surface=surface,
|
|
4043
|
+
ordinal=ordinal,
|
|
4044
|
+
start=match.start,
|
|
4045
|
+
end=match.end,
|
|
4046
|
+
)
|
|
4047
|
+
match_kinds = []
|
|
4048
|
+
if surface != "body":
|
|
4049
|
+
match_kinds.append("tool")
|
|
4050
|
+
if row_kind == "reasoning":
|
|
4051
|
+
match_kinds.append("thinking")
|
|
4052
|
+
occurrence = {
|
|
4053
|
+
"occurrence_id": occurrence_id,
|
|
4054
|
+
"item_key": item_key,
|
|
4055
|
+
"block_key": block_key,
|
|
4056
|
+
"container_block_key": container_block_key,
|
|
4057
|
+
"surface": surface,
|
|
4058
|
+
"match_kinds": match_kinds,
|
|
4059
|
+
"disclosure": disclosure if isinstance(disclosure, list) else [],
|
|
4060
|
+
"fragments": [
|
|
4061
|
+
{
|
|
4062
|
+
"leaf_key": fragment.leaf_key,
|
|
4063
|
+
"start": fragment.start,
|
|
4064
|
+
"end": fragment.end,
|
|
4065
|
+
}
|
|
4066
|
+
for fragment in fragments
|
|
4067
|
+
],
|
|
4068
|
+
}
|
|
4069
|
+
boundary = (render_order, message_id, surface, ordinal)
|
|
4070
|
+
pair = (occurrence, boundary)
|
|
4071
|
+
index = total
|
|
4072
|
+
if len(head) < limit:
|
|
4073
|
+
head.append(pair)
|
|
4074
|
+
elif index == limit:
|
|
4075
|
+
head_next = pair
|
|
4076
|
+
tail.append(pair)
|
|
4077
|
+
if cursor_index == index:
|
|
4078
|
+
if tuple(decoded_cursor["b"]) != boundary:
|
|
4079
|
+
raise StaleFindCursor(cursor)
|
|
4080
|
+
cursor_valid = True
|
|
4081
|
+
if (
|
|
4082
|
+
requested_start is not None
|
|
4083
|
+
and requested_end is not None
|
|
4084
|
+
and requested_start <= index < requested_end
|
|
4085
|
+
):
|
|
4086
|
+
requested.append(pair)
|
|
4087
|
+
elif requested_end is not None and index == requested_end:
|
|
4088
|
+
requested_next = pair
|
|
4089
|
+
if around is not None and around_index is None:
|
|
4090
|
+
if occurrence["occurrence_id"] == around:
|
|
4091
|
+
around_index = index
|
|
4092
|
+
around_page.append(pair)
|
|
4093
|
+
elif around_index is not None:
|
|
4094
|
+
if len(around_page) < limit:
|
|
4095
|
+
around_page.append(pair)
|
|
4096
|
+
elif index == around_index + limit:
|
|
4097
|
+
around_next = pair
|
|
4098
|
+
total += 1
|
|
4099
|
+
|
|
4100
|
+
if not cursor_valid:
|
|
4101
|
+
raise StaleFindCursor(cursor)
|
|
4102
|
+
selection_stale = around is not None and around_index is None
|
|
4103
|
+
next_pair = None
|
|
4104
|
+
if around is not None and around_index is not None:
|
|
4105
|
+
start_index = around_index
|
|
4106
|
+
page_pairs = around_page
|
|
4107
|
+
next_pair = around_next
|
|
4108
|
+
elif around is not None:
|
|
4109
|
+
start_index = 0
|
|
4110
|
+
page_pairs = head
|
|
4111
|
+
next_pair = head_next
|
|
4112
|
+
elif decoded_cursor is not None:
|
|
4113
|
+
start_index = min(requested_start or 0, total)
|
|
4114
|
+
page_pairs = requested
|
|
4115
|
+
next_pair = requested_next
|
|
4116
|
+
elif direction == "previous":
|
|
4117
|
+
page_pairs = list(tail)
|
|
4118
|
+
start_index = max(0, total - len(page_pairs))
|
|
4119
|
+
else:
|
|
4120
|
+
start_index = 0
|
|
4121
|
+
page_pairs = head
|
|
4122
|
+
next_pair = head_next
|
|
4123
|
+
page_occurrences = [occurrence for occurrence, _boundary in page_pairs]
|
|
4124
|
+
|
|
4125
|
+
def cursor_for(
|
|
4126
|
+
index: int,
|
|
4127
|
+
cursor_direction: str,
|
|
4128
|
+
pair: tuple[dict[str, object], tuple[int, int, str, int]] | None,
|
|
4129
|
+
) -> str | None:
|
|
4130
|
+
if not 0 <= index < total or pair is None:
|
|
4131
|
+
return None
|
|
4132
|
+
return _encode_exact_find_cursor(
|
|
4133
|
+
query_id=query_id,
|
|
4134
|
+
generation=generation,
|
|
4135
|
+
start_index=index,
|
|
4136
|
+
direction=cursor_direction,
|
|
4137
|
+
boundary=pair[1],
|
|
4138
|
+
)
|
|
4139
|
+
|
|
4140
|
+
previous_cursor = (
|
|
4141
|
+
cursor_for(start_index, "previous", page_pairs[0])
|
|
4142
|
+
if start_index > 0 and page_pairs else None
|
|
4143
|
+
)
|
|
4144
|
+
next_index = start_index + len(page_occurrences)
|
|
4145
|
+
next_cursor = cursor_for(next_index, "next", next_pair)
|
|
4146
|
+
return {
|
|
4147
|
+
**base,
|
|
4148
|
+
"total": total,
|
|
4149
|
+
"selection_stale": selection_stale,
|
|
4150
|
+
"page": {
|
|
4151
|
+
"start_index": start_index,
|
|
4152
|
+
"previous_cursor": previous_cursor,
|
|
4153
|
+
"next_cursor": next_cursor,
|
|
4154
|
+
"occurrences": page_occurrences,
|
|
4155
|
+
},
|
|
4156
|
+
}
|
|
4157
|
+
|
|
1546
4158
|
# Claude cap parity: the anchor list caps at 500 (bin/_lib_conversation_query.py
|
|
1547
4159
|
# ::_FIND_ANCHOR_CAP), with anchors_truncated when more anchors exist pre-cap.
|
|
1548
4160
|
_CODEX_FIND_ANCHOR_CAP = 500
|
|
@@ -1685,22 +4297,24 @@ def find_in_codex_conversation(
|
|
|
1685
4297
|
return base
|
|
1686
4298
|
# Collapse matched physical positions to canonical item_key (mirror-safe), then
|
|
1687
4299
|
# emit anchors in detail document order.
|
|
1688
|
-
|
|
4300
|
+
# The ORDER must come from the same segment index as the map (#463 S1). It
|
|
4301
|
+
# used to be rebuilt by walking `kern.canonical_items` and keying each entry
|
|
4302
|
+
# with `_item_key_for_item`, which produces a TURN key; a follower segment's
|
|
4303
|
+
# key can never equal one, so every hit past segment 0 of a turn was dropped
|
|
4304
|
+
# from the anchor list and from `total`. The FindBar then reported fewer
|
|
4305
|
+
# matches than exist and could navigate to none of the missing ones.
|
|
4306
|
+
pos_map, order = _pos_to_item_key_and_order(conn, conversation_key)
|
|
1689
4307
|
by_item: dict[str, set] = {}
|
|
1690
4308
|
for pos, labels in matched.items():
|
|
1691
4309
|
item_key = pos_map.get(pos)
|
|
1692
4310
|
if item_key is None:
|
|
1693
4311
|
continue
|
|
1694
4312
|
by_item.setdefault(item_key, set()).update(labels)
|
|
1695
|
-
|
|
1696
|
-
|
|
1697
|
-
|
|
1698
|
-
|
|
1699
|
-
|
|
1700
|
-
if item_key in by_item:
|
|
1701
|
-
anchors.append({
|
|
1702
|
-
"item_key": item_key,
|
|
1703
|
-
"match_kinds": sorted(l for l in by_item[item_key] if l != "prose")})
|
|
4313
|
+
anchors = [
|
|
4314
|
+
{"item_key": item_key,
|
|
4315
|
+
"match_kinds": sorted(l for l in by_item[item_key] if l != "prose")}
|
|
4316
|
+
for item_key in order if item_key in by_item
|
|
4317
|
+
]
|
|
1704
4318
|
total = len(anchors)
|
|
1705
4319
|
return {**base, "total": total, "anchors": anchors[:cap],
|
|
1706
4320
|
"anchors_truncated": total > cap}
|
|
@@ -1892,6 +4506,17 @@ def read_codex_payload(
|
|
|
1892
4506
|
response = {"status": "ok", "block_key": block_key, "which": which,
|
|
1893
4507
|
"content": content, "truncated": truncated}
|
|
1894
4508
|
if card is not None:
|
|
4509
|
+
# The SAME ordinal substitution the paged assembly applies (spec sections
|
|
4510
|
+
# 4.3 and 6.5). Without it this route published the provider's own
|
|
4511
|
+
# session id while the paged detail published the conversation-local
|
|
4512
|
+
# ordinal, so one field carried two meanings depending on which route
|
|
4513
|
+
# served it — and a client validator can only be written against one.
|
|
4514
|
+
# A session the index does not know becomes `ref: null`, never the raw
|
|
4515
|
+
# id, because `_apply_session_ordinals` fails closed.
|
|
4516
|
+
index_rows, _detail_bytes = _load_conversation_index_rows(
|
|
4517
|
+
conn, conversation_key)
|
|
4518
|
+
_envelope, ordinals = _build_session_index(index_rows)
|
|
4519
|
+
_apply_session_ordinals(card, ordinals)
|
|
1895
4520
|
response["card"] = card
|
|
1896
4521
|
return response
|
|
1897
4522
|
|