cctally 1.91.0 → 1.92.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +39 -0
- package/bin/_cctally_cache.py +863 -74
- package/bin/_cctally_config.py +57 -0
- package/bin/_cctally_core.py +39 -8
- package/bin/_cctally_dashboard.py +146 -5
- package/bin/_cctally_dashboard_conversation.py +164 -18
- package/bin/_cctally_dashboard_envelope.py +2 -0
- package/bin/_cctally_db.py +372 -10
- package/bin/_cctally_doctor.py +18 -1
- package/bin/_cctally_journal.py +535 -13
- package/bin/_cctally_journal_repair.py +6 -0
- package/bin/_cctally_parser.py +6 -0
- package/bin/_cctally_quota.py +171 -55
- package/bin/_cctally_record.py +13 -1
- package/bin/_cctally_rederive.py +4 -0
- package/bin/_cctally_store.py +311 -6
- package/bin/_cctally_transcript.py +32 -2
- package/bin/_lib_cache_report.py +8 -3
- package/bin/_lib_codex_conversation.py +851 -81
- package/bin/_lib_codex_conversation_query.py +2005 -95
- package/bin/_lib_codex_find_projection.py +370 -0
- package/bin/_lib_codex_harness_preamble.py +176 -0
- package/bin/_lib_codex_hooks.py +5 -3
- package/bin/_lib_codex_js_scan.py +254 -0
- package/bin/_lib_codex_landmarks.py +309 -0
- package/bin/_lib_codex_title_clean.py +116 -0
- package/bin/_lib_conversation_dispatch.py +153 -21
- package/bin/_lib_conversation_watch.py +4 -2
- package/bin/_lib_doctor.py +64 -0
- package/bin/_lib_quota_alert_axes.py +31 -34
- package/bin/_lib_stats_damage.py +523 -0
- package/bin/cctally +5 -0
- package/dashboard/static/assets/index-BEzzJtUd.js +97 -0
- package/dashboard/static/assets/{index-Dwirao3Y.css → index-DnWdv8um.css} +1 -1
- package/dashboard/static/dashboard.html +2 -2
- package/package.json +7 -1
- package/dashboard/static/assets/index-CILAoEja.js +0 -90
|
@@ -12,22 +12,42 @@ the caller at its I/O boundary, never here (§5.4).
|
|
|
12
12
|
Public names imported verbatim by the S7 dispatch layer — do not rename:
|
|
13
13
|
``codex_normalization_authoritative``, ``codex_item_key``,
|
|
14
14
|
``get_codex_conversation``, ``get_codex_conversation_outline``,
|
|
15
|
-
``list_codex_conversations``, ``
|
|
15
|
+
``list_codex_conversations``, ``list_codex_conversation_facets``,
|
|
16
|
+
``search_codex_conversations``,
|
|
16
17
|
``CODEX_SEARCH_KINDS``.
|
|
17
18
|
"""
|
|
18
19
|
from __future__ import annotations
|
|
19
20
|
|
|
21
|
+
import base64
|
|
22
|
+
import binascii
|
|
23
|
+
import collections
|
|
24
|
+
import contextlib
|
|
20
25
|
import hashlib
|
|
21
26
|
import json
|
|
22
27
|
import os
|
|
23
28
|
import re
|
|
24
29
|
import sqlite3
|
|
30
|
+
import threading
|
|
31
|
+
from collections import deque
|
|
25
32
|
|
|
26
33
|
import _lib_codex_conversation as kern
|
|
34
|
+
import _lib_codex_landmarks as landmarks
|
|
27
35
|
import _lib_codex_segments as segkern
|
|
28
|
-
from
|
|
36
|
+
from _lib_codex_find_projection import (
|
|
37
|
+
ProjectedLeaf,
|
|
38
|
+
RenderLeaf,
|
|
39
|
+
iter_literal_ranges,
|
|
40
|
+
iter_regex_ranges,
|
|
41
|
+
literal_ranges,
|
|
42
|
+
project_markdown,
|
|
43
|
+
project_plain,
|
|
44
|
+
regex_ranges,
|
|
45
|
+
slice_range_to_leaves,
|
|
46
|
+
)
|
|
47
|
+
from _lib_codex_title_clean import clean_codex_title
|
|
29
48
|
from _lib_conversation import _strip_ansi
|
|
30
|
-
from _lib_conversation_query import
|
|
49
|
+
from _lib_conversation_query import (
|
|
50
|
+
_FULL_PAYLOAD_CEILING, _first_nonblank_line, _parse_outline_ts)
|
|
31
51
|
from _lib_pricing import _calculate_codex_entry_cost
|
|
32
52
|
|
|
33
53
|
# ── constants ────────────────────────────────────────────────────────────────
|
|
@@ -389,6 +409,59 @@ def _load_rows_at_positions(
|
|
|
389
409
|
return hydrated
|
|
390
410
|
|
|
391
411
|
|
|
412
|
+
def _iter_row_payloads(
|
|
413
|
+
conn: sqlite3.Connection, conversation_key: str, positions=None,
|
|
414
|
+
):
|
|
415
|
+
"""Yield ``(position, record_type, payload)`` for retained physical payloads.
|
|
416
|
+
|
|
417
|
+
ONE copy of the event-table read: the per-source-path chunking, the bound
|
|
418
|
+
``IN (…)`` construction, the defensive re-filter and the
|
|
419
|
+
``{"payload": …}`` unwrap. ``_load_row_payloads`` collects this into a dict
|
|
420
|
+
and ``_derive_outline_events`` consumes it one row at a time, which is the
|
|
421
|
+
only difference between them — duplicating the read to get the streaming
|
|
422
|
+
form would mean a later correction to the unwrap or the chunking has to be
|
|
423
|
+
made twice, and the second is easy to miss.
|
|
424
|
+
|
|
425
|
+
``positions`` scopes the read to a known set (#463 S1, Phase C). Passing
|
|
426
|
+
``None`` reads the whole conversation, which the export path still wants and
|
|
427
|
+
which #463 S4 §4.1 forbids on the outline route.
|
|
428
|
+
|
|
429
|
+
A row whose ``payload_json`` does not parse, or whose ``payload`` member is
|
|
430
|
+
not an object, yields nothing — the caller therefore sees no entry for that
|
|
431
|
+
position rather than an empty one.
|
|
432
|
+
"""
|
|
433
|
+
def _batches():
|
|
434
|
+
"""One cursor per bound chunk, opened only when its turn comes."""
|
|
435
|
+
if positions is None:
|
|
436
|
+
yield conn.execute(
|
|
437
|
+
"SELECT source_path,line_offset,record_type,payload_json "
|
|
438
|
+
"FROM codex_conversation_events WHERE conversation_key = ?",
|
|
439
|
+
(conversation_key,)), None
|
|
440
|
+
return
|
|
441
|
+
scope = set(positions)
|
|
442
|
+
for path, chunks in _chunk_positions(scope).items():
|
|
443
|
+
for chunk in chunks:
|
|
444
|
+
yield conn.execute(
|
|
445
|
+
"SELECT source_path,line_offset,record_type,payload_json "
|
|
446
|
+
"FROM codex_conversation_events "
|
|
447
|
+
"WHERE conversation_key = ? AND source_path = ? "
|
|
448
|
+
f"AND line_offset IN ({','.join('?' for _ in chunk)})",
|
|
449
|
+
(conversation_key, path, *chunk)), scope
|
|
450
|
+
|
|
451
|
+
for cursor, wanted in _batches():
|
|
452
|
+
for source_path, line_offset, record_type, payload_json in cursor:
|
|
453
|
+
position = (source_path, line_offset)
|
|
454
|
+
if wanted is not None and position not in wanted:
|
|
455
|
+
continue
|
|
456
|
+
try:
|
|
457
|
+
obj = json.loads(payload_json or "{}")
|
|
458
|
+
except (json.JSONDecodeError, TypeError):
|
|
459
|
+
continue
|
|
460
|
+
payload = obj.get("payload") if isinstance(obj, dict) else None
|
|
461
|
+
if isinstance(payload, dict):
|
|
462
|
+
yield position, record_type, payload
|
|
463
|
+
|
|
464
|
+
|
|
392
465
|
def _load_row_payloads(
|
|
393
466
|
conn: sqlite3.Connection, conversation_key: str, positions=None,
|
|
394
467
|
) -> dict[tuple[str, int], tuple[str | None, dict]]:
|
|
@@ -401,42 +474,198 @@ def _load_row_payloads(
|
|
|
401
474
|
``positions`` scopes the read to one page (#463 S1, Phase C). Passing None
|
|
402
475
|
keeps the whole-conversation behaviour, which the export path still wants.
|
|
403
476
|
"""
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
(conversation_key,))
|
|
410
|
-
rows = list(cursor)
|
|
411
|
-
else:
|
|
412
|
-
wanted = set(positions)
|
|
413
|
-
rows = []
|
|
414
|
-
for path, chunks in _chunk_positions(wanted).items():
|
|
415
|
-
for chunk in chunks:
|
|
416
|
-
marks = ",".join("?" for _ in chunk)
|
|
417
|
-
rows.extend(
|
|
418
|
-
row for row in conn.execute(
|
|
419
|
-
"SELECT source_path,line_offset,record_type,payload_json "
|
|
420
|
-
"FROM codex_conversation_events "
|
|
421
|
-
"WHERE conversation_key = ? AND source_path = ? "
|
|
422
|
-
f"AND line_offset IN ({marks})",
|
|
423
|
-
(conversation_key, path, *chunk))
|
|
424
|
-
if (row[0], row[1]) in wanted)
|
|
425
|
-
for source_path, line_offset, record_type, payload_json in rows:
|
|
426
|
-
try:
|
|
427
|
-
obj = json.loads(payload_json or "{}")
|
|
428
|
-
except (json.JSONDecodeError, TypeError):
|
|
429
|
-
continue
|
|
430
|
-
payload = obj.get("payload") if isinstance(obj, dict) else None
|
|
431
|
-
if isinstance(payload, dict):
|
|
432
|
-
result[(source_path, line_offset)] = (record_type, payload)
|
|
433
|
-
return result
|
|
477
|
+
return {
|
|
478
|
+
position: (record_type, payload)
|
|
479
|
+
for position, record_type, payload
|
|
480
|
+
in _iter_row_payloads(conn, conversation_key, positions)
|
|
481
|
+
}
|
|
434
482
|
|
|
435
483
|
|
|
436
484
|
def _row_payload(row, payloads: dict) -> tuple[str | None, dict] | None:
|
|
437
485
|
return payloads.get((row.source_path, row.line_offset))
|
|
438
486
|
|
|
439
487
|
|
|
488
|
+
# The three lifecycle events that carry an outcome the §6.4 classification reads.
|
|
489
|
+
# A `tool_output` row is the fourth source and is selected by `kind`, not by
|
|
490
|
+
# event type, because it is a response item rather than an event.
|
|
491
|
+
_S4_OUTCOME_EVENTS = frozenset(
|
|
492
|
+
{"patch_apply_end", "web_search_end", "mcp_tool_call_end"})
|
|
493
|
+
|
|
494
|
+
|
|
495
|
+
# ── the outline derivation cache (#463 S4, D2's fallback) ────────────────────
|
|
496
|
+
#
|
|
497
|
+
# Task 11's re-measurement breached §4.7's first ceiling. On the heaviest
|
|
498
|
+
# production conversation the warm outline costs 332 ms, of which the event
|
|
499
|
+
# payload pass is 242 ms, against 229 ms for the detail page that opens beside
|
|
500
|
+
# it — so the outline HAD become the critical path on open, and the route is
|
|
501
|
+
# refetched on every live-tail growth push. §4.7 says a breach escalates to D2's
|
|
502
|
+
# fallback inside the session rather than deferring it, and this is that.
|
|
503
|
+
#
|
|
504
|
+
# **Why a position-keyed cache is sound.** A derivation entry is keyed by
|
|
505
|
+
# ``(source_path, line_offset)`` — a byte offset into an append-only rollout
|
|
506
|
+
# file. A line at a given offset is immutable, so the verdict for a position
|
|
507
|
+
# cannot change while the row exists, and a re-ingest rewrites the same bytes at
|
|
508
|
+
# the same offsets. The cache therefore EXTENDS on append rather than only
|
|
509
|
+
# hitting or missing: a growth push decodes the newly-in-scope positions and
|
|
510
|
+
# reuses every earlier one.
|
|
511
|
+
#
|
|
512
|
+
# **What the watermark is for.** Immutability covers append; it does not cover
|
|
513
|
+
# DELETION, and a deleted event row is exactly the case where the outline must
|
|
514
|
+
# stop reporting a verdict it can no longer support. The stored watermark is the
|
|
515
|
+
# cached prefix's ``(row count, max id)``, and reuse requires the count of rows
|
|
516
|
+
# still at ``id <= max id`` to equal it. Ids are ``AUTOINCREMENT`` and only
|
|
517
|
+
# increase, so no later insert can land inside that prefix — a preserved count
|
|
518
|
+
# therefore proves no row of the prefix was deleted. The check is one covering
|
|
519
|
+
# index read, measured at 0.18 ms on a conversation with 5,950 event rows.
|
|
520
|
+
#
|
|
521
|
+
# **Absence stays absence.** A position that was in scope and produced no entry
|
|
522
|
+
# — payload gone, unparseable, or a shape no decoder recognises — is recorded as
|
|
523
|
+
# covered and is not retried. That is deliberate: ``_outline_error_count`` reads
|
|
524
|
+
# absence as the third state "could not classify", and a payload does not appear
|
|
525
|
+
# later at an offset it was missing from.
|
|
526
|
+
#
|
|
527
|
+
# In-process only, so no cross-version staleness is possible: the binary that
|
|
528
|
+
# filled an entry is the binary that reads it.
|
|
529
|
+
_OUTLINE_DERIVATION_CACHE_MAX = 4
|
|
530
|
+
_outline_derivation_cache: "collections.OrderedDict[str, dict]" = (
|
|
531
|
+
collections.OrderedDict())
|
|
532
|
+
_outline_derivation_lock = threading.Lock()
|
|
533
|
+
|
|
534
|
+
|
|
535
|
+
def reset_outline_derivation_cache() -> None:
|
|
536
|
+
"""Drop every cached derivation. For tests and for measurement runs."""
|
|
537
|
+
with _outline_derivation_lock:
|
|
538
|
+
_outline_derivation_cache.clear()
|
|
539
|
+
|
|
540
|
+
|
|
541
|
+
def _outline_event_watermark(
|
|
542
|
+
conn: sqlite3.Connection, conversation_key: str, prefix_max_id: int | None,
|
|
543
|
+
) -> tuple[int, int | None, int]:
|
|
544
|
+
"""``(row count, max id, rows still at id <= prefix_max_id)`` in one read."""
|
|
545
|
+
count, max_id, prefix = conn.execute(
|
|
546
|
+
"SELECT COUNT(*), MAX(id), COALESCE(SUM(id <= ?), 0) "
|
|
547
|
+
"FROM codex_conversation_events WHERE conversation_key = ?",
|
|
548
|
+
(prefix_max_id if prefix_max_id is not None else -1, conversation_key),
|
|
549
|
+
).fetchone()
|
|
550
|
+
return count, max_id, prefix
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
def _derive_outline_events(
|
|
554
|
+
conn: sqlite3.Connection, conversation_key: str, rows: list,
|
|
555
|
+
) -> landmarks.EventDerivation:
|
|
556
|
+
"""The outline's read-time pass over the retained payloads (#463 S4, §4.1).
|
|
557
|
+
|
|
558
|
+
SCOPED and STREAMING, and Task 1 measured both halves of that on the
|
|
559
|
+
production store rather than assuming them.
|
|
560
|
+
|
|
561
|
+
*Scoped*: the position set comes from ``rows`` — the wide
|
|
562
|
+
``codex_conversation_messages`` read the outline has already performed — so
|
|
563
|
+
the pass never selects an event row no landmark can come from. That is the
|
|
564
|
+
idiom ``get_codex_conversation`` already uses two call sites above.
|
|
565
|
+
``positions=None`` would decode every event row of the conversation, which on
|
|
566
|
+
the heaviest one is 92.1 MB of JSON. Measured, scoping is worth 24% of that
|
|
567
|
+
read (109-122 ms and 183.1 MB of Python heap against 160 ms and 211.7 MB) —
|
|
568
|
+
real, but far less than the spec's framing implies, because 93.7% of that
|
|
569
|
+
conversation's payload bytes ARE the rows S4 needs. Do not treat scoping as
|
|
570
|
+
having solved the cost.
|
|
571
|
+
|
|
572
|
+
*Streaming*: each payload is decoded to the small fact the outline wants and
|
|
573
|
+
then dropped, rather than being retained in a dict the way
|
|
574
|
+
``_load_row_payloads`` retains it. Both go through the one
|
|
575
|
+
``_iter_row_payloads`` read; retaining is the only thing that differs.
|
|
576
|
+
Measured like for like — load AND decode,
|
|
577
|
+
to completion — the retaining form costs 254 ms and 182.9 MB of Python heap
|
|
578
|
+
on the heaviest conversation and this one costs 232 ms and 22.7 MB. That is
|
|
579
|
+
9% faster and 8x less memory, and peak RSS is in scope for §4.7's gate
|
|
580
|
+
precisely because this route is refetched on every live-tail growth push
|
|
581
|
+
rather than once per open.
|
|
582
|
+
|
|
583
|
+
Read-time only. Every decoder call takes ``for_storage=False`` and nothing
|
|
584
|
+
here is written back.
|
|
585
|
+
"""
|
|
586
|
+
outcomes: dict[tuple[str, int], object] = {}
|
|
587
|
+
headings_wanted: dict[tuple[str, int], object] = {}
|
|
588
|
+
for row in rows:
|
|
589
|
+
position = (row.source_path, row.line_offset)
|
|
590
|
+
if row.kind == "tool_output":
|
|
591
|
+
outcomes[position] = "output"
|
|
592
|
+
elif row.kind == "event" and row.event_type in _S4_OUTCOME_EVENTS:
|
|
593
|
+
outcomes[position] = row.event_type
|
|
594
|
+
elif row.kind == "reasoning":
|
|
595
|
+
# The same stored-detail gate `_reasoning_headings` applies, so the
|
|
596
|
+
# outline's headings are the set the reader route publishes and not
|
|
597
|
+
# a second, wider one.
|
|
598
|
+
detail = _parse_detail(row.detail_json)
|
|
599
|
+
if isinstance(detail, dict) and isinstance(detail.get("reasoning"), dict):
|
|
600
|
+
headings_wanted[position] = True
|
|
601
|
+
wanted = set(outcomes) | set(headings_wanted)
|
|
602
|
+
derivation = landmarks.EventDerivation()
|
|
603
|
+
if not wanted:
|
|
604
|
+
return derivation
|
|
605
|
+
|
|
606
|
+
with _outline_derivation_lock:
|
|
607
|
+
entry = _outline_derivation_cache.get(conversation_key)
|
|
608
|
+
count, max_id, prefix = _outline_event_watermark(
|
|
609
|
+
conn, conversation_key, entry["max_id"] if entry else None)
|
|
610
|
+
reusable = (entry is not None
|
|
611
|
+
and entry["count"] == prefix
|
|
612
|
+
and count >= entry["count"])
|
|
613
|
+
covered: set = set()
|
|
614
|
+
if reusable:
|
|
615
|
+
covered = entry["covered"]
|
|
616
|
+
derivation.errors_by_position.update(entry["derivation"].errors_by_position)
|
|
617
|
+
derivation.patch_files_by_position.update(
|
|
618
|
+
entry["derivation"].patch_files_by_position)
|
|
619
|
+
derivation.headings_by_position.update(
|
|
620
|
+
entry["derivation"].headings_by_position)
|
|
621
|
+
missing = wanted - covered
|
|
622
|
+
|
|
623
|
+
for position, _record_type, payload in (
|
|
624
|
+
_iter_row_payloads(conn, conversation_key, missing) if missing else ()):
|
|
625
|
+
kind = outcomes.get(position)
|
|
626
|
+
if kind == "output":
|
|
627
|
+
decoded = kern.decode_tool_output_card(payload, for_storage=False)
|
|
628
|
+
if decoded is not None:
|
|
629
|
+
derivation.errors_by_position[position] = (
|
|
630
|
+
landmarks.classify_tool_failure(
|
|
631
|
+
{"terminal_output": decoded[0]}))
|
|
632
|
+
elif kind == "patch_apply_end":
|
|
633
|
+
card = kern.decode_patch_event_card(payload, for_storage=False)
|
|
634
|
+
if card is not None:
|
|
635
|
+
derivation.errors_by_position[position] = (
|
|
636
|
+
landmarks.classify_tool_failure({"patch": card}))
|
|
637
|
+
# Counted from the UNBOUNDED raw `changes`, not off the card above,
|
|
638
|
+
# whose shared 16,000-character budget silently undercounts a large
|
|
639
|
+
# diff (§4.5).
|
|
640
|
+
derivation.patch_files_by_position[position] = (
|
|
641
|
+
landmarks.patch_file_touches(payload))
|
|
642
|
+
elif kind in _S4_OUTCOME_EVENTS:
|
|
643
|
+
card = kern.decode_secondary_event_card(payload)
|
|
644
|
+
if card is not None:
|
|
645
|
+
family = "web" if kind == "web_search_end" else "mcp"
|
|
646
|
+
derivation.errors_by_position[position] = (
|
|
647
|
+
landmarks.classify_tool_failure(
|
|
648
|
+
{family: {"completion": card}}))
|
|
649
|
+
if position in headings_wanted:
|
|
650
|
+
texts = landmarks.reasoning_heading_texts(payload)
|
|
651
|
+
if texts:
|
|
652
|
+
derivation.headings_by_position[position] = texts
|
|
653
|
+
|
|
654
|
+
# Stored AFTER the pass, so a raising derivation leaves the previous entry
|
|
655
|
+
# rather than a half-filled one. The value is the object just returned: it
|
|
656
|
+
# is never mutated again, because the next extension copies its three maps
|
|
657
|
+
# into a fresh `EventDerivation` above rather than adding to this one.
|
|
658
|
+
with _outline_derivation_lock:
|
|
659
|
+
_outline_derivation_cache[conversation_key] = {
|
|
660
|
+
"count": count, "max_id": max_id,
|
|
661
|
+
"covered": covered | wanted, "derivation": derivation,
|
|
662
|
+
}
|
|
663
|
+
_outline_derivation_cache.move_to_end(conversation_key)
|
|
664
|
+
while len(_outline_derivation_cache) > _OUTLINE_DERIVATION_CACHE_MAX:
|
|
665
|
+
_outline_derivation_cache.popitem(last=False)
|
|
666
|
+
return derivation
|
|
667
|
+
|
|
668
|
+
|
|
440
669
|
def _row_display(row) -> str:
|
|
441
670
|
"""The row's display/search text from whichever column carries it."""
|
|
442
671
|
return row.text or row.search_thinking or row.search_tool or ""
|
|
@@ -470,6 +699,50 @@ def _item_kind(item: dict) -> str:
|
|
|
470
699
|
return item["anchor_row"].kind # unturned: the row's own provider kind
|
|
471
700
|
|
|
472
701
|
|
|
702
|
+
def _item_model(item: dict) -> str | None:
|
|
703
|
+
"""The model a canonical tier-1 item states, or ``None`` (§4.2).
|
|
704
|
+
|
|
705
|
+
The anchor row first, then the item's own rows in order. Reading the anchor
|
|
706
|
+
row ALONE under-reported Codex model usage about sixfold: most Codex
|
|
707
|
+
response items anchor on a ``reasoning`` row, which carries no model, so a
|
|
708
|
+
conversation with 13 outline turns holding assistant rows and 82 assistant
|
|
709
|
+
rows in total rendered ``gpt-5.6-sol x2``, and 182 of 200 Codex
|
|
710
|
+
conversations in the production store reported a model total under a third
|
|
711
|
+
of their turn count. §4.2 defines the counting unit as the canonical tier-1
|
|
712
|
+
assistant TURN, mirroring the Claude side, whose histogram sums to exactly
|
|
713
|
+
``stats.turns.assistant``; the anchor row is one row of that turn.
|
|
714
|
+
"""
|
|
715
|
+
anchor = item["anchor_row"].model
|
|
716
|
+
if anchor:
|
|
717
|
+
return anchor
|
|
718
|
+
for row in item["rows"]:
|
|
719
|
+
if row.model:
|
|
720
|
+
return row.model
|
|
721
|
+
return None
|
|
722
|
+
|
|
723
|
+
|
|
724
|
+
def _outline_outcome_positions(rows: list) -> set:
|
|
725
|
+
"""Every position whose row could carry an outcome verdict this request."""
|
|
726
|
+
return {(row.source_path, row.line_offset) for row in rows
|
|
727
|
+
if row.kind == "tool_output"
|
|
728
|
+
or (row.kind == "event" and row.event_type in _S4_OUTCOME_EVENTS)}
|
|
729
|
+
|
|
730
|
+
|
|
731
|
+
def _outline_failing_calls(derivation, outcome_positions: set,
|
|
732
|
+
call_by_position: dict) -> set:
|
|
733
|
+
"""The calls this request found failing, charged to the call they fold into.
|
|
734
|
+
|
|
735
|
+
Intersected with ``outcome_positions`` because ``errors_by_position`` is
|
|
736
|
+
CACHED: the derivation is retained per conversation and EXTENDED on a
|
|
737
|
+
growth push, so the map can hold verdicts for positions the current request
|
|
738
|
+
never read, while every other consumer indexes it by a current position.
|
|
739
|
+
Defensive rather than a reproduction of an observed miscount.
|
|
740
|
+
"""
|
|
741
|
+
return {call_by_position.get(position, position)
|
|
742
|
+
for position in derivation.failing_positions()
|
|
743
|
+
if position in outcome_positions}
|
|
744
|
+
|
|
745
|
+
|
|
473
746
|
def _item_meta(item: dict) -> dict | None:
|
|
474
747
|
if item["klass"] != "meta":
|
|
475
748
|
return None
|
|
@@ -518,37 +791,163 @@ def _reasoning_headings(detail, payload, block_key: str):
|
|
|
518
791
|
returns ``None`` and the caller omits the field entirely, so the client falls
|
|
519
792
|
back to today's ``title``/``summary`` rendering. Decomposition never fails the
|
|
520
793
|
request and never partially populates.
|
|
794
|
+
|
|
795
|
+
#463 S4 — the summary parse itself moved to
|
|
796
|
+
``_lib_codex_landmarks.reasoning_heading_texts``, so the reader route here
|
|
797
|
+
and the outline's landmark derivation decompose by ONE rule. This function
|
|
798
|
+
keeps the two things the outline does not want: the stored-detail gate, and
|
|
799
|
+
the ``<block_key>#<ordinal>`` identity, which the outline mints from its own
|
|
800
|
+
block keys rather than from the reader's.
|
|
521
801
|
"""
|
|
522
802
|
if not isinstance(detail, dict) or not isinstance(detail.get("reasoning"), dict):
|
|
523
803
|
return None
|
|
524
|
-
|
|
525
|
-
return None
|
|
526
|
-
summary = payload.get("summary")
|
|
527
|
-
if not isinstance(summary, list) or not summary:
|
|
528
|
-
return None
|
|
529
|
-
entries = []
|
|
530
|
-
for entry in summary:
|
|
531
|
-
if not isinstance(entry, dict):
|
|
532
|
-
return None
|
|
533
|
-
text = entry.get("text")
|
|
534
|
-
# Mirror `_join_content_texts`, which is what produced the stored
|
|
535
|
-
# summary: it keeps non-empty string `text` leaves and ignores the rest.
|
|
536
|
-
if text is None:
|
|
537
|
-
continue
|
|
538
|
-
if not isinstance(text, str):
|
|
539
|
-
return None
|
|
540
|
-
if text:
|
|
541
|
-
entries.append(text)
|
|
542
|
-
headings = decompose_reasoning_headings(entries)
|
|
804
|
+
headings = landmarks.reasoning_heading_texts(payload)
|
|
543
805
|
if not headings:
|
|
544
806
|
return None
|
|
545
807
|
return [{"key": f"{block_key}#{ordinal}", "text": text}
|
|
546
808
|
for ordinal, text in enumerate(headings)]
|
|
547
809
|
|
|
548
810
|
|
|
811
|
+
# ── the conversation-level session index (#463 S3, spec section 3.2) ─────────
|
|
812
|
+
#
|
|
813
|
+
# Page-local adaptation cannot decide whether a session label is unique across
|
|
814
|
+
# the conversation or whether an opener exists, because later pages adapt
|
|
815
|
+
# independently and live-tail can append. So the server publishes a bounded
|
|
816
|
+
# conversation-scoped index and the client never computes either fact itself.
|
|
817
|
+
#
|
|
818
|
+
# 870 sessions across 223 conversations, roughly four per conversation, so this
|
|
819
|
+
# cap is generous. A conversation past it publishes what fits and marks itself
|
|
820
|
+
# truncated rather than publishing a partial map that looks complete.
|
|
821
|
+
_SESSION_INDEX_MAX = 64
|
|
822
|
+
|
|
823
|
+
|
|
824
|
+
def _stored_write_stdin_session(detail) -> str | None:
|
|
825
|
+
"""The session id a `write_stdin` call names, from its STORED arguments.
|
|
826
|
+
|
|
827
|
+
Phase A must not load event payloads, and it does not need to: `detail.args`
|
|
828
|
+
is in the narrow index, and it is the provider's own argument JSON.
|
|
829
|
+
"""
|
|
830
|
+
if not isinstance(detail, dict) or detail.get("name") != "write_stdin":
|
|
831
|
+
return None
|
|
832
|
+
args = detail.get("args")
|
|
833
|
+
if not isinstance(args, str) or not args:
|
|
834
|
+
return None
|
|
835
|
+
try:
|
|
836
|
+
parsed = json.loads(args)
|
|
837
|
+
except (json.JSONDecodeError, TypeError, ValueError):
|
|
838
|
+
return None
|
|
839
|
+
raw = parsed.get("session_id") if isinstance(parsed, dict) else None
|
|
840
|
+
if isinstance(raw, bool) or not isinstance(raw, (str, int)):
|
|
841
|
+
return None
|
|
842
|
+
return str(raw)
|
|
843
|
+
|
|
844
|
+
|
|
845
|
+
def _stored_session_announcement(detail) -> str | None:
|
|
846
|
+
"""The session id a tool output ANNOUNCES, from its stored card.
|
|
847
|
+
|
|
848
|
+
`Process running with session ID <id>` is the evidence linking a shell
|
|
849
|
+
session to the call that opened it — 698 of 870 sessions, 80.2%. The line is
|
|
850
|
+
read through the same anchored preamble reader the card path uses rather than
|
|
851
|
+
by searching the text, so a user's own output cannot be mistaken for one.
|
|
852
|
+
|
|
853
|
+
Spec section 3.2 names `search_tool` as the column this comes from. It is
|
|
854
|
+
read from the stored card in `detail_json` instead, which carries the same
|
|
855
|
+
bytes at the head of its first part and IS in the narrow index —
|
|
856
|
+
`_load_conversation_index_rows` excludes `search_tool` along with the other
|
|
857
|
+
two bulk columns, and adding it back would undo S1's Phase A saving.
|
|
858
|
+
"""
|
|
859
|
+
card = detail.get("card") if isinstance(detail, dict) else None
|
|
860
|
+
if not isinstance(card, dict) or card.get("type") != "terminal_output":
|
|
861
|
+
return None
|
|
862
|
+
parts = card.get("parts")
|
|
863
|
+
if not isinstance(parts, list) or not parts:
|
|
864
|
+
return None
|
|
865
|
+
head = parts[0]
|
|
866
|
+
text = head.get("text") if isinstance(head, dict) else None
|
|
867
|
+
if not isinstance(text, str):
|
|
868
|
+
return None
|
|
869
|
+
parsed = kern.parse_harness_preamble(text)
|
|
870
|
+
return parsed[0]["session_announcement"] if parsed is not None else None
|
|
871
|
+
|
|
872
|
+
|
|
873
|
+
def _build_session_index(rows) -> tuple[dict, dict[str, str]]:
|
|
874
|
+
"""``(envelope, ordinal_by_provider_session_id)`` over the WHOLE conversation.
|
|
875
|
+
|
|
876
|
+
Ordinals are assigned in first-appearance order over the conversation's
|
|
877
|
+
physical row order, so they are stable across pages and live-tail appends and
|
|
878
|
+
no client-side uniqueness decision is made from a partial window.
|
|
879
|
+
|
|
880
|
+
The envelope's `sessions` map is keyed by the ordinal in decimal, which is
|
|
881
|
+
exactly what a `session_ref` card's `ref` carries, so the client's lookup is
|
|
882
|
+
direct. Nothing in the envelope is derived from the provider's own session
|
|
883
|
+
id — that token is removed rather than scrubbed (spec section 4.3).
|
|
884
|
+
"""
|
|
885
|
+
owners: dict[str, list] = {}
|
|
886
|
+
for row in rows:
|
|
887
|
+
if row.kind == "tool_call" and row.call_id:
|
|
888
|
+
owners.setdefault(row.call_id, []).append(row)
|
|
889
|
+
ordinals: dict[str, int] = {}
|
|
890
|
+
openers: dict[str, str | None] = {}
|
|
891
|
+
truncated = False
|
|
892
|
+
for row in rows:
|
|
893
|
+
session = None
|
|
894
|
+
opener_row = None
|
|
895
|
+
detail = _parse_detail(row.detail_json)
|
|
896
|
+
if row.kind == "tool_call":
|
|
897
|
+
session = _stored_write_stdin_session(detail)
|
|
898
|
+
elif row.kind == "tool_output":
|
|
899
|
+
session = _stored_session_announcement(detail)
|
|
900
|
+
if session is not None:
|
|
901
|
+
# The opener is the CALL that owns the announcing output, not the
|
|
902
|
+
# output row: a uniquely-owned output folds into its call and has
|
|
903
|
+
# no block of its own, so its key would name nothing on the page.
|
|
904
|
+
owning = owners.get(row.call_id or "", [])
|
|
905
|
+
opener_row = owning[0] if len(owning) == 1 else row
|
|
906
|
+
if session is None:
|
|
907
|
+
continue
|
|
908
|
+
if session not in ordinals:
|
|
909
|
+
if len(ordinals) >= _SESSION_INDEX_MAX:
|
|
910
|
+
truncated = True
|
|
911
|
+
continue
|
|
912
|
+
ordinals[session] = len(ordinals) + 1
|
|
913
|
+
openers[session] = None
|
|
914
|
+
if opener_row is not None and openers.get(session) is None:
|
|
915
|
+
openers[session] = _block_key_for_row(opener_row)
|
|
916
|
+
envelope = {
|
|
917
|
+
"sessions": {
|
|
918
|
+
str(ordinal): {"ordinal": ordinal,
|
|
919
|
+
"opener_block_key": openers.get(session)}
|
|
920
|
+
for session, ordinal in ordinals.items()
|
|
921
|
+
},
|
|
922
|
+
"truncated": truncated,
|
|
923
|
+
}
|
|
924
|
+
return envelope, {session: str(ordinal) for session, ordinal in ordinals.items()}
|
|
925
|
+
|
|
926
|
+
|
|
927
|
+
def _apply_session_ordinals(card, ordinals: dict[str, str]) -> None:
|
|
928
|
+
"""Replace every SHELL session reference with its conversation-local ordinal.
|
|
929
|
+
|
|
930
|
+
Fails closed: a reference the index does not know becomes ``None`` rather
|
|
931
|
+
than falling back to the provider's id. `cell` scope is left alone — a cell
|
|
932
|
+
id is a small per-conversation sandbox ordinal that identifies nothing
|
|
933
|
+
outside the sandbox, and it is never presented as a shell session.
|
|
934
|
+
"""
|
|
935
|
+
if not isinstance(card, dict):
|
|
936
|
+
return
|
|
937
|
+
if card.get("type") == "session_ref" and card.get("scope") == "shell":
|
|
938
|
+
card["ref"] = ordinals.get(card.get("ref"))
|
|
939
|
+
return
|
|
940
|
+
if card.get("type") == "program":
|
|
941
|
+
for entry in card.get("invocations") or []:
|
|
942
|
+
if (isinstance(entry, dict) and entry.get("kind") == "session"
|
|
943
|
+
and entry.get("scope") == "shell"):
|
|
944
|
+
entry["ref"] = ordinals.get(entry.get("ref"))
|
|
945
|
+
|
|
946
|
+
|
|
549
947
|
def _item_blocks_with_rows(
|
|
550
948
|
item: dict, payloads: dict | None = None, *, preserve_marker_text: bool = False,
|
|
551
949
|
call_owner_count: dict | None = None, decompose_headings: bool = False,
|
|
950
|
+
session_ordinals: dict[str, str] | None = None,
|
|
552
951
|
) -> list[list]:
|
|
553
952
|
"""Assemble an item's blocks (the historical ``_build_item_blocks`` behaviour)
|
|
554
953
|
AND expose each block's underlying rows, so the detail renderer and the payload
|
|
@@ -597,11 +996,28 @@ def _item_blocks_with_rows(
|
|
|
597
996
|
text = kern._join_content_texts(payload.get("content"))
|
|
598
997
|
elif retained[0] == "event_msg":
|
|
599
998
|
text = kern._stringify(payload.get("message"))
|
|
999
|
+
if r.kind == "assistant":
|
|
1000
|
+
# #463 S3 section 5.5. Read-time detection over the row's own stored
|
|
1001
|
+
# text, so it reaches every historical marker with no payload load.
|
|
1002
|
+
# It is written to `external_call` and never to `markers`, which
|
|
1003
|
+
# selects export payload hydration.
|
|
1004
|
+
external = kern._external_call_from_text(text)
|
|
1005
|
+
# Fail closed on the span: it is published as offsets into the very
|
|
1006
|
+
# string served as `block["text"]`, and a span that does not resolve
|
|
1007
|
+
# would make the client hide the wrong run of prose. `text` is the
|
|
1008
|
+
# same object the block below carries, including the
|
|
1009
|
+
# `preserve_marker_text` replacement, so the check is against what is
|
|
1010
|
+
# actually served rather than against what was read.
|
|
1011
|
+
if external is not None and kern.external_call_span_resolves(
|
|
1012
|
+
text, external):
|
|
1013
|
+
detail = dict(detail) if isinstance(detail, dict) else {}
|
|
1014
|
+
detail["external_call"] = external
|
|
600
1015
|
if r.kind == "tool_call" and isinstance(payload, dict):
|
|
601
1016
|
card = kern.decode_tool_call_card(payload)
|
|
602
1017
|
if card is None:
|
|
603
1018
|
card = kern.decode_secondary_tool_call_card(payload)
|
|
604
1019
|
if card is not None:
|
|
1020
|
+
_apply_session_ordinals(card, session_ordinals or {})
|
|
605
1021
|
detail = dict(detail) if isinstance(detail, dict) else {}
|
|
606
1022
|
detail["card"] = card
|
|
607
1023
|
output_card = (kern.decode_tool_output_card(payload)
|
|
@@ -792,6 +1208,7 @@ def _item_blocks_with_rows(
|
|
|
792
1208
|
def _build_item_blocks(
|
|
793
1209
|
item: dict, payloads: dict | None = None, *, preserve_marker_text: bool = False,
|
|
794
1210
|
call_owner_count: dict | None = None, decompose_headings: bool = False,
|
|
1211
|
+
session_ordinals: dict[str, str] | None = None,
|
|
795
1212
|
) -> list[dict]:
|
|
796
1213
|
"""Assemble an item's blocks, folding each ``tool_output`` into its
|
|
797
1214
|
``tool_call`` block via ``call_id`` when that call_id has exactly one owner
|
|
@@ -800,7 +1217,8 @@ def _build_item_blocks(
|
|
|
800
1217
|
return [entry[0] for entry in _item_blocks_with_rows(
|
|
801
1218
|
item, payloads, preserve_marker_text=preserve_marker_text,
|
|
802
1219
|
call_owner_count=call_owner_count,
|
|
803
|
-
decompose_headings=decompose_headings
|
|
1220
|
+
decompose_headings=decompose_headings,
|
|
1221
|
+
session_ordinals=session_ordinals)]
|
|
804
1222
|
|
|
805
1223
|
|
|
806
1224
|
def _item_lifecycle(item: dict) -> dict | None:
|
|
@@ -912,9 +1330,32 @@ def _attribute_costs(conn: sqlite3.Connection, conversation_key: str, effective_
|
|
|
912
1330
|
return turn_cost, turn_tokens, unattr_cost, unattr_tokens, total, conv_tokens
|
|
913
1331
|
|
|
914
1332
|
|
|
1333
|
+
def _conversation_totals(
|
|
1334
|
+
conn: sqlite3.Connection, conversation_key: str, effective_speed: str,
|
|
1335
|
+
) -> tuple[float, dict]:
|
|
1336
|
+
"""Lean priced and token totals over one conversation's accounting rows.
|
|
1337
|
+
|
|
1338
|
+
Unlike ``_attribute_costs``, this does not reconstruct the event-to-turn map:
|
|
1339
|
+
callers that need only conversation totals (outline, browse, child summaries)
|
|
1340
|
+
can sum the compact accounting rows directly. The row order and pricing
|
|
1341
|
+
primitive stay identical to the detail envelope's whole-conversation pass.
|
|
1342
|
+
"""
|
|
1343
|
+
total = 0.0
|
|
1344
|
+
tokens = _zero_tokens()
|
|
1345
|
+
for model, inp, cin, out, rout in conn.execute(
|
|
1346
|
+
"SELECT model, input_tokens, cached_input_tokens, output_tokens, "
|
|
1347
|
+
"reasoning_output_tokens FROM codex_session_entries WHERE conversation_key = ? "
|
|
1348
|
+
"ORDER BY source_path, line_offset",
|
|
1349
|
+
(conversation_key,),
|
|
1350
|
+
):
|
|
1351
|
+
total += _calculate_codex_entry_cost(
|
|
1352
|
+
model or "", inp or 0, cin or 0, out or 0, rout or 0, speed=effective_speed)
|
|
1353
|
+
_add_tokens(tokens, inp, out, cin, rout)
|
|
1354
|
+
return total, _tokens_union(tokens)
|
|
1355
|
+
|
|
1356
|
+
|
|
915
1357
|
def _conversation_total_cost(conn: sqlite3.Connection, conversation_key: str, effective_speed: str) -> float:
|
|
916
|
-
"""Lean priced total
|
|
917
|
-
child summaries) — same primitive as ``_attribute_costs`` (§5.4)."""
|
|
1358
|
+
"""Lean priced total for browse rows and child summaries (§5.4)."""
|
|
918
1359
|
total = 0.0
|
|
919
1360
|
for model, inp, cin, out, rout in conn.execute(
|
|
920
1361
|
"SELECT model, input_tokens, cached_input_tokens, output_tokens, "
|
|
@@ -997,9 +1438,19 @@ def _short_native(native: str | None) -> str:
|
|
|
997
1438
|
|
|
998
1439
|
def _display_chain(fields: dict) -> str:
|
|
999
1440
|
"""Read-time display fallback (§4.3): stored title → project_label → short
|
|
1000
|
-
native-thread-id prefix.
|
|
1001
|
-
|
|
1002
|
-
|
|
1441
|
+
native-thread-id prefix.
|
|
1442
|
+
|
|
1443
|
+
#463 S4 §5 — the stored title is cleaned of recognized harness markup here.
|
|
1444
|
+
This is ONE of three read paths that need it, not the universal chokepoint
|
|
1445
|
+
the first draft assumed: the outline turn label is built independently from
|
|
1446
|
+
anchor-row text and the `kind=title` search path reads rollup titles
|
|
1447
|
+
directly, so both clean through the same helper rather than through this
|
|
1448
|
+
call. A construct that strips to nothing falls through the chain below on
|
|
1449
|
+
its own, which is what makes `strip` expressible at read time at all.
|
|
1450
|
+
"""
|
|
1451
|
+
return (clean_codex_title(fields.get("title"))
|
|
1452
|
+
or fields.get("project_label")
|
|
1453
|
+
or _short_native(fields.get("native_thread_id")) or "")
|
|
1003
1454
|
|
|
1004
1455
|
|
|
1005
1456
|
def _conversation_display_title(conn: sqlite3.Connection, conversation_key: str, rows: list | None = None) -> str:
|
|
@@ -1538,6 +1989,7 @@ def _fold_groups_for_item(item: dict, call_owner_count: dict,
|
|
|
1538
1989
|
def _build_segment_index(
|
|
1539
1990
|
conversation_key: str, items: list[dict], detail_bytes: dict, *,
|
|
1540
1991
|
segmented: bool, block_budget: int | None = None,
|
|
1992
|
+
fold_groups: bool = False,
|
|
1541
1993
|
) -> list[dict]:
|
|
1542
1994
|
"""Phase A's output: an ordered index of segments, with no block content.
|
|
1543
1995
|
|
|
@@ -1554,6 +2006,12 @@ def _build_segment_index(
|
|
|
1554
2006
|
when omitted, never as a default argument value: a default argument binds
|
|
1555
2007
|
once at import, so a test that lowered the budget would silently keep the
|
|
1556
2008
|
imported figure and pass vacuously.
|
|
2009
|
+
|
|
2010
|
+
``fold_groups`` publishes the fold-group membership as ``_fold_groups``. Only
|
|
2011
|
+
the outline reads it (#463 S4 — a ``tool_error`` landmark anchors on the call
|
|
2012
|
+
a failure folds into); the detail route S1 bounded and the search position
|
|
2013
|
+
map do not, and building it for them costs a tuple per row per request for a
|
|
2014
|
+
value nobody reads.
|
|
1557
2015
|
"""
|
|
1558
2016
|
index: list[dict] = []
|
|
1559
2017
|
for item_index, item in enumerate(items):
|
|
@@ -1605,6 +2063,16 @@ def _build_segment_index(
|
|
|
1605
2063
|
"_turn_id": item["turn_id"],
|
|
1606
2064
|
"_anchor_row": anchor,
|
|
1607
2065
|
"_rows": segment_rows,
|
|
2066
|
+
# #463 S4 — the fold-group membership, as physical positions.
|
|
2067
|
+
# `_fold_groups_for_item` computes it payload-free and this
|
|
2068
|
+
# index discarded it, so nothing downstream could say WHICH
|
|
2069
|
+
# `tool_call` a failing `tool_output` belongs to — only that the
|
|
2070
|
+
# segment contained one. A `tool_error` landmark anchors on the
|
|
2071
|
+
# call, so the membership has to survive Phase A.
|
|
2072
|
+
"_fold_groups": [
|
|
2073
|
+
[(row.source_path, row.line_offset) for row in group.rows]
|
|
2074
|
+
for group in segment.groups
|
|
2075
|
+
] if fold_groups else (),
|
|
1608
2076
|
"_call_owner_count": call_owner_count,
|
|
1609
2077
|
"_meta": _item_meta(item) if head else None,
|
|
1610
2078
|
"_lifecycle": _item_lifecycle(item) if head else None,
|
|
@@ -1690,6 +2158,11 @@ def get_codex_conversation(
|
|
|
1690
2158
|
# byte-identical to what the export golden already pins.
|
|
1691
2159
|
index = _build_segment_index(
|
|
1692
2160
|
conversation_key, items, detail_bytes, segmented=not legacy_export)
|
|
2161
|
+
# #463 S3 section 3.2. Built here, in Phase A, from the narrow index that is
|
|
2162
|
+
# already loaded for the whole conversation: it needs no extra payload read
|
|
2163
|
+
# and no extra column, and it must be whole-conversation because ordinals and
|
|
2164
|
+
# opener presence are global facts a page cannot decide.
|
|
2165
|
+
session_index, session_ordinals = _build_session_index(rows)
|
|
1693
2166
|
|
|
1694
2167
|
# ── Phase B: paginate the index ──────────────────────────────────────────
|
|
1695
2168
|
page_index, page = _paginate_items(
|
|
@@ -1761,7 +2234,8 @@ def get_codex_conversation(
|
|
|
1761
2234
|
# marker-bearing payloads, so populating `headings` there would
|
|
1762
2235
|
# force a whole-conversation payload read for a field the
|
|
1763
2236
|
# exporter never reads.
|
|
1764
|
-
decompose_headings=not legacy_export
|
|
2237
|
+
decompose_headings=not legacy_export,
|
|
2238
|
+
session_ordinals=session_ordinals),
|
|
1765
2239
|
"cost_usd": cost,
|
|
1766
2240
|
"tokens": tokens,
|
|
1767
2241
|
}
|
|
@@ -1782,6 +2256,7 @@ def get_codex_conversation(
|
|
|
1782
2256
|
"title": _conversation_display_title(conn, conversation_key),
|
|
1783
2257
|
"items": page_items,
|
|
1784
2258
|
"page": page,
|
|
2259
|
+
"session_index": session_index,
|
|
1785
2260
|
"children": _children_of(conn, conversation_key, effective_speed),
|
|
1786
2261
|
"parent": _parent_of(conn, conversation_key),
|
|
1787
2262
|
"total_cost_usd": total,
|
|
@@ -1793,15 +2268,299 @@ def get_codex_conversation(
|
|
|
1793
2268
|
# ── outline assembly (§5.6) ───────────────────────────────────────────────────
|
|
1794
2269
|
|
|
1795
2270
|
|
|
1796
|
-
def
|
|
1797
|
-
|
|
1798
|
-
|
|
1799
|
-
|
|
1800
|
-
|
|
1801
|
-
|
|
1802
|
-
|
|
1803
|
-
|
|
1804
|
-
|
|
2271
|
+
def _tool_call_name(row) -> str | None:
|
|
2272
|
+
"""The tool a ``tool_call`` row invoked, from its STORED detail.
|
|
2273
|
+
|
|
2274
|
+
Ingest writes ``{"name": payload["name"] or <record type>, …}`` for every
|
|
2275
|
+
tool call, so this needs no payload and stays outside the scoped pass.
|
|
2276
|
+
"""
|
|
2277
|
+
detail = _parse_detail(row.detail_json)
|
|
2278
|
+
name = detail.get("name") if isinstance(detail, dict) else None
|
|
2279
|
+
return name if isinstance(name, str) and name else None
|
|
2280
|
+
|
|
2281
|
+
|
|
2282
|
+
def _conversation_duration_seconds(rows) -> int | None:
|
|
2283
|
+
"""Wall span of the conversation, as MIN to MAX over row timestamps (§4.2).
|
|
2284
|
+
|
|
2285
|
+
Never last minus first. §3.4 records that ``timestamp_utc`` is monotone
|
|
2286
|
+
within a turn only — item and segment emission is physical order, not
|
|
2287
|
+
timestamp order — and Task 1 found five decreases across turns in the
|
|
2288
|
+
corpus, on which the naive form returns a negative duration.
|
|
2289
|
+
|
|
2290
|
+
The outline's own caller cannot exhibit that today, because
|
|
2291
|
+
``_load_conversation_rows`` reads ``ORDER BY timestamp_utc, source_path,
|
|
2292
|
+
line_offset`` and the two forms therefore coincide on the rows it passes.
|
|
2293
|
+
The rule is stated for the next caller, which is much more likely to hand
|
|
2294
|
+
over an item-anchor list.
|
|
2295
|
+
"""
|
|
2296
|
+
stamps = [row.timestamp_utc for row in rows if row.timestamp_utc]
|
|
2297
|
+
if not stamps:
|
|
2298
|
+
return None
|
|
2299
|
+
first = _parse_outline_ts(min(stamps))
|
|
2300
|
+
last = _parse_outline_ts(max(stamps))
|
|
2301
|
+
if first is None or last is None:
|
|
2302
|
+
return None
|
|
2303
|
+
return int((last - first).total_seconds())
|
|
2304
|
+
|
|
2305
|
+
|
|
2306
|
+
def _outline_error_count(
|
|
2307
|
+
failed_calls: set, outcome_positions: set,
|
|
2308
|
+
derivation: landmarks.EventDerivation,
|
|
2309
|
+
) -> int | None:
|
|
2310
|
+
"""How many calls failed, or ``None`` when that cannot be answered (D3).
|
|
2311
|
+
|
|
2312
|
+
Three states, because 0 and null are different claims. A conversation with
|
|
2313
|
+
no outcome-bearing row at all reports 0: nothing failed because nothing ran,
|
|
2314
|
+
and that is determinable. A conversation whose outcome rows produced no
|
|
2315
|
+
verdict at all — every retained payload gone, unparseable, or of a shape no
|
|
2316
|
+
decoder recognises — reports null, because the stored projection answers
|
|
2317
|
+
nothing here (Task 1 measured stored ``is_error`` true for 0 of 63,150
|
|
2318
|
+
production ``tool_output`` rows) and 0 would assert an absence nobody proved.
|
|
2319
|
+
|
|
2320
|
+
A PARTIAL read reports what it found rather than declining. The alternative
|
|
2321
|
+
would suppress real failures the pass did see because of one unreadable
|
|
2322
|
+
neighbour, which is the worse error of the two.
|
|
2323
|
+
"""
|
|
2324
|
+
if not outcome_positions:
|
|
2325
|
+
return 0
|
|
2326
|
+
if not (outcome_positions & set(derivation.errors_by_position)):
|
|
2327
|
+
return None
|
|
2328
|
+
return len(failed_calls)
|
|
2329
|
+
|
|
2330
|
+
|
|
2331
|
+
# The Codex tool names that open the `plan` landmark family. Codex's decoded
|
|
2332
|
+
# plan card is named `update_plan`, and both existing CLIENT plan predicates
|
|
2333
|
+
# recognise only Claude's `ExitPlanMode` and `AskUserQuestion` — so publishing
|
|
2334
|
+
# raw Codex tool names into tier-1 `tools` would NOT have made the plan jump
|
|
2335
|
+
# work, it would have been a silent no-op (§3.2). The mapping is explicit here
|
|
2336
|
+
# and the outline target derivation reads landmark KINDS rather than inferring
|
|
2337
|
+
# from names.
|
|
2338
|
+
_S4_PLAN_TOOLS = frozenset({"update_plan"})
|
|
2339
|
+
|
|
2340
|
+
|
|
2341
|
+
def _landmark_label(row) -> str:
|
|
2342
|
+
"""What a landmark row says. Never a raw provider identifier (§8).
|
|
2343
|
+
|
|
2344
|
+
§3.6 enumerates exactly TWO sources for a landmark label — reasoning heading
|
|
2345
|
+
text, or a tool name — and rests the decision not to scrub these labels on
|
|
2346
|
+
that enumeration. So a row that is neither a named ``tool_call`` nor a typed
|
|
2347
|
+
``event`` falls back to its own KIND, which is normalizer vocabulary, and
|
|
2348
|
+
never to the row's stored text.
|
|
2349
|
+
|
|
2350
|
+
That branch is reachable: a failing ``tool_output`` whose ``call_id`` is
|
|
2351
|
+
owned by two or more ``tool_call`` rows in its turn does not fold, becomes
|
|
2352
|
+
its own group head, and enters ``failed_calls`` directly. Its ``text`` column
|
|
2353
|
+
is the harness preamble that ``decode_tool_output_card(for_storage=False)``
|
|
2354
|
+
exists to remove, and ``test_s3_no_raw_session_id_reaches_any_served_route``
|
|
2355
|
+
documents that preamble as carrying the provider ``session_id``.
|
|
2356
|
+
"""
|
|
2357
|
+
if row.kind == "tool_call":
|
|
2358
|
+
name = _tool_call_name(row)
|
|
2359
|
+
if name is not None:
|
|
2360
|
+
return name
|
|
2361
|
+
elif row.kind == "event" and row.event_type:
|
|
2362
|
+
return row.event_type
|
|
2363
|
+
return row.kind or ""
|
|
2364
|
+
|
|
2365
|
+
|
|
2366
|
+
def _clean_outline_label(text: str) -> str:
|
|
2367
|
+
"""An outline turn label, cleaned, and never cleaned away to nothing (§5.1).
|
|
2368
|
+
|
|
2369
|
+
Unlike ``_display_chain``, which §5.3 leans on as "already a fallback chain"
|
|
2370
|
+
when it justifies the ``strip`` disposition, this path has no chain: it
|
|
2371
|
+
cleans the anchor row's first non-blank line and publishes the result. Two
|
|
2372
|
+
allowlisted grammars can consume the whole string — ``<recommended_plugins>``
|
|
2373
|
+
(6 of the census's 438 titles) and ``<command-name>`` with no sibling tag —
|
|
2374
|
+
and the client's ``cleanQualifiedTitle(turn.label) ?? turn.label`` passes an
|
|
2375
|
+
empty string straight through, so the reader would get a row with no text at
|
|
2376
|
+
all. The uncleaned line is the pre-S4 label, which is legible.
|
|
2377
|
+
"""
|
|
2378
|
+
cleaned = clean_codex_title(text)
|
|
2379
|
+
return cleaned if cleaned.strip() else text
|
|
2380
|
+
|
|
2381
|
+
|
|
2382
|
+
def _build_landmarks(
|
|
2383
|
+
index: list[dict], derivation: landmarks.EventDerivation,
|
|
2384
|
+
failed_calls: set,
|
|
2385
|
+
) -> list[dict]:
|
|
2386
|
+
"""Tier 2 — the landmarks a jump can reach (§3.2).
|
|
2387
|
+
|
|
2388
|
+
Three kinds and deliberately NOT one entry per tool call: a 523-call turn
|
|
2389
|
+
would contribute 523 rows, which is noise rather than navigation.
|
|
2390
|
+
|
|
2391
|
+
Emission is PHYSICAL order — the segment index in order, and each segment's
|
|
2392
|
+
rows in order — because §3.4 records that ``timestamp_utc`` is monotone
|
|
2393
|
+
within a turn only and no consumer sorts by it.
|
|
2394
|
+
|
|
2395
|
+
``landmark_key`` is always COMPOUND — ``<block_key>#<discriminator>`` — and
|
|
2396
|
+
unique across every kind. A reasoning heading discriminates by its zero-based
|
|
2397
|
+
ordinal, which is the identity the reader route already mints for the same
|
|
2398
|
+
heading, because one block yields several headings and ``block_key`` alone
|
|
2399
|
+
would collide. Every other kind discriminates by the kind itself.
|
|
2400
|
+
|
|
2401
|
+
That is what lets one ``tool_call`` block carry BOTH a ``tool_error`` and a
|
|
2402
|
+
``plan`` landmark. It has to: §3.2 gives the plan kind one entry per plan
|
|
2403
|
+
call, and a failed plan call is one, so filing it only as the error made the
|
|
2404
|
+
jump cluster's plan family report zero — asserting no plan activity in a
|
|
2405
|
+
conversation that has some, which is the claim the spec's own "0 is a claim,
|
|
2406
|
+
hiding is not" rule forbids. The error is emitted first, because it is the
|
|
2407
|
+
more urgent of the two claims about the same call.
|
|
2408
|
+
|
|
2409
|
+
A block carrying ``detail.external_call`` produces no landmark of any kind
|
|
2410
|
+
(§3.2). That holds by construction rather than by a filter here: the marker
|
|
2411
|
+
is published on ``assistant`` blocks only, and no kind below comes from an
|
|
2412
|
+
assistant row. ``test_external_call_block_produces_no_landmark`` pins it.
|
|
2413
|
+
"""
|
|
2414
|
+
out: list[dict] = []
|
|
2415
|
+
for entry in index:
|
|
2416
|
+
for row in entry["_rows"]:
|
|
2417
|
+
position = (row.source_path, row.line_offset)
|
|
2418
|
+
block_key = _block_key_for_row(row)
|
|
2419
|
+
common = {
|
|
2420
|
+
"block_key": block_key,
|
|
2421
|
+
"item_key": entry["item_key"],
|
|
2422
|
+
"parent_item_key": entry["turn_item_key"],
|
|
2423
|
+
"timestamp_utc": row.timestamp_utc,
|
|
2424
|
+
}
|
|
2425
|
+
if row.kind == "reasoning":
|
|
2426
|
+
for ordinal, text in enumerate(
|
|
2427
|
+
derivation.headings_by_position.get(position, ())):
|
|
2428
|
+
out.append({"landmark_key": f"{block_key}#{ordinal}",
|
|
2429
|
+
"kind": "reasoning", "label": text, **common})
|
|
2430
|
+
continue
|
|
2431
|
+
if position in failed_calls:
|
|
2432
|
+
out.append({"landmark_key": f"{block_key}#tool_error",
|
|
2433
|
+
"kind": "tool_error",
|
|
2434
|
+
"label": _landmark_label(row), **common})
|
|
2435
|
+
if (row.kind == "tool_call"
|
|
2436
|
+
and _tool_call_name(row) in _S4_PLAN_TOOLS):
|
|
2437
|
+
out.append({"landmark_key": f"{block_key}#plan", "kind": "plan",
|
|
2438
|
+
"label": _landmark_label(row), **common})
|
|
2439
|
+
return out
|
|
2440
|
+
|
|
2441
|
+
|
|
2442
|
+
# The literal ingest writes into `codex_conversation_file_touches.tool`, kept so
|
|
2443
|
+
# the outline wire field and the file-search projection mean the same thing.
|
|
2444
|
+
# Every touch S4 derives comes from a `patch_apply_end`, which is the completion
|
|
2445
|
+
# of an `apply_patch` call.
|
|
2446
|
+
_PATCH_TOUCH_TOOL = "apply_patch"
|
|
2447
|
+
|
|
2448
|
+
|
|
2449
|
+
def _conversation_files(
|
|
2450
|
+
segment_index: list[dict], derivation: landmarks.EventDerivation,
|
|
2451
|
+
) -> list[dict]:
|
|
2452
|
+
"""The whole-conversation file list, DERIVED read-time (§1.2, §4.3).
|
|
2453
|
+
|
|
2454
|
+
The stored ``codex_conversation_file_touches`` table is the source for
|
|
2455
|
+
cross-conversation ``kind=files`` search after #489 repaired dict-shaped
|
|
2456
|
+
ingest and backfilled retained history. It is deliberately not the outline
|
|
2457
|
+
source: this payload pass has the richer evidence the outline contract needs.
|
|
2458
|
+
|
|
2459
|
+
Deriving it instead buys three things the table could not have supplied: a
|
|
2460
|
+
real segment anchor per touch, so a file row jumps to its change rather than
|
|
2461
|
+
to the top of a turn; the true per-file count; and first-touch DOCUMENT
|
|
2462
|
+
order, which is what ``OutlineFile`` has always promised while the SQL
|
|
2463
|
+
ordered alphabetically by path.
|
|
2464
|
+
|
|
2465
|
+
``added``/``removed`` are summed over the touches, and go ``None`` as soon as
|
|
2466
|
+
ONE touch of that file cannot be counted. Summing only the countable touches
|
|
2467
|
+
would publish a number for a file that changed more — a file edited once with
|
|
2468
|
+
a real diff and then moved by a count-free ``update`` would report the first
|
|
2469
|
+
figure — and nothing in ``touches[]`` marks such a total as partial. §4.5 is
|
|
2470
|
+
explicit that an undeterminable count is null and the badge renders nothing
|
|
2471
|
+
rather than an understated number; that rule has to reach the aggregate, not
|
|
2472
|
+
only the individual touch. The per-touch counts themselves come from the
|
|
2473
|
+
UNBOUNDED raw ``changes`` entry (§4.5) — see ``landmarks.patch_file_touches``.
|
|
2474
|
+
"""
|
|
2475
|
+
files: dict[str, dict] = {}
|
|
2476
|
+
undetermined: dict[str, set[str]] = {}
|
|
2477
|
+
for entry in segment_index:
|
|
2478
|
+
for row in entry["_rows"]:
|
|
2479
|
+
position = (row.source_path, row.line_offset)
|
|
2480
|
+
for touch in derivation.patch_files_by_position.get(position, ()):
|
|
2481
|
+
record = files.get(touch["path"])
|
|
2482
|
+
if record is None:
|
|
2483
|
+
record = files[touch["path"]] = {
|
|
2484
|
+
"file_path": touch["path"], "tool": _PATCH_TOUCH_TOOL,
|
|
2485
|
+
"count": 0, "touches": [],
|
|
2486
|
+
"added": None, "removed": None}
|
|
2487
|
+
record["count"] += 1
|
|
2488
|
+
record["touches"].append({
|
|
2489
|
+
"item_key": entry["item_key"],
|
|
2490
|
+
"timestamp_utc": row.timestamp_utc,
|
|
2491
|
+
# The raw change KIND — `add`/`delete`/`update` from the dict
|
|
2492
|
+
# shape, `modified` from the list one — never the tool name.
|
|
2493
|
+
"op": touch["op"],
|
|
2494
|
+
})
|
|
2495
|
+
for field in ("added", "removed"):
|
|
2496
|
+
if touch[field] is None:
|
|
2497
|
+
undetermined.setdefault(touch["path"], set()).add(field)
|
|
2498
|
+
else:
|
|
2499
|
+
record[field] = (record[field] or 0) + touch[field]
|
|
2500
|
+
for path, fields in undetermined.items():
|
|
2501
|
+
for field in fields:
|
|
2502
|
+
files[path][field] = None
|
|
2503
|
+
return list(files.values())
|
|
2504
|
+
|
|
2505
|
+
|
|
2506
|
+
# Connections whose read snapshot THIS module opened, by identity. A
|
|
2507
|
+
# ``sqlite3.Connection`` supports neither attribute assignment nor a weak
|
|
2508
|
+
# reference, so ownership cannot be recorded on the object; ``id`` is unique
|
|
2509
|
+
# among live objects and the connection is alive for the whole ``with`` body, so
|
|
2510
|
+
# the token cannot be confused with another connection's while it is registered.
|
|
2511
|
+
_OWNED_READ_SNAPSHOTS: set[int] = set()
|
|
2512
|
+
|
|
2513
|
+
|
|
2514
|
+
@contextlib.contextmanager
|
|
2515
|
+
def _read_snapshot(conn: sqlite3.Connection):
|
|
2516
|
+
"""One consistent read snapshot across a multi-query envelope (#463 S4 §4.1).
|
|
2517
|
+
|
|
2518
|
+
The outline route uses one connection but opened no explicit read
|
|
2519
|
+
transaction, so its several queries each took their own snapshot. Concurrent
|
|
2520
|
+
APPEND is benign there — extra raw event rows have no normalized mapping yet
|
|
2521
|
+
— but a concurrent delete or truncation between the wide message read and the
|
|
2522
|
+
payload read can expose a message row whose payload is already gone, and the
|
|
2523
|
+
derivation would then report an absence that never existed.
|
|
2524
|
+
|
|
2525
|
+
A deferred ``BEGIN`` takes the snapshot on the first read and holds it for
|
|
2526
|
+
every later one. It is released with ``rollback``, which is the honest end of
|
|
2527
|
+
a transaction that wrote nothing.
|
|
2528
|
+
|
|
2529
|
+
**A transaction this module did not open is refused, not inherited.**
|
|
2530
|
+
``conn.in_transaction`` is true for an outer WRITE transaction exactly as it
|
|
2531
|
+
is for an outer read snapshot, and Python's ``sqlite3`` exposes no
|
|
2532
|
+
``txn_state``, so the two cannot be told apart here. Only one of them is safe
|
|
2533
|
+
to borrow: inside a write, the envelope would read that writer's uncommitted
|
|
2534
|
+
and possibly half-applied state — a message row whose events are already
|
|
2535
|
+
deleted — with no snapshot of its own and no way to notice. Treating both
|
|
2536
|
+
alike is silent; refusing is not. A caller that wants several envelopes on
|
|
2537
|
+
one snapshot opens it through this same helper, which nests without issuing
|
|
2538
|
+
the second ``BEGIN`` SQLite would refuse.
|
|
2539
|
+
|
|
2540
|
+
The caller sweep behind that decision, pinned by
|
|
2541
|
+
``test_every_outline_caller_arrives_outside_a_transaction``: the three call
|
|
2542
|
+
paths into ``get_codex_conversation_outline`` are
|
|
2543
|
+
``_lib_conversation_dispatch.neutral_outline`` (the dashboard route, on a
|
|
2544
|
+
connection ``open_conversations_db`` returns fresh per request and closes
|
|
2545
|
+
after), ``bin/build-codex-reader-fixtures.py``, and the tests. None holds a
|
|
2546
|
+
transaction at the call.
|
|
2547
|
+
"""
|
|
2548
|
+
token = id(conn)
|
|
2549
|
+
if token in _OWNED_READ_SNAPSHOTS:
|
|
2550
|
+
yield
|
|
2551
|
+
return
|
|
2552
|
+
if conn.in_transaction:
|
|
2553
|
+
raise RuntimeError(
|
|
2554
|
+
"this envelope needs its own read snapshot, and the connection is "
|
|
2555
|
+
"already inside a transaction it did not open; wrap the outer "
|
|
2556
|
+
"scope in _read_snapshot instead")
|
|
2557
|
+
conn.execute("BEGIN")
|
|
2558
|
+
_OWNED_READ_SNAPSHOTS.add(token)
|
|
2559
|
+
try:
|
|
2560
|
+
yield
|
|
2561
|
+
finally:
|
|
2562
|
+
_OWNED_READ_SNAPSHOTS.discard(token)
|
|
2563
|
+
conn.rollback()
|
|
1805
2564
|
|
|
1806
2565
|
|
|
1807
2566
|
def get_codex_conversation_outline(
|
|
@@ -1827,7 +2586,24 @@ def get_codex_conversation_outline(
|
|
|
1827
2586
|
report true for a segment that has not been fetched, so the drain would never
|
|
1828
2587
|
run and the jump would land nowhere. Membership for navigation and membership
|
|
1829
2588
|
for "this item subsumes that key" are different relations.
|
|
2589
|
+
|
|
2590
|
+
#463 S4 — the route now makes TWO conversation reads under one snapshot: the
|
|
2591
|
+
wide message read below, and a scoped read-time pass over the retained event
|
|
2592
|
+
payloads (``_derive_outline_events``) whose position set comes from that
|
|
2593
|
+
first read. The pass is what gives the outline a failure verdict per call,
|
|
2594
|
+
the authored reasoning headings, and the per-file patch touches; §1.2 records
|
|
2595
|
+
why the stored ``codex_conversation_file_touches`` search projection is not
|
|
2596
|
+
the OUTLINE source, and Task 1 measured that the stored card carries 0 of the
|
|
2597
|
+
corpus's 896 tool failures, so read-time is not a preference here.
|
|
1830
2598
|
"""
|
|
2599
|
+
with _read_snapshot(conn):
|
|
2600
|
+
return _outline_envelope(
|
|
2601
|
+
conn, conversation_key, effective_speed=effective_speed)
|
|
2602
|
+
|
|
2603
|
+
|
|
2604
|
+
def _outline_envelope(
|
|
2605
|
+
conn: sqlite3.Connection, conversation_key: str, *, effective_speed: str
|
|
2606
|
+
) -> dict:
|
|
1831
2607
|
if not codex_normalization_authoritative(conn):
|
|
1832
2608
|
return {"status": "normalization_pending", "conversation_key": conversation_key,
|
|
1833
2609
|
"turns": [], "files": [], "children": []}
|
|
@@ -1841,11 +2617,26 @@ def get_codex_conversation_outline(
|
|
|
1841
2617
|
kept, _suppressed = kern.pair_mirrors(rows)
|
|
1842
2618
|
items = kern.canonical_items(kept)
|
|
1843
2619
|
segment_keys: dict[int, list[str]] = {}
|
|
1844
|
-
|
|
1845
|
-
|
|
2620
|
+
fold_groups: list[list[tuple[str, int]]] = []
|
|
2621
|
+
segment_index = _build_segment_index(
|
|
2622
|
+
conversation_key, items, detail_bytes, segmented=True, fold_groups=True)
|
|
2623
|
+
for entry in segment_index:
|
|
1846
2624
|
segment_keys.setdefault(entry["_item_index"], []).append(entry["item_key"])
|
|
2625
|
+
fold_groups.extend(entry["_fold_groups"])
|
|
2626
|
+
# `rows` here is the wide read directly above, and that is what makes the
|
|
2627
|
+
# payload pass SCOPED rather than a second whole-conversation decode (§4.1).
|
|
2628
|
+
derivation = _derive_outline_events(conn, conversation_key, rows)
|
|
2629
|
+
call_by_position = landmarks.fold_owner_by_position(fold_groups)
|
|
2630
|
+
outcome_positions = _outline_outcome_positions(rows)
|
|
2631
|
+
# A failing outcome row is charged to the `tool_call` it folds into, so the
|
|
2632
|
+
# same failure cannot be counted twice when a call and its output both carry
|
|
2633
|
+
# one, and so a turn's `tools` entry can say WHICH call failed.
|
|
2634
|
+
failed_calls = _outline_failing_calls(
|
|
2635
|
+
derivation, outcome_positions, call_by_position)
|
|
1847
2636
|
turns: list[dict] = []
|
|
1848
2637
|
kind_totals: dict[str, int] = {}
|
|
2638
|
+
tool_counts: dict[str, int] = {}
|
|
2639
|
+
models: dict[str, int] = {}
|
|
1849
2640
|
# Keyed on the ITEM index, which is what _build_segment_index records. Using
|
|
1850
2641
|
# ``len(turns)`` would be correct only for as long as this loop appends a
|
|
1851
2642
|
# turn for every item without exception; a later ``continue`` would misalign
|
|
@@ -1857,11 +2648,38 @@ def get_codex_conversation_outline(
|
|
|
1857
2648
|
if meta is not None:
|
|
1858
2649
|
label = _META_LABEL_TEXT.get(meta["meta_label"], "Harness context")
|
|
1859
2650
|
else:
|
|
1860
|
-
|
|
2651
|
+
# Built from anchor-row TEXT, which is why it does not reach
|
|
2652
|
+
# `_display_chain` and has to clean through the shared helper here
|
|
2653
|
+
# (§5.1). A label with no recognized markup passes through byte for
|
|
2654
|
+
# byte, so this cannot move an ordinary prose label.
|
|
2655
|
+
label = _clean_outline_label(
|
|
2656
|
+
_first_nonblank_line(_strip_ansi(anchor_text))) if anchor_text else ""
|
|
1861
2657
|
kinds: dict[str, int] = {}
|
|
2658
|
+
tools: list[dict] = []
|
|
2659
|
+
tool_slot: dict[str | None, int] = {}
|
|
2660
|
+
tool_call_count = 0
|
|
2661
|
+
first_failure_name: str | None = None
|
|
2662
|
+
thinking: list[str] = []
|
|
1862
2663
|
for r in it["rows"]:
|
|
1863
2664
|
kinds[r.kind] = kinds.get(r.kind, 0) + 1
|
|
1864
2665
|
kind_totals[r.kind] = kind_totals.get(r.kind, 0) + 1
|
|
2666
|
+
position = (r.source_path, r.line_offset)
|
|
2667
|
+
if r.kind == "tool_call":
|
|
2668
|
+
tool_call_count += 1
|
|
2669
|
+
name = _tool_call_name(r)
|
|
2670
|
+
failed = position in failed_calls
|
|
2671
|
+
if failed and first_failure_name is None:
|
|
2672
|
+
first_failure_name = name
|
|
2673
|
+
if name is not None:
|
|
2674
|
+
tool_counts[name] = tool_counts.get(name, 0) + 1
|
|
2675
|
+
slot = tool_slot.get(name)
|
|
2676
|
+
if slot is None:
|
|
2677
|
+
tool_slot[name] = len(tools)
|
|
2678
|
+
tools.append({"name": name, "is_error": failed})
|
|
2679
|
+
elif failed:
|
|
2680
|
+
tools[slot]["is_error"] = True
|
|
2681
|
+
elif r.kind == "reasoning":
|
|
2682
|
+
thinking.extend(derivation.headings_by_position.get(position, ()))
|
|
1865
2683
|
item_key = _item_key_for_item(conversation_key, it)
|
|
1866
2684
|
turn = {
|
|
1867
2685
|
"item_key": item_key,
|
|
@@ -1871,15 +2689,43 @@ def get_codex_conversation_outline(
|
|
|
1871
2689
|
"timestamp_utc": it["anchor_row"].timestamp_utc,
|
|
1872
2690
|
"kinds": kinds,
|
|
1873
2691
|
}
|
|
2692
|
+
# Additive, and only where there is something to say: a turn with no
|
|
2693
|
+
# calls publishes neither an empty array nor a zero count, matching how
|
|
2694
|
+
# the Claude outline omits `tools` and `thinking`.
|
|
2695
|
+
if tools:
|
|
2696
|
+
turn["tools"] = tools
|
|
2697
|
+
turn["tool_call_count"] = tool_call_count
|
|
2698
|
+
turn["first_failure_name"] = first_failure_name
|
|
2699
|
+
if thinking:
|
|
2700
|
+
turn["thinking"] = thinking
|
|
2701
|
+
item_model = _item_model(it) if _item_kind(it) == "assistant" else None
|
|
2702
|
+
if item_model:
|
|
2703
|
+
turn["model"] = item_model
|
|
2704
|
+
models[item_model] = models.get(item_model, 0) + 1
|
|
1874
2705
|
if meta is not None:
|
|
1875
2706
|
turn.update(meta)
|
|
1876
2707
|
turns.append(turn)
|
|
2708
|
+
total_cost, tokens = _conversation_totals(
|
|
2709
|
+
conn, conversation_key, effective_speed)
|
|
1877
2710
|
return {
|
|
1878
2711
|
"status": "ok",
|
|
1879
2712
|
"conversation_key": conversation_key,
|
|
1880
2713
|
"turns": turns,
|
|
1881
|
-
|
|
1882
|
-
|
|
2714
|
+
# Tier 2, deliberately a SEPARATE array (§3.3): `adaptQualifiedOutline`
|
|
2715
|
+
# derives `stats.turns.{human,assistant,tool_result,meta}` by filtering
|
|
2716
|
+
# `turns[]` on kind, so putting landmarks there would inflate counts
|
|
2717
|
+
# meant to describe the conversation's structure.
|
|
2718
|
+
"landmarks": _build_landmarks(
|
|
2719
|
+
segment_index, derivation, failed_calls),
|
|
2720
|
+
"stats": {
|
|
2721
|
+
"items": len(items), "kinds": kind_totals,
|
|
2722
|
+
"cost_usd": total_cost, "tokens": tokens,
|
|
2723
|
+
"tool_counts": tool_counts, "models": models,
|
|
2724
|
+
"duration_seconds": _conversation_duration_seconds(rows),
|
|
2725
|
+
"error_count": _outline_error_count(
|
|
2726
|
+
failed_calls, outcome_positions, derivation),
|
|
2727
|
+
},
|
|
2728
|
+
"files": _conversation_files(segment_index, derivation),
|
|
1883
2729
|
"children": _children_of(conn, conversation_key, effective_speed),
|
|
1884
2730
|
}
|
|
1885
2731
|
|
|
@@ -1893,6 +2739,16 @@ def _is_fork(fields: dict) -> bool:
|
|
|
1893
2739
|
|
|
1894
2740
|
|
|
1895
2741
|
def _browse_row(conn: sqlite3.Connection, conversation_key: str, effective_speed: str, fields: dict) -> dict:
|
|
2742
|
+
return _browse_row_from_fields(
|
|
2743
|
+
conversation_key, fields,
|
|
2744
|
+
cost_usd=_conversation_total_cost(conn, conversation_key, effective_speed),
|
|
2745
|
+
parent=_parent_of(conn, conversation_key),
|
|
2746
|
+
)
|
|
2747
|
+
|
|
2748
|
+
|
|
2749
|
+
def _browse_row_from_fields(
|
|
2750
|
+
conversation_key: str, fields: dict, *, cost_usd: float, parent,
|
|
2751
|
+
) -> dict:
|
|
1896
2752
|
return {
|
|
1897
2753
|
"conversation_key": conversation_key,
|
|
1898
2754
|
"title": _display_chain(fields),
|
|
@@ -1901,9 +2757,9 @@ def _browse_row(conn: sqlite3.Connection, conversation_key: str, effective_speed
|
|
|
1901
2757
|
"started_utc": fields["started"],
|
|
1902
2758
|
"last_activity_utc": fields["last"],
|
|
1903
2759
|
"count": fields["item_count"],
|
|
1904
|
-
"cost_usd":
|
|
2760
|
+
"cost_usd": cost_usd,
|
|
1905
2761
|
"models": list(fields["models"]),
|
|
1906
|
-
"parent":
|
|
2762
|
+
"parent": parent,
|
|
1907
2763
|
"is_fork": _is_fork(fields),
|
|
1908
2764
|
}
|
|
1909
2765
|
|
|
@@ -1948,6 +2804,200 @@ def _paginate_rows(rows: list[dict], *, cursor: str | None, limit: int):
|
|
|
1948
2804
|
return window, page
|
|
1949
2805
|
|
|
1950
2806
|
|
|
2807
|
+
def _stored_rollups_present(conn: sqlite3.Connection) -> bool:
|
|
2808
|
+
"""Whether the authoritative stored-rollup branch has been materialized.
|
|
2809
|
+
|
|
2810
|
+
Normal writes update normalized rows and their rollups in one transaction.
|
|
2811
|
+
The only supported no-rollup state is the pre-first-recompute window, where
|
|
2812
|
+
the live branch keeps the rail available. This constant-time probe avoids
|
|
2813
|
+
reintroducing the whole-message-table DISTINCT scan on every cold browse.
|
|
2814
|
+
"""
|
|
2815
|
+
return conn.execute(
|
|
2816
|
+
"SELECT 1 FROM codex_conversation_rollups LIMIT 1").fetchone() is not None
|
|
2817
|
+
|
|
2818
|
+
|
|
2819
|
+
def _live_browse_fields(conn: sqlite3.Connection) -> list[tuple[str, dict]]:
|
|
2820
|
+
"""Live-recompute fallback used only while no stored rollups exist."""
|
|
2821
|
+
out = []
|
|
2822
|
+
for (conversation_key,) in conn.execute(
|
|
2823
|
+
"SELECT DISTINCT conversation_key FROM codex_conversation_messages"):
|
|
2824
|
+
fields = _rollup_fields(conn, conversation_key)
|
|
2825
|
+
if fields is not None:
|
|
2826
|
+
out.append((conversation_key, fields))
|
|
2827
|
+
return out
|
|
2828
|
+
|
|
2829
|
+
|
|
2830
|
+
def _facets_from_fields(fields_rows: list[tuple[str, dict]]) -> dict:
|
|
2831
|
+
return _browse_facets([
|
|
2832
|
+
{
|
|
2833
|
+
"project_key": fields["project_key"],
|
|
2834
|
+
"project_label": fields["project_label"],
|
|
2835
|
+
"models": list(fields["models"]),
|
|
2836
|
+
}
|
|
2837
|
+
for _conversation_key, fields in fields_rows
|
|
2838
|
+
])
|
|
2839
|
+
|
|
2840
|
+
|
|
2841
|
+
def _stored_browse_facets(conn: sqlite3.Connection) -> dict:
|
|
2842
|
+
fields_rows = []
|
|
2843
|
+
for conversation_key, project_key, project_label, models_json in conn.execute(
|
|
2844
|
+
"SELECT conversation_key, project_key, project_label, models_json "
|
|
2845
|
+
"FROM codex_conversation_rollups"
|
|
2846
|
+
):
|
|
2847
|
+
try:
|
|
2848
|
+
parsed = json.loads(models_json) if models_json else []
|
|
2849
|
+
models = parsed if isinstance(parsed, list) else []
|
|
2850
|
+
except (TypeError, json.JSONDecodeError):
|
|
2851
|
+
models = []
|
|
2852
|
+
fields_rows.append((conversation_key, {
|
|
2853
|
+
"project_key": project_key,
|
|
2854
|
+
"project_label": project_label,
|
|
2855
|
+
"models": models,
|
|
2856
|
+
}))
|
|
2857
|
+
return _facets_from_fields(fields_rows)
|
|
2858
|
+
|
|
2859
|
+
|
|
2860
|
+
def _stored_filter_sql(alias: str, project_key: str | None, model: str | None):
|
|
2861
|
+
clauses = []
|
|
2862
|
+
params = []
|
|
2863
|
+
if project_key is not None:
|
|
2864
|
+
clauses.append(f"{alias}.project_key = ?")
|
|
2865
|
+
params.append(project_key)
|
|
2866
|
+
if model is not None:
|
|
2867
|
+
# models_json is the writer's canonical JSON array. Searching for the
|
|
2868
|
+
# complete JSON string literal is exact and does not require JSON1.
|
|
2869
|
+
clauses.append(f"instr(COALESCE({alias}.models_json, ''), ?) > 0")
|
|
2870
|
+
params.append(json.dumps(model))
|
|
2871
|
+
return (" AND ".join(clauses) if clauses else "1"), params
|
|
2872
|
+
|
|
2873
|
+
|
|
2874
|
+
def _page_costs(
|
|
2875
|
+
conn: sqlite3.Connection, conversation_keys: list[str], effective_speed: str,
|
|
2876
|
+
) -> dict[str, float]:
|
|
2877
|
+
if not conversation_keys:
|
|
2878
|
+
return {}
|
|
2879
|
+
placeholders = ",".join("?" for _ in conversation_keys)
|
|
2880
|
+
totals = {key: 0.0 for key in conversation_keys}
|
|
2881
|
+
for ck, model, inp, cin, out, rout in conn.execute(
|
|
2882
|
+
"SELECT conversation_key, model, input_tokens, cached_input_tokens, "
|
|
2883
|
+
"output_tokens, reasoning_output_tokens FROM codex_session_entries "
|
|
2884
|
+
f"WHERE conversation_key IN ({placeholders}) "
|
|
2885
|
+
"ORDER BY conversation_key, id",
|
|
2886
|
+
conversation_keys,
|
|
2887
|
+
):
|
|
2888
|
+
totals[ck] += _calculate_codex_entry_cost(
|
|
2889
|
+
model or "", inp or 0, cin or 0, out or 0, rout or 0,
|
|
2890
|
+
speed=effective_speed)
|
|
2891
|
+
return totals
|
|
2892
|
+
|
|
2893
|
+
|
|
2894
|
+
def _stored_browse_page(
|
|
2895
|
+
conn: sqlite3.Connection, *, effective_speed: str,
|
|
2896
|
+
project_key: str | None, model: str | None, limit: int,
|
|
2897
|
+
cursor: str | None,
|
|
2898
|
+
):
|
|
2899
|
+
where_sql, filter_params = _stored_filter_sql("r", project_key, model)
|
|
2900
|
+
total = conn.execute(
|
|
2901
|
+
f"SELECT COUNT(*) FROM codex_conversation_rollups r WHERE {where_sql}",
|
|
2902
|
+
filter_params,
|
|
2903
|
+
).fetchone()[0]
|
|
2904
|
+
|
|
2905
|
+
cursor_row = None
|
|
2906
|
+
if cursor is not None:
|
|
2907
|
+
cursor_where, cursor_params = _stored_filter_sql("c", project_key, model)
|
|
2908
|
+
cursor_row = conn.execute(
|
|
2909
|
+
"SELECT COALESCE(c.last_activity_utc, ''), c.conversation_key "
|
|
2910
|
+
"FROM codex_conversation_rollups c "
|
|
2911
|
+
f"WHERE c.conversation_key = ? AND {cursor_where}",
|
|
2912
|
+
[cursor, *cursor_params],
|
|
2913
|
+
).fetchone()
|
|
2914
|
+
|
|
2915
|
+
page_where = [where_sql]
|
|
2916
|
+
page_params = list(filter_params)
|
|
2917
|
+
if cursor_row is not None:
|
|
2918
|
+
cursor_last, cursor_key = cursor_row
|
|
2919
|
+
page_where.append(
|
|
2920
|
+
"(COALESCE(r.last_activity_utc, '') < ? OR "
|
|
2921
|
+
"(COALESCE(r.last_activity_utc, '') = ? AND r.conversation_key < ?))")
|
|
2922
|
+
page_params.extend((cursor_last, cursor_last, cursor_key))
|
|
2923
|
+
|
|
2924
|
+
sql = (
|
|
2925
|
+
"SELECT r.conversation_key, r.item_count, r.started_utc, "
|
|
2926
|
+
"r.last_activity_utc, r.project_key, r.project_label, r.models_json, "
|
|
2927
|
+
"r.title, r.parent_thread_id, r.source_root_key, t.native_thread_id, "
|
|
2928
|
+
"pt.conversation_key, pr.title, pr.project_label, pt.native_thread_id "
|
|
2929
|
+
"FROM codex_conversation_rollups r "
|
|
2930
|
+
"LEFT JOIN codex_conversation_threads t "
|
|
2931
|
+
"ON t.conversation_key = r.conversation_key "
|
|
2932
|
+
"LEFT JOIN codex_conversation_threads pt "
|
|
2933
|
+
"ON pt.source_root_key = r.source_root_key "
|
|
2934
|
+
"AND pt.native_thread_id = r.parent_thread_id "
|
|
2935
|
+
"AND pt.conversation_key != r.conversation_key "
|
|
2936
|
+
"LEFT JOIN codex_conversation_rollups pr "
|
|
2937
|
+
"ON pr.conversation_key = pt.conversation_key "
|
|
2938
|
+
f"WHERE {' AND '.join(page_where)} "
|
|
2939
|
+
"ORDER BY COALESCE(r.last_activity_utc, '') DESC, r.conversation_key DESC"
|
|
2940
|
+
)
|
|
2941
|
+
if limit:
|
|
2942
|
+
sql += " LIMIT ?"
|
|
2943
|
+
page_params.append(limit + 1)
|
|
2944
|
+
raw_rows = list(conn.execute(sql, page_params))
|
|
2945
|
+
has_more = bool(limit and len(raw_rows) > limit)
|
|
2946
|
+
if has_more:
|
|
2947
|
+
raw_rows = raw_rows[:limit]
|
|
2948
|
+
|
|
2949
|
+
keys = [row[0] for row in raw_rows]
|
|
2950
|
+
costs = _page_costs(conn, keys, effective_speed)
|
|
2951
|
+
rows = []
|
|
2952
|
+
for row in raw_rows:
|
|
2953
|
+
(conversation_key, item_count, started, last, row_project_key,
|
|
2954
|
+
project_label, models_json, title, parent_thread_id, source_root_key,
|
|
2955
|
+
native_thread_id, parent_key, parent_title, parent_project_label,
|
|
2956
|
+
parent_native_thread_id) = row
|
|
2957
|
+
try:
|
|
2958
|
+
parsed = json.loads(models_json) if models_json else []
|
|
2959
|
+
models = parsed if isinstance(parsed, list) else []
|
|
2960
|
+
except (TypeError, json.JSONDecodeError):
|
|
2961
|
+
models = []
|
|
2962
|
+
fields = {
|
|
2963
|
+
"item_count": item_count,
|
|
2964
|
+
"started": started,
|
|
2965
|
+
"last": last,
|
|
2966
|
+
"project_key": row_project_key,
|
|
2967
|
+
"project_label": project_label,
|
|
2968
|
+
"models": models,
|
|
2969
|
+
"title": title,
|
|
2970
|
+
"parent_thread_id": parent_thread_id,
|
|
2971
|
+
"source_root_key": source_root_key,
|
|
2972
|
+
"native_thread_id": native_thread_id,
|
|
2973
|
+
}
|
|
2974
|
+
parent = None
|
|
2975
|
+
if parent_key is not None:
|
|
2976
|
+
parent = {
|
|
2977
|
+
"conversation_key": parent_key,
|
|
2978
|
+
"title": _display_chain({
|
|
2979
|
+
"title": parent_title,
|
|
2980
|
+
"project_label": parent_project_label,
|
|
2981
|
+
"native_thread_id": parent_native_thread_id,
|
|
2982
|
+
}),
|
|
2983
|
+
}
|
|
2984
|
+
rows.append(_browse_row_from_fields(
|
|
2985
|
+
conversation_key, fields, cost_usd=costs.get(conversation_key, 0.0),
|
|
2986
|
+
parent=parent))
|
|
2987
|
+
next_cursor = rows[-1]["conversation_key"] if (rows and has_more) else None
|
|
2988
|
+
return rows, {"total": total, "returned": len(rows), "cursor": next_cursor}
|
|
2989
|
+
|
|
2990
|
+
|
|
2991
|
+
def list_codex_conversation_facets(conn: sqlite3.Connection) -> dict:
|
|
2992
|
+
"""Facet-only browse projection; never builds or prices a discarded page."""
|
|
2993
|
+
if not codex_normalization_authoritative(conn):
|
|
2994
|
+
return {"status": "normalization_pending",
|
|
2995
|
+
"facets": {"projects": [], "models": []}}
|
|
2996
|
+
facets = (_stored_browse_facets(conn) if _stored_rollups_present(conn)
|
|
2997
|
+
else _facets_from_fields(_live_browse_fields(conn)))
|
|
2998
|
+
return {"status": "ok", "facets": facets}
|
|
2999
|
+
|
|
3000
|
+
|
|
1951
3001
|
def list_codex_conversations(
|
|
1952
3002
|
conn: sqlite3.Connection,
|
|
1953
3003
|
*,
|
|
@@ -1966,20 +3016,22 @@ def list_codex_conversations(
|
|
|
1966
3016
|
if not codex_normalization_authoritative(conn):
|
|
1967
3017
|
return {"status": "normalization_pending", "rows": [],
|
|
1968
3018
|
"facets": {"projects": [], "models": []}, "page": {"total": 0}}
|
|
1969
|
-
|
|
1970
|
-
|
|
1971
|
-
|
|
1972
|
-
|
|
1973
|
-
|
|
1974
|
-
|
|
1975
|
-
|
|
1976
|
-
|
|
1977
|
-
|
|
1978
|
-
|
|
1979
|
-
|
|
1980
|
-
if (project_key is None or row["project_key"] == project_key)
|
|
1981
|
-
and (model is None or model in (row["models"] or []))
|
|
3019
|
+
if _stored_rollups_present(conn):
|
|
3020
|
+
facets = _stored_browse_facets(conn)
|
|
3021
|
+
page_rows, page = _stored_browse_page(
|
|
3022
|
+
conn, effective_speed=effective_speed, project_key=project_key,
|
|
3023
|
+
model=model, limit=limit, cursor=cursor)
|
|
3024
|
+
return {"status": "ok", "rows": page_rows, "facets": facets, "page": page}
|
|
3025
|
+
|
|
3026
|
+
fields_rows = _live_browse_fields(conn)
|
|
3027
|
+
rows = [
|
|
3028
|
+
_browse_row(conn, conversation_key, effective_speed, fields)
|
|
3029
|
+
for conversation_key, fields in fields_rows
|
|
1982
3030
|
]
|
|
3031
|
+
facets = _facets_from_fields(fields_rows)
|
|
3032
|
+
filtered = [row for row in rows
|
|
3033
|
+
if (project_key is None or row["project_key"] == project_key)
|
|
3034
|
+
and (model is None or model in (row["models"] or []))]
|
|
1983
3035
|
filtered.sort(key=_recent_sort_key, reverse=True)
|
|
1984
3036
|
page_rows, page = _paginate_rows(filtered, cursor=cursor, limit=limit)
|
|
1985
3037
|
return {"status": "ok", "rows": page_rows, "facets": facets, "page": page}
|
|
@@ -2057,6 +3109,395 @@ def _pos_to_item_key(conn: sqlite3.Connection, conversation_key: str) -> dict:
|
|
|
2057
3109
|
return _pos_to_item_key_and_order(conn, conversation_key)[0]
|
|
2058
3110
|
|
|
2059
3111
|
|
|
3112
|
+
# ── #482 visible render-leaf projection ─────────────────────────────────────
|
|
3113
|
+
|
|
3114
|
+
CODEX_FIND_PROJECTION_VERSION = 1
|
|
3115
|
+
_COMPLETION_EVENT_TYPES = {
|
|
3116
|
+
"patch_apply_end",
|
|
3117
|
+
"web_search_end",
|
|
3118
|
+
"mcp_tool_call_end",
|
|
3119
|
+
}
|
|
3120
|
+
|
|
3121
|
+
|
|
3122
|
+
def _find_surface(row) -> str | None:
|
|
3123
|
+
if row.kind in {"user", "assistant", "reasoning"}:
|
|
3124
|
+
return "body"
|
|
3125
|
+
if row.kind == "tool_call":
|
|
3126
|
+
return "call"
|
|
3127
|
+
if row.kind == "tool_output":
|
|
3128
|
+
return "output"
|
|
3129
|
+
if row.kind == "event" and row.event_type in _COMPLETION_EVENT_TYPES:
|
|
3130
|
+
return "completion"
|
|
3131
|
+
return None
|
|
3132
|
+
|
|
3133
|
+
|
|
3134
|
+
def _project_plain_leaves(leaves: list[RenderLeaf]):
|
|
3135
|
+
"""Project structured-card leaves with a non-searchable visual boundary.
|
|
3136
|
+
|
|
3137
|
+
Native card fields render in separate block/inline containers. A newline
|
|
3138
|
+
between fields prevents a match from crossing that visual boundary while
|
|
3139
|
+
keeping each leaf's offsets local to the exact string its React component
|
|
3140
|
+
receives.
|
|
3141
|
+
"""
|
|
3142
|
+
text_parts: list[str] = []
|
|
3143
|
+
projected: list[ProjectedLeaf] = []
|
|
3144
|
+
cursor = 0
|
|
3145
|
+
for leaf in leaves:
|
|
3146
|
+
if not leaf.text:
|
|
3147
|
+
continue
|
|
3148
|
+
if text_parts:
|
|
3149
|
+
text_parts.append("\n")
|
|
3150
|
+
cursor += 1
|
|
3151
|
+
start = cursor
|
|
3152
|
+
text_parts.append(leaf.text)
|
|
3153
|
+
cursor += len(leaf.text)
|
|
3154
|
+
projected.append(ProjectedLeaf(leaf.key, start, cursor))
|
|
3155
|
+
return "".join(text_parts), tuple(projected)
|
|
3156
|
+
|
|
3157
|
+
|
|
3158
|
+
def _project_markdown_fields(fields: list[tuple[str, str]]):
|
|
3159
|
+
text_parts: list[str] = []
|
|
3160
|
+
leaves: list[ProjectedLeaf] = []
|
|
3161
|
+
cursor = 0
|
|
3162
|
+
for field, source in fields:
|
|
3163
|
+
if not source:
|
|
3164
|
+
continue
|
|
3165
|
+
if text_parts:
|
|
3166
|
+
text_parts.append("\n")
|
|
3167
|
+
cursor += 1
|
|
3168
|
+
projected_text, projected_leaves = project_markdown(source)
|
|
3169
|
+
text_parts.append(projected_text)
|
|
3170
|
+
leaves.extend(
|
|
3171
|
+
ProjectedLeaf(f"{field}/{leaf.key}", cursor + leaf.start, cursor + leaf.end)
|
|
3172
|
+
for leaf in projected_leaves
|
|
3173
|
+
)
|
|
3174
|
+
cursor += len(projected_text)
|
|
3175
|
+
return "".join(text_parts), tuple(leaves)
|
|
3176
|
+
|
|
3177
|
+
|
|
3178
|
+
def _patch_diff_leaves(files) -> list[RenderLeaf]:
|
|
3179
|
+
leaves: list[RenderLeaf] = []
|
|
3180
|
+
for file_index, file in enumerate(files or []):
|
|
3181
|
+
if not isinstance(file, dict):
|
|
3182
|
+
continue
|
|
3183
|
+
for field in ("path", "move_path"):
|
|
3184
|
+
value = file.get(field)
|
|
3185
|
+
if isinstance(value, str) and value:
|
|
3186
|
+
leaves.append(RenderLeaf(f"files.{file_index}.{field}", value))
|
|
3187
|
+
diff = file.get("unified_diff")
|
|
3188
|
+
if not isinstance(diff, str):
|
|
3189
|
+
continue
|
|
3190
|
+
hunk_index = -1
|
|
3191
|
+
row_index = 0
|
|
3192
|
+
for line in diff.replace("\r\n", "\n").replace("\r", "\n").split("\n"):
|
|
3193
|
+
if line.startswith("@@"):
|
|
3194
|
+
hunk_index += 1
|
|
3195
|
+
row_index = 0
|
|
3196
|
+
continue
|
|
3197
|
+
if hunk_index < 0 or not line or line.startswith(("--- ", "+++ ", "\\")):
|
|
3198
|
+
continue
|
|
3199
|
+
if line[0] not in {"+", "-", " "}:
|
|
3200
|
+
continue
|
|
3201
|
+
leaves.append(RenderLeaf(
|
|
3202
|
+
f"files.{file_index}.diff.{hunk_index}.{row_index}", line[1:]))
|
|
3203
|
+
row_index += 1
|
|
3204
|
+
return leaves
|
|
3205
|
+
|
|
3206
|
+
|
|
3207
|
+
def _json_card_text(value) -> str:
|
|
3208
|
+
return value if isinstance(value, str) else json.dumps(
|
|
3209
|
+
value, ensure_ascii=False, indent=2, separators=(",", ": "))
|
|
3210
|
+
|
|
3211
|
+
|
|
3212
|
+
def _project_completion_payload(payload: dict):
|
|
3213
|
+
patch = kern.decode_patch_event_card(payload)
|
|
3214
|
+
if patch is not None:
|
|
3215
|
+
leaves = _patch_diff_leaves(patch.get("files"))
|
|
3216
|
+
for field in ("stdout", "stderr"):
|
|
3217
|
+
value = patch.get(field)
|
|
3218
|
+
if isinstance(value, str) and value:
|
|
3219
|
+
leaves.append(RenderLeaf(field, _strip_ansi(value)))
|
|
3220
|
+
return _project_plain_leaves(leaves) if leaves else None
|
|
3221
|
+
|
|
3222
|
+
completion = kern.decode_secondary_event_card(payload)
|
|
3223
|
+
if completion is None:
|
|
3224
|
+
return None
|
|
3225
|
+
if completion.get("type") == "web_search_completion":
|
|
3226
|
+
leaves: list[RenderLeaf] = []
|
|
3227
|
+
for index, result in enumerate(completion.get("results") or []):
|
|
3228
|
+
if not isinstance(result, dict):
|
|
3229
|
+
continue
|
|
3230
|
+
for field in ("title", "domain", "snippet", "ref_id"):
|
|
3231
|
+
value = result.get(field)
|
|
3232
|
+
if isinstance(value, str) and value:
|
|
3233
|
+
leaves.append(RenderLeaf(f"results.{index}.{field}", value))
|
|
3234
|
+
error = completion.get("error")
|
|
3235
|
+
if error is not None:
|
|
3236
|
+
leaves.append(RenderLeaf("error", _json_card_text(error)))
|
|
3237
|
+
return _project_plain_leaves(leaves) if leaves else None
|
|
3238
|
+
if completion.get("type") == "mcp_completion":
|
|
3239
|
+
leaves = [
|
|
3240
|
+
RenderLeaf("arguments", _json_card_text(completion.get("arguments"))),
|
|
3241
|
+
RenderLeaf("result", _json_card_text(completion.get("result"))),
|
|
3242
|
+
]
|
|
3243
|
+
return _project_plain_leaves(leaves)
|
|
3244
|
+
return None
|
|
3245
|
+
|
|
3246
|
+
|
|
3247
|
+
def _project_find_row(row, *, payload: dict | None = None, block: dict | None = None):
|
|
3248
|
+
if row.kind == "event" and row.event_type in _COMPLETION_EVENT_TYPES and payload:
|
|
3249
|
+
completion = _project_completion_payload(payload)
|
|
3250
|
+
if completion is not None:
|
|
3251
|
+
return completion
|
|
3252
|
+
text = _row_display(row)
|
|
3253
|
+
if not text:
|
|
3254
|
+
return None
|
|
3255
|
+
if row.kind == "reasoning" and isinstance(block, dict):
|
|
3256
|
+
detail = block.get("detail")
|
|
3257
|
+
reasoning = detail.get("reasoning") if isinstance(detail, dict) else None
|
|
3258
|
+
if isinstance(reasoning, dict):
|
|
3259
|
+
visible_headings = block.get("_find_visible_headings")
|
|
3260
|
+
if isinstance(visible_headings, list):
|
|
3261
|
+
leaves = [
|
|
3262
|
+
RenderLeaf(leaf_key, text)
|
|
3263
|
+
for leaf_key, text in visible_headings
|
|
3264
|
+
if isinstance(leaf_key, str) and isinstance(text, str) and text
|
|
3265
|
+
]
|
|
3266
|
+
if leaves:
|
|
3267
|
+
return _project_plain_leaves(leaves)
|
|
3268
|
+
if reasoning.get("body") is None:
|
|
3269
|
+
return None
|
|
3270
|
+
fields = [
|
|
3271
|
+
(field, reasoning[field])
|
|
3272
|
+
for field in ("title", "summary", "body")
|
|
3273
|
+
if isinstance(reasoning.get(field), str) and reasoning[field]
|
|
3274
|
+
]
|
|
3275
|
+
if fields:
|
|
3276
|
+
return _project_markdown_fields(fields)
|
|
3277
|
+
if row.kind in {"user", "assistant", "reasoning"}:
|
|
3278
|
+
return project_markdown(text)
|
|
3279
|
+
if row.kind == "tool_output" and isinstance(block, dict):
|
|
3280
|
+
detail = block.get("detail")
|
|
3281
|
+
card = detail.get("card") if isinstance(detail, dict) else None
|
|
3282
|
+
if isinstance(card, dict) and card.get("type") == "terminal":
|
|
3283
|
+
output = card.get("output")
|
|
3284
|
+
parts = output.get("parts") if isinstance(output, dict) else None
|
|
3285
|
+
if isinstance(parts, list):
|
|
3286
|
+
stdout = "".join(
|
|
3287
|
+
part.get("text", "") for part in parts
|
|
3288
|
+
if isinstance(part, dict) and part.get("type") == "text"
|
|
3289
|
+
and part.get("stream") != "stderr"
|
|
3290
|
+
)
|
|
3291
|
+
stderr = "".join(
|
|
3292
|
+
part.get("text", "") for part in parts
|
|
3293
|
+
if isinstance(part, dict) and part.get("type") == "text"
|
|
3294
|
+
and part.get("stream") == "stderr"
|
|
3295
|
+
)
|
|
3296
|
+
leaves = []
|
|
3297
|
+
if stdout:
|
|
3298
|
+
leaves.append(RenderLeaf("stdout", _strip_ansi(stdout)))
|
|
3299
|
+
if stderr:
|
|
3300
|
+
leaves.append(RenderLeaf("stderr", _strip_ansi(stderr)))
|
|
3301
|
+
leaves.extend(
|
|
3302
|
+
RenderLeaf(f"raw.{index}", part["text"])
|
|
3303
|
+
for index, part in enumerate(parts)
|
|
3304
|
+
if isinstance(part, dict) and part.get("type") == "raw"
|
|
3305
|
+
and isinstance(part.get("text"), str) and part["text"]
|
|
3306
|
+
)
|
|
3307
|
+
if leaves:
|
|
3308
|
+
return _project_plain_leaves(leaves)
|
|
3309
|
+
if row.kind == "tool_call" and isinstance(block, dict):
|
|
3310
|
+
detail = block.get("detail")
|
|
3311
|
+
detail = detail if isinstance(detail, dict) else {}
|
|
3312
|
+
card = detail.get("card")
|
|
3313
|
+
if isinstance(card, dict):
|
|
3314
|
+
if card.get("type") == "patch":
|
|
3315
|
+
return None
|
|
3316
|
+
if card.get("type") == "web_search" and isinstance(card.get("query"), str):
|
|
3317
|
+
return project_plain((RenderLeaf("query", card["query"]),))
|
|
3318
|
+
if card.get("type") == "mcp":
|
|
3319
|
+
return None
|
|
3320
|
+
if card.get("type") == "terminal":
|
|
3321
|
+
commands = [
|
|
3322
|
+
RenderLeaf(f"commands.{index}", entry["command"])
|
|
3323
|
+
for index, entry in enumerate(card.get("commands") or [])
|
|
3324
|
+
if isinstance(entry, dict) and isinstance(entry.get("command"), str)
|
|
3325
|
+
]
|
|
3326
|
+
if commands:
|
|
3327
|
+
return _project_plain_leaves(commands)
|
|
3328
|
+
args = detail.get("args")
|
|
3329
|
+
if isinstance(args, str) and args:
|
|
3330
|
+
return project_plain((RenderLeaf("t0", args),))
|
|
3331
|
+
return project_plain((RenderLeaf("t0", text),))
|
|
3332
|
+
|
|
3333
|
+
|
|
3334
|
+
def materialize_codex_find_projection(
|
|
3335
|
+
conn: sqlite3.Connection,
|
|
3336
|
+
conversation_keys,
|
|
3337
|
+
) -> None:
|
|
3338
|
+
"""Replace #482 projection rows for the affected conversations.
|
|
3339
|
+
|
|
3340
|
+
The existing item/block builder is the only authority for native folds.
|
|
3341
|
+
Every searchable physical row keeps its own block key; a folded output or
|
|
3342
|
+
completion separately records the visual call block that owns it.
|
|
3343
|
+
"""
|
|
3344
|
+
keys = sorted({key for key in conversation_keys if key})
|
|
3345
|
+
if not keys:
|
|
3346
|
+
return
|
|
3347
|
+
for conversation_key in keys:
|
|
3348
|
+
conn.execute(
|
|
3349
|
+
"DELETE FROM codex_find_projection WHERE conversation_key=?",
|
|
3350
|
+
(conversation_key,),
|
|
3351
|
+
)
|
|
3352
|
+
rows = [
|
|
3353
|
+
kern.CodexNormalizedRow(*row)
|
|
3354
|
+
for row in conn.execute(
|
|
3355
|
+
"SELECT " + _ROW_COLS + " FROM codex_conversation_messages "
|
|
3356
|
+
"WHERE conversation_key=? "
|
|
3357
|
+
"ORDER BY timestamp_utc,source_path,line_offset",
|
|
3358
|
+
(conversation_key,),
|
|
3359
|
+
)
|
|
3360
|
+
]
|
|
3361
|
+
if not rows:
|
|
3362
|
+
continue
|
|
3363
|
+
kept, _suppressed = kern.pair_mirrors(rows)
|
|
3364
|
+
items = kern.canonical_items(kept)
|
|
3365
|
+
payloads = _load_row_payloads(conn, conversation_key)
|
|
3366
|
+
pos_to_item = _pos_to_item_key(conn, conversation_key)
|
|
3367
|
+
row_ids = {
|
|
3368
|
+
(source_path, line_offset): message_id
|
|
3369
|
+
for message_id, source_path, line_offset in conn.execute(
|
|
3370
|
+
"SELECT id,source_path,line_offset "
|
|
3371
|
+
"FROM codex_conversation_messages WHERE conversation_key=?",
|
|
3372
|
+
(conversation_key,),
|
|
3373
|
+
)
|
|
3374
|
+
}
|
|
3375
|
+
render_order = 0
|
|
3376
|
+
seen: set[tuple[str, int]] = set()
|
|
3377
|
+
seen_reasoning_by_turn: dict[str, set[str]] = {}
|
|
3378
|
+
|
|
3379
|
+
def store(row, *, container_block_key: str, block: dict | None = None) -> None:
|
|
3380
|
+
nonlocal render_order
|
|
3381
|
+
position = (row.source_path, row.line_offset)
|
|
3382
|
+
if position in seen:
|
|
3383
|
+
return
|
|
3384
|
+
surface = _find_surface(row)
|
|
3385
|
+
retained = _row_payload(row, payloads)
|
|
3386
|
+
payload = retained[1] if retained is not None else None
|
|
3387
|
+
projected = _project_find_row(row, payload=payload, block=block)
|
|
3388
|
+
message_id = row_ids.get(position)
|
|
3389
|
+
if surface is None or projected is None or message_id is None:
|
|
3390
|
+
return
|
|
3391
|
+
text, leaves = projected
|
|
3392
|
+
if not text:
|
|
3393
|
+
return
|
|
3394
|
+
physical_block_key = _block_key_for_row(row)
|
|
3395
|
+
item_key = pos_to_item.get(position)
|
|
3396
|
+
if item_key is None:
|
|
3397
|
+
return
|
|
3398
|
+
disclosure = (
|
|
3399
|
+
[container_block_key]
|
|
3400
|
+
if row.kind == "reasoning" or surface != "body"
|
|
3401
|
+
else []
|
|
3402
|
+
)
|
|
3403
|
+
conn.execute(
|
|
3404
|
+
"INSERT INTO codex_find_projection "
|
|
3405
|
+
"(message_id,conversation_key,item_key,block_key,"
|
|
3406
|
+
"container_block_key,surface,render_order,projected_text,"
|
|
3407
|
+
"leaves_json,disclosure_json,projection_version) "
|
|
3408
|
+
"VALUES (?,?,?,?,?,?,?,?,?,?,?)",
|
|
3409
|
+
(
|
|
3410
|
+
message_id,
|
|
3411
|
+
conversation_key,
|
|
3412
|
+
item_key,
|
|
3413
|
+
physical_block_key,
|
|
3414
|
+
container_block_key,
|
|
3415
|
+
surface,
|
|
3416
|
+
render_order,
|
|
3417
|
+
text,
|
|
3418
|
+
json.dumps(
|
|
3419
|
+
[
|
|
3420
|
+
{"key": leaf.key, "start": leaf.start, "end": leaf.end}
|
|
3421
|
+
for leaf in leaves
|
|
3422
|
+
],
|
|
3423
|
+
sort_keys=True,
|
|
3424
|
+
separators=(",", ":"),
|
|
3425
|
+
),
|
|
3426
|
+
json.dumps(disclosure, separators=(",", ":")),
|
|
3427
|
+
CODEX_FIND_PROJECTION_VERSION,
|
|
3428
|
+
),
|
|
3429
|
+
)
|
|
3430
|
+
seen.add(position)
|
|
3431
|
+
render_order += 1
|
|
3432
|
+
|
|
3433
|
+
for item in items:
|
|
3434
|
+
built_entries = _item_blocks_with_rows(
|
|
3435
|
+
item, payloads, decompose_headings=True,
|
|
3436
|
+
)
|
|
3437
|
+
completion_owner: dict[str, str] = {}
|
|
3438
|
+
for candidate_block, _candidate_primary, _candidate_output in built_entries:
|
|
3439
|
+
candidate_container = (
|
|
3440
|
+
candidate_block.get("block_key")
|
|
3441
|
+
or _block_key_for_row(_candidate_primary)
|
|
3442
|
+
)
|
|
3443
|
+
candidate_detail = candidate_block.get("detail")
|
|
3444
|
+
candidate_card = (
|
|
3445
|
+
candidate_detail.get("card")
|
|
3446
|
+
if isinstance(candidate_detail, dict) else None
|
|
3447
|
+
)
|
|
3448
|
+
completion = (
|
|
3449
|
+
candidate_card.get("completion")
|
|
3450
|
+
if isinstance(candidate_card, dict) else None
|
|
3451
|
+
)
|
|
3452
|
+
event_key = (
|
|
3453
|
+
completion.get("event_block_key")
|
|
3454
|
+
if isinstance(completion, dict) else None
|
|
3455
|
+
)
|
|
3456
|
+
if isinstance(event_key, str):
|
|
3457
|
+
completion_owner[event_key] = candidate_container
|
|
3458
|
+
|
|
3459
|
+
for block, primary, output in built_entries:
|
|
3460
|
+
container = block.get("block_key") or _block_key_for_row(primary)
|
|
3461
|
+
container = completion_owner.get(_block_key_for_row(primary), container)
|
|
3462
|
+
if primary.kind == "reasoning":
|
|
3463
|
+
detail = block.get("detail")
|
|
3464
|
+
reasoning = (
|
|
3465
|
+
detail.get("reasoning") if isinstance(detail, dict) else None
|
|
3466
|
+
)
|
|
3467
|
+
headings = (
|
|
3468
|
+
reasoning.get("headings")
|
|
3469
|
+
if isinstance(reasoning, dict) else None
|
|
3470
|
+
)
|
|
3471
|
+
if isinstance(headings, list):
|
|
3472
|
+
turn_key = primary.turn_id or item.get("turn_id") or ""
|
|
3473
|
+
prior = seen_reasoning_by_turn.setdefault(turn_key, set())
|
|
3474
|
+
visible = []
|
|
3475
|
+
for heading_index, heading in enumerate(headings):
|
|
3476
|
+
text = heading.get("text") if isinstance(heading, dict) else None
|
|
3477
|
+
if not isinstance(text, str) or text in prior:
|
|
3478
|
+
continue
|
|
3479
|
+
prior.add(text)
|
|
3480
|
+
visible.append((f"headings.{heading_index}", text))
|
|
3481
|
+
block["_find_visible_headings"] = visible
|
|
3482
|
+
store(primary, container_block_key=container, block=block)
|
|
3483
|
+
if output is not None:
|
|
3484
|
+
store(output, container_block_key=container, block=block)
|
|
3485
|
+
# Completion folds that intentionally produce no standalone block
|
|
3486
|
+
# still own a searchable physical surface and point at their call.
|
|
3487
|
+
for row in item["rows"]:
|
|
3488
|
+
if row.event_type not in _COMPLETION_EVENT_TYPES:
|
|
3489
|
+
continue
|
|
3490
|
+
physical_key = _block_key_for_row(row)
|
|
3491
|
+
container = completion_owner.get(physical_key, physical_key)
|
|
3492
|
+
store(row, container_block_key=container)
|
|
3493
|
+
|
|
3494
|
+
conn.execute(
|
|
3495
|
+
"INSERT INTO cache_meta(key,value) VALUES"
|
|
3496
|
+
"('codex_find_projection_generation','1') "
|
|
3497
|
+
"ON CONFLICT(key) DO UPDATE SET value=CAST(value AS INTEGER)+1"
|
|
3498
|
+
)
|
|
3499
|
+
|
|
3500
|
+
|
|
2060
3501
|
def _fts_query(query: str, column: str | None) -> str:
|
|
2061
3502
|
"""A safe FTS5 query: each whitespace term becomes a quoted phrase, joined by
|
|
2062
3503
|
implicit AND (term-wise AND — the documented divergence from LIKE's single
|
|
@@ -2120,7 +3561,60 @@ def _excerpt(text: str | None) -> str:
|
|
|
2120
3561
|
return collapsed[:200]
|
|
2121
3562
|
|
|
2122
3563
|
|
|
2123
|
-
def
|
|
3564
|
+
def _search_display_text(text: str | None) -> str:
|
|
3565
|
+
"""Readable search projection for retained structured content arrays.
|
|
3566
|
+
|
|
3567
|
+
Tool outputs must retain their provider JSON in ``search_tool`` so every
|
|
3568
|
+
leaf stays searchable. The rail, however, needs the same ordered text
|
|
3569
|
+
leaves a reader sees—not the serialized wrapper. Unknown JSON and future
|
|
3570
|
+
shapes fall back byte-for-byte to the retained string.
|
|
3571
|
+
"""
|
|
3572
|
+
if not text:
|
|
3573
|
+
return ""
|
|
3574
|
+
raw = str(text)
|
|
3575
|
+
if not raw.lstrip().startswith("["):
|
|
3576
|
+
return raw
|
|
3577
|
+
try:
|
|
3578
|
+
parsed = json.loads(raw)
|
|
3579
|
+
except (TypeError, json.JSONDecodeError):
|
|
3580
|
+
# Search columns are capped. A large content array can therefore end
|
|
3581
|
+
# mid-string and cease to be valid JSON even though one or more leading
|
|
3582
|
+
# text parts are complete. `_canonical_json` sorts object keys, so a
|
|
3583
|
+
# text-bearing part starts as `{\"text\":...}`. Decode only those
|
|
3584
|
+
# complete JSON string literals; never regex-unescape provider bytes.
|
|
3585
|
+
parts = []
|
|
3586
|
+
for match in re.finditer(r'(?:\A\[\{|,\{)"text":', raw):
|
|
3587
|
+
try:
|
|
3588
|
+
value, _end = json.JSONDecoder().raw_decode(raw, match.end())
|
|
3589
|
+
except (TypeError, json.JSONDecodeError):
|
|
3590
|
+
continue
|
|
3591
|
+
if isinstance(value, str) and value:
|
|
3592
|
+
parts.append(value)
|
|
3593
|
+
return "\n".join(parts) + ("\n…" if parts else "") or raw
|
|
3594
|
+
joined = kern._join_content_texts(parsed)
|
|
3595
|
+
return joined if joined else raw
|
|
3596
|
+
|
|
3597
|
+
|
|
3598
|
+
def _search_excerpt(text: str | None, query: str, width: int = 200) -> str:
|
|
3599
|
+
"""Whitespace-collapsed, match-centred excerpt from readable search text."""
|
|
3600
|
+
collapsed = " ".join(_search_display_text(text).split())
|
|
3601
|
+
if not collapsed:
|
|
3602
|
+
return ""
|
|
3603
|
+
needle = " ".join(query.split())
|
|
3604
|
+
found = collapsed.casefold().find(needle.casefold()) if needle else -1
|
|
3605
|
+
if found < 0 or len(collapsed) <= width:
|
|
3606
|
+
return collapsed[:width]
|
|
3607
|
+
start = max(0, found - (width // 3))
|
|
3608
|
+
end = min(len(collapsed), start + width)
|
|
3609
|
+
start = max(0, end - width)
|
|
3610
|
+
excerpt = collapsed[start:end]
|
|
3611
|
+
return (("… " if start else "") + excerpt
|
|
3612
|
+
+ (" …" if end < len(collapsed) else ""))
|
|
3613
|
+
|
|
3614
|
+
|
|
3615
|
+
def _collapse_message_hits(
|
|
3616
|
+
conn: sqlite3.Connection, matched_rows: list, query: str,
|
|
3617
|
+
) -> list[dict]:
|
|
2124
3618
|
"""Collapse matched physical rows to canonical ``item_key`` BEFORE totals /
|
|
2125
3619
|
badges (§6.2) — both members of a mirror pair map to one item_key, so mirror
|
|
2126
3620
|
rows never double-count (turned or unturned)."""
|
|
@@ -2143,7 +3637,7 @@ def _collapse_message_hits(conn: sqlite3.Connection, matched_rows: list) -> list
|
|
|
2143
3637
|
"last_activity_utc": last_act, "project_label": project_label})
|
|
2144
3638
|
hit["_badges"].add(_badge_for_kind(kind))
|
|
2145
3639
|
if hit["snippet"] is None:
|
|
2146
|
-
hit["snippet"] =
|
|
3640
|
+
hit["snippet"] = _search_excerpt(disp, query)
|
|
2147
3641
|
return [
|
|
2148
3642
|
{"conversation_key": h["conversation_key"], "item_key": h["item_key"],
|
|
2149
3643
|
"title": h["title"], "snippet": h["snippet"], "badges": sorted(h["_badges"]),
|
|
@@ -2154,22 +3648,31 @@ def _collapse_message_hits(conn: sqlite3.Connection, matched_rows: list) -> list
|
|
|
2154
3648
|
|
|
2155
3649
|
def _search_title(conn: sqlite3.Connection, query: str) -> list[dict]:
|
|
2156
3650
|
"""Title search over the rollup table — identical LIKE semantics in both FTS
|
|
2157
|
-
and LIKE modes (§6.2). Conversation-level hits (no item anchor).
|
|
3651
|
+
and LIKE modes (§6.2). Conversation-level hits (no item anchor).
|
|
3652
|
+
|
|
3653
|
+
#463 S4 §5.1 — the third read path that needs cleaning, and the one that is
|
|
3654
|
+
user-facing on the CLI: `cctally transcript search --source codex
|
|
3655
|
+
--kind title` prints this `snippet` and emits this `title` in its JSON. The
|
|
3656
|
+
MATCH still runs against the stored value, so a query that names markup
|
|
3657
|
+
still finds its conversation; only what is shown is cleaned.
|
|
3658
|
+
"""
|
|
2158
3659
|
like = f"%{query}%"
|
|
2159
3660
|
hits = []
|
|
2160
3661
|
for ck, title, last_act, project_label in conn.execute(
|
|
2161
3662
|
"SELECT conversation_key, title, last_activity_utc, project_label "
|
|
2162
3663
|
"FROM codex_conversation_rollups WHERE title LIKE ?", (like,)):
|
|
3664
|
+
cleaned = clean_codex_title(title)
|
|
2163
3665
|
hits.append(
|
|
2164
|
-
{"conversation_key": ck, "item_key": None, "title":
|
|
2165
|
-
"snippet": _excerpt(
|
|
3666
|
+
{"conversation_key": ck, "item_key": None, "title": cleaned,
|
|
3667
|
+
"snippet": _excerpt(cleaned), "badges": ["title"],
|
|
2166
3668
|
"last_activity_utc": last_act, "project_label": project_label})
|
|
2167
3669
|
return hits
|
|
2168
3670
|
|
|
2169
3671
|
|
|
2170
3672
|
def _search_files(conn: sqlite3.Connection, query: str) -> list[dict]:
|
|
2171
3673
|
"""File-touch search — matches file paths, collapsed to the owning message's
|
|
2172
|
-
canonical item_key (§6.2).
|
|
3674
|
+
canonical item_key (§6.2). ``message_id`` is an application-level link, so
|
|
3675
|
+
an orphan is skipped and cannot suppress valid rows."""
|
|
2173
3676
|
like = f"%{query}%"
|
|
2174
3677
|
pos_cache: dict[str, dict] = {}
|
|
2175
3678
|
fields_cache: dict[str, tuple] = {}
|
|
@@ -2244,7 +3747,7 @@ def search_codex_conversations(
|
|
|
2244
3747
|
hits = _search_files(conn, query)
|
|
2245
3748
|
else:
|
|
2246
3749
|
hits = _collapse_message_hits(
|
|
2247
|
-
conn, _matched_message_rows(conn, query, kind, mode))
|
|
3750
|
+
conn, _matched_message_rows(conn, query, kind, mode), query)
|
|
2248
3751
|
hits.sort(key=lambda h: (h["conversation_key"], h["item_key"] or ""))
|
|
2249
3752
|
total = len(hits)
|
|
2250
3753
|
page_hits, page = _paginate_hits(hits, cursor=cursor, limit=limit)
|
|
@@ -2256,6 +3759,402 @@ def search_codex_conversations(
|
|
|
2256
3759
|
|
|
2257
3760
|
# ── in-conversation find (§3.1) ───────────────────────────────────────────────
|
|
2258
3761
|
|
|
3762
|
+
_CODEX_EXACT_FIND_SCHEMA_VERSION = 2
|
|
3763
|
+
_CODEX_EXACT_FIND_DEFAULT_LIMIT = 100
|
|
3764
|
+
_CODEX_EXACT_FIND_MAX_LIMIT = 200
|
|
3765
|
+
_CODEX_EXACT_FIND_CURSOR_PREFIX = "ofc1."
|
|
3766
|
+
_CODEX_EXACT_FIND_QUERY_DOMAIN = b"cctally-codex-find-query-v1\0"
|
|
3767
|
+
_CODEX_EXACT_FIND_OCCURRENCE_DOMAIN = b"cctally-codex-find-occurrence-v1\0"
|
|
3768
|
+
|
|
3769
|
+
|
|
3770
|
+
class InvalidFindCursor(ValueError):
|
|
3771
|
+
"""The external exact-find cursor is malformed."""
|
|
3772
|
+
|
|
3773
|
+
|
|
3774
|
+
class StaleFindCursor(ValueError):
|
|
3775
|
+
"""The exact-find cursor belongs to another query or projection generation."""
|
|
3776
|
+
|
|
3777
|
+
|
|
3778
|
+
def _exact_find_query_id(
|
|
3779
|
+
query: str, *, regex: bool, case_sensitive: bool, kind: str
|
|
3780
|
+
) -> str:
|
|
3781
|
+
payload = json.dumps(
|
|
3782
|
+
{
|
|
3783
|
+
"case": case_sensitive,
|
|
3784
|
+
"kind": kind,
|
|
3785
|
+
"projection": CODEX_FIND_PROJECTION_VERSION,
|
|
3786
|
+
"query": query,
|
|
3787
|
+
"regex": regex,
|
|
3788
|
+
},
|
|
3789
|
+
ensure_ascii=False,
|
|
3790
|
+
sort_keys=True,
|
|
3791
|
+
separators=(",", ":"),
|
|
3792
|
+
).encode("utf-8")
|
|
3793
|
+
return hashlib.sha256(_CODEX_EXACT_FIND_QUERY_DOMAIN + payload).hexdigest()
|
|
3794
|
+
|
|
3795
|
+
|
|
3796
|
+
def _exact_find_occurrence_id(
|
|
3797
|
+
query_id: str,
|
|
3798
|
+
*,
|
|
3799
|
+
block_key: str,
|
|
3800
|
+
surface: str,
|
|
3801
|
+
ordinal: int,
|
|
3802
|
+
start: int,
|
|
3803
|
+
end: int,
|
|
3804
|
+
) -> str:
|
|
3805
|
+
payload = json.dumps(
|
|
3806
|
+
[
|
|
3807
|
+
CODEX_FIND_PROJECTION_VERSION,
|
|
3808
|
+
query_id,
|
|
3809
|
+
block_key,
|
|
3810
|
+
surface,
|
|
3811
|
+
ordinal,
|
|
3812
|
+
start,
|
|
3813
|
+
end,
|
|
3814
|
+
],
|
|
3815
|
+
ensure_ascii=False,
|
|
3816
|
+
separators=(",", ":"),
|
|
3817
|
+
).encode("utf-8")
|
|
3818
|
+
digest = hashlib.sha256(_CODEX_EXACT_FIND_OCCURRENCE_DOMAIN + payload).digest()
|
|
3819
|
+
return "o1." + base64.urlsafe_b64encode(digest).decode("ascii").rstrip("=")
|
|
3820
|
+
|
|
3821
|
+
|
|
3822
|
+
def _encode_exact_find_cursor(
|
|
3823
|
+
*,
|
|
3824
|
+
query_id: str,
|
|
3825
|
+
generation: int,
|
|
3826
|
+
start_index: int,
|
|
3827
|
+
direction: str,
|
|
3828
|
+
boundary: tuple[int, int, str, int],
|
|
3829
|
+
) -> str:
|
|
3830
|
+
payload = json.dumps(
|
|
3831
|
+
{
|
|
3832
|
+
"b": list(boundary),
|
|
3833
|
+
"d": direction,
|
|
3834
|
+
"g": generation,
|
|
3835
|
+
"i": start_index,
|
|
3836
|
+
"q": query_id,
|
|
3837
|
+
"v": CODEX_FIND_PROJECTION_VERSION,
|
|
3838
|
+
},
|
|
3839
|
+
sort_keys=True,
|
|
3840
|
+
separators=(",", ":"),
|
|
3841
|
+
).encode("utf-8")
|
|
3842
|
+
return _CODEX_EXACT_FIND_CURSOR_PREFIX + base64.urlsafe_b64encode(
|
|
3843
|
+
payload
|
|
3844
|
+
).decode("ascii").rstrip("=")
|
|
3845
|
+
|
|
3846
|
+
|
|
3847
|
+
def _decode_exact_find_cursor(cursor: str) -> dict[str, object]:
|
|
3848
|
+
if not isinstance(cursor, str) or not cursor.startswith(
|
|
3849
|
+
_CODEX_EXACT_FIND_CURSOR_PREFIX
|
|
3850
|
+
):
|
|
3851
|
+
raise InvalidFindCursor(cursor)
|
|
3852
|
+
encoded = cursor[len(_CODEX_EXACT_FIND_CURSOR_PREFIX):]
|
|
3853
|
+
try:
|
|
3854
|
+
raw = base64.urlsafe_b64decode(encoded + "=" * (-len(encoded) % 4))
|
|
3855
|
+
canonical = base64.urlsafe_b64encode(raw).decode("ascii").rstrip("=")
|
|
3856
|
+
if canonical != encoded:
|
|
3857
|
+
raise InvalidFindCursor(cursor)
|
|
3858
|
+
payload = json.loads(raw.decode("utf-8"))
|
|
3859
|
+
except (binascii.Error, ValueError, TypeError, UnicodeDecodeError, json.JSONDecodeError):
|
|
3860
|
+
raise InvalidFindCursor(cursor) from None
|
|
3861
|
+
if not isinstance(payload, dict) or set(payload) != {"b", "d", "g", "i", "q", "v"}:
|
|
3862
|
+
raise InvalidFindCursor(cursor)
|
|
3863
|
+
boundary = payload.get("b")
|
|
3864
|
+
if (
|
|
3865
|
+
payload.get("d") not in {"next", "previous"}
|
|
3866
|
+
or type(payload.get("g")) is not int
|
|
3867
|
+
or type(payload.get("i")) is not int
|
|
3868
|
+
or payload["i"] < 0
|
|
3869
|
+
or not isinstance(payload.get("q"), str)
|
|
3870
|
+
or payload.get("v") != CODEX_FIND_PROJECTION_VERSION
|
|
3871
|
+
or not isinstance(boundary, list)
|
|
3872
|
+
or len(boundary) != 4
|
|
3873
|
+
or type(boundary[0]) is not int
|
|
3874
|
+
or type(boundary[1]) is not int
|
|
3875
|
+
or not isinstance(boundary[2], str)
|
|
3876
|
+
or type(boundary[3]) is not int
|
|
3877
|
+
):
|
|
3878
|
+
raise InvalidFindCursor(cursor)
|
|
3879
|
+
return payload
|
|
3880
|
+
|
|
3881
|
+
|
|
3882
|
+
def _exact_find_base(
|
|
3883
|
+
query_id: str,
|
|
3884
|
+
*,
|
|
3885
|
+
status: str,
|
|
3886
|
+
regex: bool,
|
|
3887
|
+
kind: str,
|
|
3888
|
+
) -> dict[str, object]:
|
|
3889
|
+
return {
|
|
3890
|
+
"schema_version": _CODEX_EXACT_FIND_SCHEMA_VERSION,
|
|
3891
|
+
"semantics": "occurrence",
|
|
3892
|
+
"status": status,
|
|
3893
|
+
"query_id": query_id,
|
|
3894
|
+
"selection_stale": False,
|
|
3895
|
+
"mode": "regex" if regex else "literal",
|
|
3896
|
+
"kind": kind,
|
|
3897
|
+
"search_depth": "full",
|
|
3898
|
+
}
|
|
3899
|
+
|
|
3900
|
+
|
|
3901
|
+
def find_occurrences_in_codex_conversation(
|
|
3902
|
+
conn: sqlite3.Connection,
|
|
3903
|
+
conversation_key: str,
|
|
3904
|
+
query: str,
|
|
3905
|
+
*,
|
|
3906
|
+
regex: bool,
|
|
3907
|
+
case_sensitive: bool,
|
|
3908
|
+
kind: str,
|
|
3909
|
+
limit: int = _CODEX_EXACT_FIND_DEFAULT_LIMIT,
|
|
3910
|
+
cursor: str | None = None,
|
|
3911
|
+
direction: str = "next",
|
|
3912
|
+
around: str | None = None,
|
|
3913
|
+
) -> dict[str, object]:
|
|
3914
|
+
"""Return occurrence-exact matches over the materialized visible projection.
|
|
3915
|
+
|
|
3916
|
+
Matching never crosses a physical projection surface. Coordinates are
|
|
3917
|
+
Unicode-scalar offsets into stable render leaves, while paging cursors are
|
|
3918
|
+
bound to both query semantics and the current projection generation.
|
|
3919
|
+
"""
|
|
3920
|
+
if kind not in CODEX_FIND_KINDS:
|
|
3921
|
+
raise ValueError(f"unknown kind: {kind}")
|
|
3922
|
+
if not isinstance(limit, int) or not 1 <= limit <= _CODEX_EXACT_FIND_MAX_LIMIT:
|
|
3923
|
+
raise ValueError("find limit must be between 1 and 200")
|
|
3924
|
+
if direction not in {"next", "previous"}:
|
|
3925
|
+
raise ValueError("find direction must be next or previous")
|
|
3926
|
+
if cursor is not None and around is not None:
|
|
3927
|
+
raise ValueError("find cursor and around are mutually exclusive")
|
|
3928
|
+
q = (query or "").strip()
|
|
3929
|
+
query_id = _exact_find_query_id(
|
|
3930
|
+
q, regex=regex, case_sensitive=case_sensitive, kind=kind
|
|
3931
|
+
)
|
|
3932
|
+
exists = conn.execute(
|
|
3933
|
+
"SELECT 1 FROM codex_conversation_messages WHERE conversation_key=? LIMIT 1",
|
|
3934
|
+
(conversation_key,),
|
|
3935
|
+
).fetchone()
|
|
3936
|
+
if exists is None:
|
|
3937
|
+
return {"status": "not_found", "conversation_key": conversation_key}
|
|
3938
|
+
complete = conn.execute(
|
|
3939
|
+
"SELECT 1 FROM cache_meta WHERE "
|
|
3940
|
+
"key='codex_find_projection_complete_version' AND value=?",
|
|
3941
|
+
(str(CODEX_FIND_PROJECTION_VERSION),),
|
|
3942
|
+
).fetchone()
|
|
3943
|
+
base = _exact_find_base(query_id, status="ready", regex=regex, kind=kind)
|
|
3944
|
+
empty_page = {
|
|
3945
|
+
"start_index": 0,
|
|
3946
|
+
"previous_cursor": None,
|
|
3947
|
+
"next_cursor": None,
|
|
3948
|
+
"occurrences": [],
|
|
3949
|
+
}
|
|
3950
|
+
if complete is None:
|
|
3951
|
+
return {**base, "status": "indexing", "page": empty_page}
|
|
3952
|
+
generation_row = conn.execute(
|
|
3953
|
+
"SELECT value FROM cache_meta WHERE key='codex_find_projection_generation'"
|
|
3954
|
+
).fetchone()
|
|
3955
|
+
try:
|
|
3956
|
+
generation = int(generation_row[0]) if generation_row else 0
|
|
3957
|
+
except (TypeError, ValueError):
|
|
3958
|
+
generation = 0
|
|
3959
|
+
|
|
3960
|
+
decoded_cursor = None
|
|
3961
|
+
if cursor is not None:
|
|
3962
|
+
decoded_cursor = _decode_exact_find_cursor(cursor)
|
|
3963
|
+
if (
|
|
3964
|
+
decoded_cursor["q"] != query_id
|
|
3965
|
+
or decoded_cursor["g"] != generation
|
|
3966
|
+
or decoded_cursor["d"] != direction
|
|
3967
|
+
):
|
|
3968
|
+
raise StaleFindCursor(cursor)
|
|
3969
|
+
|
|
3970
|
+
if not q or (regex and len(q) > _CODEX_FIND_REGEX_MAX_LEN):
|
|
3971
|
+
return {**base, "total": 0, "page": empty_page}
|
|
3972
|
+
pattern = re.compile(q, 0 if case_sensitive else re.IGNORECASE) if regex else None
|
|
3973
|
+
kind_predicate = {
|
|
3974
|
+
"all": "1=1",
|
|
3975
|
+
"prompts": "m.kind='user'",
|
|
3976
|
+
"assistant": "m.kind='assistant'",
|
|
3977
|
+
"tools": "p.surface IN ('call','output','completion')",
|
|
3978
|
+
"thinking": "m.kind='reasoning'",
|
|
3979
|
+
}[kind]
|
|
3980
|
+
rows = conn.execute(
|
|
3981
|
+
"SELECT p.message_id,p.item_key,p.block_key,p.container_block_key,"
|
|
3982
|
+
"p.surface,p.render_order,"
|
|
3983
|
+
"p.projected_text,p.leaves_json,p.disclosure_json,m.kind "
|
|
3984
|
+
"FROM codex_find_projection p "
|
|
3985
|
+
"JOIN codex_conversation_messages m ON m.id=p.message_id "
|
|
3986
|
+
"WHERE p.conversation_key=? AND p.projection_version=? AND "
|
|
3987
|
+
+ kind_predicate
|
|
3988
|
+
+ " ORDER BY p.render_order,p.message_id,p.surface",
|
|
3989
|
+
(conversation_key, CODEX_FIND_PROJECTION_VERSION),
|
|
3990
|
+
)
|
|
3991
|
+
requested: list[tuple[dict[str, object], tuple[int, int, str, int]]] = []
|
|
3992
|
+
around_page: list[tuple[dict[str, object], tuple[int, int, str, int]]] = []
|
|
3993
|
+
head: list[tuple[dict[str, object], tuple[int, int, str, int]]] = []
|
|
3994
|
+
tail: deque[tuple[dict[str, object], tuple[int, int, str, int]]] = deque(
|
|
3995
|
+
maxlen=limit
|
|
3996
|
+
)
|
|
3997
|
+
head_next = None
|
|
3998
|
+
requested_next = None
|
|
3999
|
+
around_next = None
|
|
4000
|
+
around_index = None
|
|
4001
|
+
cursor_valid = decoded_cursor is None
|
|
4002
|
+
cursor_index = int(decoded_cursor["i"]) if decoded_cursor is not None else None
|
|
4003
|
+
requested_start = None
|
|
4004
|
+
requested_end = None
|
|
4005
|
+
if decoded_cursor is not None:
|
|
4006
|
+
if direction == "previous":
|
|
4007
|
+
requested_end = cursor_index
|
|
4008
|
+
requested_start = max(0, cursor_index - limit)
|
|
4009
|
+
else:
|
|
4010
|
+
requested_start = cursor_index
|
|
4011
|
+
requested_end = cursor_index + limit
|
|
4012
|
+
total = 0
|
|
4013
|
+
for (
|
|
4014
|
+
message_id,
|
|
4015
|
+
item_key,
|
|
4016
|
+
block_key,
|
|
4017
|
+
container_block_key,
|
|
4018
|
+
surface,
|
|
4019
|
+
render_order,
|
|
4020
|
+
text,
|
|
4021
|
+
leaves_json,
|
|
4022
|
+
disclosure_json,
|
|
4023
|
+
row_kind,
|
|
4024
|
+
) in rows:
|
|
4025
|
+
try:
|
|
4026
|
+
leaves = tuple(ProjectedLeaf(**leaf) for leaf in json.loads(leaves_json))
|
|
4027
|
+
disclosure = json.loads(disclosure_json)
|
|
4028
|
+
except (TypeError, ValueError, json.JSONDecodeError):
|
|
4029
|
+
continue
|
|
4030
|
+
ranges = (
|
|
4031
|
+
iter_regex_ranges(text, pattern)
|
|
4032
|
+
if pattern is not None
|
|
4033
|
+
else iter_literal_ranges(text, q, case_sensitive=case_sensitive)
|
|
4034
|
+
)
|
|
4035
|
+
for ordinal, match in enumerate(ranges):
|
|
4036
|
+
fragments = slice_range_to_leaves(match, leaves)
|
|
4037
|
+
if not fragments:
|
|
4038
|
+
continue
|
|
4039
|
+
occurrence_id = _exact_find_occurrence_id(
|
|
4040
|
+
query_id,
|
|
4041
|
+
block_key=block_key,
|
|
4042
|
+
surface=surface,
|
|
4043
|
+
ordinal=ordinal,
|
|
4044
|
+
start=match.start,
|
|
4045
|
+
end=match.end,
|
|
4046
|
+
)
|
|
4047
|
+
match_kinds = []
|
|
4048
|
+
if surface != "body":
|
|
4049
|
+
match_kinds.append("tool")
|
|
4050
|
+
if row_kind == "reasoning":
|
|
4051
|
+
match_kinds.append("thinking")
|
|
4052
|
+
occurrence = {
|
|
4053
|
+
"occurrence_id": occurrence_id,
|
|
4054
|
+
"item_key": item_key,
|
|
4055
|
+
"block_key": block_key,
|
|
4056
|
+
"container_block_key": container_block_key,
|
|
4057
|
+
"surface": surface,
|
|
4058
|
+
"match_kinds": match_kinds,
|
|
4059
|
+
"disclosure": disclosure if isinstance(disclosure, list) else [],
|
|
4060
|
+
"fragments": [
|
|
4061
|
+
{
|
|
4062
|
+
"leaf_key": fragment.leaf_key,
|
|
4063
|
+
"start": fragment.start,
|
|
4064
|
+
"end": fragment.end,
|
|
4065
|
+
}
|
|
4066
|
+
for fragment in fragments
|
|
4067
|
+
],
|
|
4068
|
+
}
|
|
4069
|
+
boundary = (render_order, message_id, surface, ordinal)
|
|
4070
|
+
pair = (occurrence, boundary)
|
|
4071
|
+
index = total
|
|
4072
|
+
if len(head) < limit:
|
|
4073
|
+
head.append(pair)
|
|
4074
|
+
elif index == limit:
|
|
4075
|
+
head_next = pair
|
|
4076
|
+
tail.append(pair)
|
|
4077
|
+
if cursor_index == index:
|
|
4078
|
+
if tuple(decoded_cursor["b"]) != boundary:
|
|
4079
|
+
raise StaleFindCursor(cursor)
|
|
4080
|
+
cursor_valid = True
|
|
4081
|
+
if (
|
|
4082
|
+
requested_start is not None
|
|
4083
|
+
and requested_end is not None
|
|
4084
|
+
and requested_start <= index < requested_end
|
|
4085
|
+
):
|
|
4086
|
+
requested.append(pair)
|
|
4087
|
+
elif requested_end is not None and index == requested_end:
|
|
4088
|
+
requested_next = pair
|
|
4089
|
+
if around is not None and around_index is None:
|
|
4090
|
+
if occurrence["occurrence_id"] == around:
|
|
4091
|
+
around_index = index
|
|
4092
|
+
around_page.append(pair)
|
|
4093
|
+
elif around_index is not None:
|
|
4094
|
+
if len(around_page) < limit:
|
|
4095
|
+
around_page.append(pair)
|
|
4096
|
+
elif index == around_index + limit:
|
|
4097
|
+
around_next = pair
|
|
4098
|
+
total += 1
|
|
4099
|
+
|
|
4100
|
+
if not cursor_valid:
|
|
4101
|
+
raise StaleFindCursor(cursor)
|
|
4102
|
+
selection_stale = around is not None and around_index is None
|
|
4103
|
+
next_pair = None
|
|
4104
|
+
if around is not None and around_index is not None:
|
|
4105
|
+
start_index = around_index
|
|
4106
|
+
page_pairs = around_page
|
|
4107
|
+
next_pair = around_next
|
|
4108
|
+
elif around is not None:
|
|
4109
|
+
start_index = 0
|
|
4110
|
+
page_pairs = head
|
|
4111
|
+
next_pair = head_next
|
|
4112
|
+
elif decoded_cursor is not None:
|
|
4113
|
+
start_index = min(requested_start or 0, total)
|
|
4114
|
+
page_pairs = requested
|
|
4115
|
+
next_pair = requested_next
|
|
4116
|
+
elif direction == "previous":
|
|
4117
|
+
page_pairs = list(tail)
|
|
4118
|
+
start_index = max(0, total - len(page_pairs))
|
|
4119
|
+
else:
|
|
4120
|
+
start_index = 0
|
|
4121
|
+
page_pairs = head
|
|
4122
|
+
next_pair = head_next
|
|
4123
|
+
page_occurrences = [occurrence for occurrence, _boundary in page_pairs]
|
|
4124
|
+
|
|
4125
|
+
def cursor_for(
|
|
4126
|
+
index: int,
|
|
4127
|
+
cursor_direction: str,
|
|
4128
|
+
pair: tuple[dict[str, object], tuple[int, int, str, int]] | None,
|
|
4129
|
+
) -> str | None:
|
|
4130
|
+
if not 0 <= index < total or pair is None:
|
|
4131
|
+
return None
|
|
4132
|
+
return _encode_exact_find_cursor(
|
|
4133
|
+
query_id=query_id,
|
|
4134
|
+
generation=generation,
|
|
4135
|
+
start_index=index,
|
|
4136
|
+
direction=cursor_direction,
|
|
4137
|
+
boundary=pair[1],
|
|
4138
|
+
)
|
|
4139
|
+
|
|
4140
|
+
previous_cursor = (
|
|
4141
|
+
cursor_for(start_index, "previous", page_pairs[0])
|
|
4142
|
+
if start_index > 0 and page_pairs else None
|
|
4143
|
+
)
|
|
4144
|
+
next_index = start_index + len(page_occurrences)
|
|
4145
|
+
next_cursor = cursor_for(next_index, "next", next_pair)
|
|
4146
|
+
return {
|
|
4147
|
+
**base,
|
|
4148
|
+
"total": total,
|
|
4149
|
+
"selection_stale": selection_stale,
|
|
4150
|
+
"page": {
|
|
4151
|
+
"start_index": start_index,
|
|
4152
|
+
"previous_cursor": previous_cursor,
|
|
4153
|
+
"next_cursor": next_cursor,
|
|
4154
|
+
"occurrences": page_occurrences,
|
|
4155
|
+
},
|
|
4156
|
+
}
|
|
4157
|
+
|
|
2259
4158
|
# Claude cap parity: the anchor list caps at 500 (bin/_lib_conversation_query.py
|
|
2260
4159
|
# ::_FIND_ANCHOR_CAP), with anchors_truncated when more anchors exist pre-cap.
|
|
2261
4160
|
_CODEX_FIND_ANCHOR_CAP = 500
|
|
@@ -2607,6 +4506,17 @@ def read_codex_payload(
|
|
|
2607
4506
|
response = {"status": "ok", "block_key": block_key, "which": which,
|
|
2608
4507
|
"content": content, "truncated": truncated}
|
|
2609
4508
|
if card is not None:
|
|
4509
|
+
# The SAME ordinal substitution the paged assembly applies (spec sections
|
|
4510
|
+
# 4.3 and 6.5). Without it this route published the provider's own
|
|
4511
|
+
# session id while the paged detail published the conversation-local
|
|
4512
|
+
# ordinal, so one field carried two meanings depending on which route
|
|
4513
|
+
# served it — and a client validator can only be written against one.
|
|
4514
|
+
# A session the index does not know becomes `ref: null`, never the raw
|
|
4515
|
+
# id, because `_apply_session_ordinals` fails closed.
|
|
4516
|
+
index_rows, _detail_bytes = _load_conversation_index_rows(
|
|
4517
|
+
conn, conversation_key)
|
|
4518
|
+
_envelope, ordinals = _build_session_index(index_rows)
|
|
4519
|
+
_apply_session_ordinals(card, ordinals)
|
|
2610
4520
|
response["card"] = card
|
|
2611
4521
|
return response
|
|
2612
4522
|
|