@titan-design/session-graph 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -14,7 +14,8 @@ import { openSessionGraph, refreshCorpus } from "@titan-design/session-graph";
14
14
 
15
15
  const graph = openSessionGraph("~/.local/state/miner/index.sqlite3");
16
16
  const summary = await refreshCorpus(graph, await discoverTranscripts());
17
- // summary.indexed, summary.unchanged, summary.rewound, summary.quarantined, summary.missing
17
+ // summary.indexed, summary.unchanged, summary.rewound, summary.quarantined, summary.missing,
18
+ // summary.facetsBackfilled, summary.facetBacklog
18
19
  ```
19
20
 
20
21
  `openSessionGraph` takes an optional `schemaVersion`: the highest migration version the
@@ -29,15 +30,23 @@ passes 1002.
29
30
  checks the file (`unchanged`, `appended`, `rewritten`, `missing`), reads the delta with
30
31
  `extractTranscript`, applies it in one transaction, and advances the watermark with the
31
32
  new prefix hash. A malformed line quarantines that transcript only.
32
- 2. `rollupSessions` recomputes turn aggregates (index, end, duration, tool calls, thinking
33
- time) for the sessions that changed. Recompute, never accumulate, so incremental and
33
+ 2. `backfillFacets` re-extracts the audit facet of already-indexed transcripts whose
34
+ `transcript_facet` version is below session-read's `EXTRACT_VERSION`: up to
35
+ `facetLimit` of them per pass (default 40), newest `file_mtime` first, each read from
36
+ byte 0 to its watermark. It replaces that transcript's rows in the eight audit tables
37
+ and never writes the legacy tables, so their accumulating counts cannot double.
38
+ Transcripts that are `missing` or `quarantined` are skipped. `summary.facetsBackfilled`
39
+ and `summary.facetBacklog` report progress; pass `facetLimit: Infinity` to clear the
40
+ backlog in one pass. A classifier change is a version bump, not a `resetIndex`.
41
+ 3. `rollupSessions` recomputes turn aggregates (index, end, duration, tool calls, thinking
42
+ time) for the sessions that changed, including those the backfill touched. Recompute, never accumulate, so incremental and
34
43
  full passes converge.
35
- 3. `reconcile` folds cross-transcript observations: `gh pr merge` sightings onto PRs,
44
+ 4. `reconcile` folds cross-transcript observations: `gh pr merge` sightings onto PRs,
36
45
  complete `gh pr create` sightings into new PR rows, subagent end times and parentage
37
46
  from child sessions.
38
- 4. `enrichTasks` runs if the caller passed a `resolveTasks` resolver, once over the whole
47
+ 5. `enrichTasks` runs if the caller passed a `resolveTasks` resolver, once over the whole
39
48
  task table. See below.
40
- 5. Rows whose source file has vanished are marked `missing`. Their facts stay: surviving
49
+ 6. Rows whose source file has vanished are marked `missing`. Their facts stay: surviving
41
50
  Claude Code's own pruning is much of the point.
42
51
 
43
52
  `resetIndex` clears every derived table and rewinds watermarks; the next refresh rebuilds
@@ -100,6 +109,16 @@ files are not needed to migrate. Back up the database before upgrading; restorin
100
109
  that backup is the rollback path for an older binary. Explicit `resetIndex` remains a
101
110
  destructive rebuild and requires the original sources; it is never run by migration.
102
111
 
112
+ Migration 4, `audit tables`, adds one table per session-read audit event kind
113
+ (`request`, `tool_call`, `inbound`, `context_block`, `compaction`, `queue_op`,
114
+ `session_signal`, `cost_state_observation`) plus `transcript_facet`, and adds
115
+ `session.account` and five rollup or outcome columns. It creates only empty tables and
116
+ nullable columns, so existing rows are untouched; transcripts indexed before it gain
117
+ audit rows through `backfillFacets` on later refreshes. `request` holds one row per `(transcript_id, request_id)`:
118
+ the several assistant lines of one API response collapse to the first line's offset and
119
+ the largest value of each token column. `session.account` comes from discovery, not the
120
+ transcript. `purgeTranscript` and `resetIndex` clear all nine tables.
121
+
103
122
  For snapshot-only usage across multiple physical sources, queries select one source
104
123
  by latest native usage timestamp, then greatest usage-record coverage and stable
105
124
  source ID. Source-local reset epochs cannot safely be summed across copies. This is
package/dist/index.d.ts CHANGED
@@ -39,17 +39,67 @@ declare const KIT: {
39
39
  */
40
40
  declare const DOMAIN_DDL = "\n CREATE TABLE IF NOT EXISTS fact (\n fact_id INTEGER PRIMARY KEY,\n transcript_id INTEGER NOT NULL,\n byte_offset INTEGER NOT NULL,\n byte_length INTEGER NOT NULL,\n event_type TEXT NOT NULL,\n ts TEXT NOT NULL,\n seq INTEGER NOT NULL,\n session_id TEXT NOT NULL,\n prompt_id TEXT,\n tool_use_id TEXT,\n t_indexed TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')),\n UNIQUE (transcript_id, byte_offset)\n );\n CREATE INDEX IF NOT EXISTS idx_fact_session_ts ON fact(session_id, ts);\n CREATE INDEX IF NOT EXISTS idx_fact_prompt ON fact(prompt_id);\n CREATE INDEX IF NOT EXISTS idx_fact_tool_use ON fact(tool_use_id);\n\n CREATE TABLE IF NOT EXISTS session (\n session_id TEXT PRIMARY KEY,\n transcript_id INTEGER,\n started_at TEXT,\n ended_at TEXT,\n start_type TEXT,\n cwd TEXT,\n git_branch TEXT,\n ai_title TEXT,\n seed_prompt TEXT,\n cli_version TEXT,\n turn_count INTEGER NOT NULL DEFAULT 0,\n commit_count INTEGER NOT NULL DEFAULT 0,\n push_count INTEGER NOT NULL DEFAULT 0\n );\n CREATE INDEX IF NOT EXISTS idx_session_started ON session(started_at);\n\n CREATE TABLE IF NOT EXISTS session_model_usage (\n session_id TEXT NOT NULL,\n model TEXT NOT NULL,\n input_tokens INTEGER NOT NULL DEFAULT 0,\n output_tokens INTEGER NOT NULL DEFAULT 0,\n cache_read_tokens INTEGER NOT NULL DEFAULT 0,\n cache_creation_tokens INTEGER NOT NULL DEFAULT 0,\n thinking_tokens INTEGER NOT NULL DEFAULT 0,\n request_count INTEGER NOT NULL DEFAULT 0,\n PRIMARY KEY (session_id, model)\n );\n\n CREATE TABLE IF NOT EXISTS turn (\n prompt_id TEXT PRIMARY KEY,\n session_id TEXT NOT NULL,\n turn_index INTEGER NOT NULL,\n started_at TEXT NOT NULL,\n ended_at TEXT,\n duration_ms INTEGER,\n tool_call_count INTEGER NOT NULL DEFAULT 0,\n thinking_ms INTEGER NOT NULL DEFAULT 0,\n fact_id_start INTEGER\n );\n CREATE INDEX IF NOT EXISTS idx_turn_session ON turn(session_id, turn_index);\n\n CREATE TABLE IF NOT EXISTS permission_phase (\n phase_id INTEGER PRIMARY KEY,\n session_id TEXT NOT NULL,\n from_mode TEXT,\n to_mode TEXT NOT NULL,\n trigger TEXT NOT NULL,\n t_valid TEXT NOT NULL,\n t_invalid TEXT,\n fact_id INTEGER\n );\n CREATE INDEX IF NOT EXISTS idx_phase_session ON permission_phase(session_id, t_valid);\n\n CREATE TABLE IF NOT EXISTS human_edit (\n edit_id INTEGER PRIMARY KEY,\n session_id TEXT NOT NULL,\n file_path TEXT NOT NULL,\n ts TEXT NOT NULL,\n fact_id INTEGER,\n UNIQUE (session_id, file_path, ts)\n );\n\n CREATE TABLE IF NOT EXISTS file_checkpoint (\n checkpoint_id INTEGER PRIMARY KEY,\n session_id TEXT NOT NULL,\n file_path TEXT NOT NULL,\n backup_file_name TEXT NOT NULL,\n version INTEGER NOT NULL,\n backup_time TEXT NOT NULL,\n fact_id INTEGER,\n UNIQUE (session_id, file_path, backup_file_name)\n );\n\n CREATE TABLE IF NOT EXISTS pr (\n pr_ref TEXT PRIMARY KEY, number INTEGER, repo TEXT, title TEXT, state TEXT, url TEXT, merged_at TEXT\n );\n CREATE TABLE IF NOT EXISTS branch (\n branch_ref TEXT PRIMARY KEY, repo TEXT, name TEXT NOT NULL, base TEXT, created_at TEXT, deleted_at TEXT\n );\n CREATE TABLE IF NOT EXISTS file (\n file_ref TEXT PRIMARY KEY, repo TEXT, path TEXT NOT NULL\n );\n CREATE TABLE IF NOT EXISTS task (\n task_ref TEXT PRIMARY KEY, task_id TEXT NOT NULL, initiative TEXT, title TEXT, status TEXT\n );\n CREATE TABLE IF NOT EXISTS subagent (\n agent_ref TEXT PRIMARY KEY,\n session_id TEXT,\n child_session_id TEXT,\n parent_agent_ref TEXT,\n agent_type TEXT,\n label TEXT,\n started_at TEXT,\n ended_at TEXT,\n fact_id INTEGER\n );\n CREATE INDEX IF NOT EXISTS idx_subagent_child ON subagent(child_session_id);\n CREATE TABLE IF NOT EXISTS artifact (\n artifact_ref TEXT PRIMARY KEY, kind TEXT, title TEXT, url TEXT, path TEXT, created_at TEXT\n );\n CREATE TABLE IF NOT EXISTS pr_merge_observation (\n number INTEGER NOT NULL, repo_hint TEXT, merged_at TEXT NOT NULL,\n PRIMARY KEY (number, repo_hint, merged_at)\n );\n CREATE TABLE IF NOT EXISTS pr_create_observation (\n tool_use_id TEXT PRIMARY KEY, title TEXT, number INTEGER, repo TEXT, url TEXT\n );\n";
41
41
  /** Every derived table, in an order safe to clear. The watermark table is not derived. */
42
- declare const DERIVED_TABLES: readonly ["normalized_span", "normalized_event", "normalized_source", "search_span", "edge", "turn", "permission_phase", "human_edit", "file_checkpoint", "subagent", "session_model_usage", "session", "fact", "pr", "pr_merge_observation", "pr_create_observation", "branch", "file", "task", "artifact"];
42
+ declare const DERIVED_TABLES: readonly ["normalized_span", "normalized_event", "normalized_source", "search_span", "edge", "turn", "permission_phase", "human_edit", "file_checkpoint", "subagent", "request", "tool_call", "inbound", "context_block", "compaction", "queue_op", "session_signal", "cost_state_observation", "transcript_facet", "session_origin", "session_external_event", "episode", "session_model_usage", "session", "fact", "pr", "pr_merge_observation", "pr_create_observation", "branch", "file", "task", "artifact"];
43
43
  declare const MIGRATIONS: Migration[];
44
44
 
45
+ /** Recorded in `_migration`; store-sqlite refuses a database whose applied name differs, so never rename it. */
46
+ declare const AUDIT_MIGRATION_NAME = "audit tables";
47
+ /** Tables written per transcript by `applyAudit` and dropped by `purgeTranscript`. */
48
+ declare const AUDIT_TABLES: readonly ["request", "tool_call", "inbound", "context_block", "compaction", "queue_op", "session_signal", "cost_state_observation"];
49
+ /** Per-transcript read state for each audit facet; cleared with the audit rows so a re-read starts clean. */
50
+ declare const FACET_TABLE = "transcript_facet";
51
+ declare const AUDIT_DDL = "\n CREATE TABLE IF NOT EXISTS request (\n transcript_id INTEGER NOT NULL,\n request_id TEXT NOT NULL,\n byte_offset INTEGER NOT NULL, -- first line seen for this request\n session_id TEXT NOT NULL,\n message_id TEXT,\n ts TEXT NOT NULL,\n model TEXT NOT NULL,\n input_tokens INTEGER NOT NULL DEFAULT 0,\n cache_read_tokens INTEGER NOT NULL DEFAULT 0,\n cache_creation_tokens INTEGER NOT NULL DEFAULT 0,\n cache_creation_5m INTEGER NOT NULL DEFAULT 0,\n cache_creation_1h INTEGER NOT NULL DEFAULT 0,\n output_tokens INTEGER NOT NULL DEFAULT 0,\n thinking_tokens INTEGER NOT NULL DEFAULT 0,\n context_tokens INTEGER NOT NULL DEFAULT 0, -- input + cache_read + cache_creation\n service_tier TEXT,\n is_sidechain INTEGER NOT NULL DEFAULT 0,\n -- rollup-owned, recomputed for touched sessions:\n seq_in_session INTEGER,\n gap_ms INTEGER, -- since previous request in the session\n ctx_delta INTEGER, -- context_tokens - (prev.context_tokens + prev.output_tokens)\n wake_cause TEXT,\n wake_delivery TEXT,\n wake_detail TEXT, -- tool name, channel sender, or NULL\n wake_tool_family TEXT,\n wake_mcp_server TEXT,\n wake_offset INTEGER, -- byte_offset of the inbound row\n PRIMARY KEY (transcript_id, request_id)\n ) WITHOUT ROWID;\n CREATE INDEX IF NOT EXISTS idx_request_session_ts ON request(session_id, ts);\n CREATE INDEX IF NOT EXISTS idx_request_ts ON request(ts);\n CREATE INDEX IF NOT EXISTS idx_request_id ON request(request_id);\n\n CREATE TABLE IF NOT EXISTS tool_call (\n transcript_id INTEGER NOT NULL, byte_offset INTEGER NOT NULL, block_index INTEGER NOT NULL,\n session_id TEXT NOT NULL, ts TEXT NOT NULL, tool_use_id TEXT NOT NULL,\n name TEXT NOT NULL, family TEXT NOT NULL, mcp_server TEXT, input_chars INTEGER NOT NULL,\n PRIMARY KEY (transcript_id, byte_offset, block_index)\n ) WITHOUT ROWID;\n CREATE INDEX IF NOT EXISTS idx_tool_call_use ON tool_call(tool_use_id);\n\n CREATE TABLE IF NOT EXISTS inbound (\n transcript_id INTEGER NOT NULL, byte_offset INTEGER NOT NULL, block_index INTEGER NOT NULL,\n session_id TEXT NOT NULL, ts TEXT NOT NULL,\n cause TEXT NOT NULL, delivery TEXT NOT NULL, detail TEXT,\n origin_server TEXT, from_name TEXT, msg_id TEXT, tool_use_id TEXT,\n is_error INTEGER NOT NULL DEFAULT 0, content_hash TEXT NOT NULL, chars INTEGER NOT NULL,\n queued_ms INTEGER, -- rollup-owned\n PRIMARY KEY (transcript_id, byte_offset, block_index)\n ) WITHOUT ROWID;\n CREATE INDEX IF NOT EXISTS idx_inbound_session ON inbound(session_id, byte_offset);\n\n CREATE TABLE IF NOT EXISTS context_block (\n transcript_id INTEGER NOT NULL, byte_offset INTEGER NOT NULL, block_index INTEGER NOT NULL,\n session_id TEXT NOT NULL, ts TEXT NOT NULL, source TEXT NOT NULL,\n tool_use_id TEXT, attachment_type TEXT, chars INTEGER NOT NULL, is_media INTEGER NOT NULL DEFAULT 0,\n PRIMARY KEY (transcript_id, byte_offset, block_index)\n ) WITHOUT ROWID;\n CREATE INDEX IF NOT EXISTS idx_context_block_session ON context_block(session_id, source);\n\n CREATE TABLE IF NOT EXISTS compaction (\n transcript_id INTEGER NOT NULL, byte_offset INTEGER NOT NULL,\n session_id TEXT NOT NULL, ts TEXT NOT NULL, trigger TEXT,\n pre_tokens INTEGER, post_tokens INTEGER, dropped_tokens INTEGER, duration_ms INTEGER,\n mid_loop INTEGER, -- rollup-owned: previous inbound was a tool_result\n PRIMARY KEY (transcript_id, byte_offset)\n ) WITHOUT ROWID;\n\n CREATE TABLE IF NOT EXISTS queue_op (\n transcript_id INTEGER NOT NULL, byte_offset INTEGER NOT NULL,\n session_id TEXT NOT NULL, ts TEXT NOT NULL, operation TEXT NOT NULL,\n content_hash TEXT, origin_server TEXT,\n PRIMARY KEY (transcript_id, byte_offset)\n ) WITHOUT ROWID;\n CREATE INDEX IF NOT EXISTS idx_queue_op_hash ON queue_op(session_id, content_hash);\n\n CREATE TABLE IF NOT EXISTS session_signal (\n transcript_id INTEGER NOT NULL, byte_offset INTEGER NOT NULL, block_index INTEGER NOT NULL,\n session_id TEXT NOT NULL, ts TEXT NOT NULL, signal TEXT NOT NULL, detail TEXT, tool_use_id TEXT,\n -- one Bash block can emit up to three signals (commit, push, pr_create); the signal name disambiguates them\n PRIMARY KEY (transcript_id, byte_offset, block_index, signal)\n ) WITHOUT ROWID;\n CREATE INDEX IF NOT EXISTS idx_signal_session ON session_signal(session_id, ts);\n\n CREATE TABLE IF NOT EXISTS cost_state_observation (\n transcript_id INTEGER NOT NULL, byte_offset INTEGER NOT NULL,\n session_id TEXT NOT NULL, ts TEXT, total_cost_usd REAL NOT NULL, model_usage TEXT NOT NULL,\n PRIMARY KEY (transcript_id, byte_offset)\n ) WITHOUT ROWID;\n\n CREATE TABLE IF NOT EXISTS transcript_facet (\n transcript_id INTEGER NOT NULL, facet TEXT NOT NULL,\n version INTEGER NOT NULL, indexed_to INTEGER NOT NULL,\n PRIMARY KEY (transcript_id, facet)\n ) WITHOUT ROWID;\n";
52
+
53
+ /** Recorded in `_migration`; store-sqlite refuses a database whose applied name differs, so never rename it. */
54
+ declare const ORIGIN_MIGRATION_NAME = "origin, episodes, prices";
55
+ /** Re-derivable from their own sources rather than from transcripts, so `resetIndex` clears them and the next pass refills them. */
56
+ declare const ORIGIN_TABLES: readonly ["session_origin", "session_external_event"];
57
+ /** Written from transcript offsets by `replaceEpisodes`; a rewritten transcript invalidates them, so purge drops them by session. */
58
+ declare const EPISODE_TABLE = "episode";
59
+ declare const ORIGIN_DDL = "\n CREATE TABLE IF NOT EXISTS session_origin ( -- not derived from transcripts\n session_id TEXT PRIMARY KEY,\n origin_system TEXT NOT NULL, -- 'agent-chat'\n agent_id TEXT, agent_name TEXT, parent_name TEXT, parent_session_id TEXT,\n profile TEXT, model_alias TEXT, surface TEXT, isolation TEXT, depth INTEGER,\n origin_kind TEXT, -- spawned | adopted | inherited | human\n config_dir TEXT, spawn_cwd TEXT, spawned_at TEXT,\n brief_chars INTEGER, brief_excerpt TEXT, brief_path TEXT, launch_args TEXT,\n resolved_at TEXT NOT NULL\n );\n CREATE INDEX IF NOT EXISTS idx_origin_parent ON session_origin(parent_session_id);\n\n CREATE TABLE IF NOT EXISTS session_external_event ( -- teleport, handoff, retire, exit\n session_id TEXT NOT NULL, ts TEXT NOT NULL, origin_system TEXT NOT NULL,\n kind TEXT NOT NULL, detail TEXT,\n PRIMARY KEY (session_id, ts, kind)\n ) WITHOUT ROWID;\n\n CREATE TABLE IF NOT EXISTS episode ( -- provisional, see section 9\n session_id TEXT NOT NULL, episode_index INTEGER NOT NULL,\n heuristic TEXT NOT NULL, heuristic_version INTEGER NOT NULL,\n started_at TEXT NOT NULL, ended_at TEXT NOT NULL,\n start_offset INTEGER NOT NULL, end_offset INTEGER NOT NULL,\n opened_by TEXT NOT NULL, -- brief | channel_followup | idle_gap | compaction\n assignment_offset INTEGER, first_deliverable_offset INTEGER, first_deliverable_signal TEXT,\n first_status_offset INTEGER,\n PRIMARY KEY (session_id, heuristic, episode_index)\n ) WITHOUT ROWID;\n\n CREATE TABLE IF NOT EXISTS price (\n model TEXT NOT NULL, effective_from TEXT NOT NULL, table_version INTEGER NOT NULL,\n input_usd_mtok REAL NOT NULL, cache_read_usd_mtok REAL NOT NULL,\n cache_write_5m_usd_mtok REAL NOT NULL, cache_write_1h_usd_mtok REAL NOT NULL,\n output_usd_mtok REAL NOT NULL, source TEXT,\n PRIMARY KEY (model, effective_from)\n ) WITHOUT ROWID;\n";
60
+ declare const ORIGIN_VIEWS: readonly ["\n CREATE VIEW IF NOT EXISTS request_dedup AS\n SELECT transcript_id, request_id, byte_offset, session_id, message_id, ts, model,\n input_tokens, cache_read_tokens, cache_creation_tokens, cache_creation_5m, cache_creation_1h,\n output_tokens, thinking_tokens, context_tokens, service_tier, is_sidechain,\n seq_in_session, gap_ms, ctx_delta, wake_cause, wake_delivery, wake_detail,\n wake_tool_family, wake_mcp_server, wake_offset FROM (\n SELECT r.*, ROW_NUMBER() OVER (PARTITION BY r.request_id ORDER BY r.ts, r.transcript_id) AS copy_rank\n FROM request r\n ) WHERE copy_rank = 1;\n", "\n CREATE VIEW IF NOT EXISTS request_cost AS\n WITH candidate AS (\n SELECT d.request_id, p.model AS price_model, p.effective_from AS price_effective_from,\n p.input_usd_mtok, p.cache_read_usd_mtok, p.cache_write_5m_usd_mtok, p.cache_write_1h_usd_mtok, p.output_usd_mtok,\n ROW_NUMBER() OVER (PARTITION BY d.request_id ORDER BY length(p.model) DESC, p.effective_from DESC) AS match_rank\n FROM request_dedup d\n JOIN price p ON substr(d.model, 1, length(p.model)) = p.model AND p.effective_from <= d.ts\n ), component AS (\n SELECT d.*, c.price_model, c.price_effective_from, c.price_model IS NOT NULL AS priced,\n COALESCE(d.input_tokens * c.input_usd_mtok, 0) / 1e6 AS input_cost_usd,\n COALESCE(d.cache_read_tokens * c.cache_read_usd_mtok, 0) / 1e6 AS cache_read_cost_usd,\n COALESCE(CASE WHEN d.cache_creation_5m + d.cache_creation_1h = 0 THEN d.cache_creation_tokens ELSE d.cache_creation_5m END\n * c.cache_write_5m_usd_mtok, 0) / 1e6 AS cache_write_5m_cost_usd,\n COALESCE(d.cache_creation_1h * c.cache_write_1h_usd_mtok, 0) / 1e6 AS cache_write_1h_cost_usd,\n COALESCE(d.output_tokens * c.output_usd_mtok, 0) / 1e6 AS output_cost_usd\n FROM request_dedup d\n LEFT JOIN candidate c ON c.request_id = d.request_id AND c.match_rank = 1\n )\n SELECT *,\n input_cost_usd + cache_read_cost_usd + cache_write_5m_cost_usd + cache_write_1h_cost_usd + output_cost_usd AS cost_usd,\n (cache_read_tokens < 0.2 * context_tokens AND cache_creation_tokens >= 20000) AS is_cold,\n CASE WHEN context_tokens < 50000 THEN '<50k' WHEN context_tokens < 100000 THEN '50-100k'\n WHEN context_tokens < 200000 THEN '100-200k' ELSE '200k+' END AS context_band,\n CASE WHEN gap_ms IS NULL THEN NULL WHEN gap_ms < 300000 THEN '<5m'\n WHEN gap_ms < 3600000 THEN '5-60m' ELSE '>60m' END AS gap_band\n FROM component;\n", "\n CREATE VIEW IF NOT EXISTS context_contribution AS\n WITH stream AS (\n SELECT transcript_id, byte_offset, 0 AS lane, block_index, NULL AS request_id FROM context_block\n UNION ALL\n SELECT transcript_id, byte_offset, -1 AS lane, 0, request_id FROM request\n ), ordered AS (\n SELECT *, COUNT(request_id) OVER (PARTITION BY transcript_id ORDER BY byte_offset, lane, block_index ROWS UNBOUNDED PRECEDING) AS requests_before\n FROM stream\n ), owner AS (\n SELECT b.transcript_id, b.byte_offset, b.block_index, r.request_id\n FROM ordered b JOIN ordered r ON r.transcript_id = b.transcript_id AND r.lane = -1 AND r.requests_before = b.requests_before + 1\n WHERE b.lane = 0\n ), tool AS (\n SELECT tool_use_id, name, family, mcp_server,\n ROW_NUMBER() OVER (PARTITION BY tool_use_id ORDER BY transcript_id, byte_offset, block_index) AS copy_rank\n FROM tool_call\n )\n SELECT cb.transcript_id, cb.byte_offset, cb.block_index, cb.session_id, cb.ts, cb.source,\n cb.tool_use_id, cb.attachment_type, cb.chars, cb.is_media, o.request_id, d.ctx_delta,\n CASE WHEN cb.source IN ('assistant_text', 'assistant_thinking', 'assistant_tool_input') THEN NULL\n ELSE d.ctx_delta * cb.chars * 1.0 / NULLIF(SUM(CASE WHEN cb.source IN ('assistant_text', 'assistant_thinking', 'assistant_tool_input') THEN 0 ELSE cb.chars END)\n OVER (PARTITION BY o.transcript_id, o.request_id), 0) END AS est_tokens,\n t.name AS tool_name, t.family AS tool_family, t.mcp_server\n FROM context_block cb\n JOIN owner o USING (transcript_id, byte_offset, block_index)\n JOIN request_dedup d ON d.transcript_id = o.transcript_id AND d.request_id = o.request_id\n LEFT JOIN tool t ON t.tool_use_id = cb.tool_use_id AND t.copy_rank = 1;\n"];
61
+
62
+ /** Write one delta's audit events. Runs inside `applyDelta`'s transaction, so it opens none of its own. */
63
+ declare function applyAudit(db: Db, transcriptId: number, delta: TranscriptDelta): void;
64
+
65
+ /** The one facet today: the eight audit tables, all written from the same extraction. */
66
+ declare const AUDIT_FACET = "audit";
67
+ /** Transcripts re-extracted per pass, so a daemon's pass stays short while the backlog drains. */
68
+ declare const DEFAULT_FACET_LIMIT = 40;
69
+ interface BackfillOptions {
70
+ /** Transcripts visited this pass. `Infinity` clears the whole backlog. */
71
+ limit?: number;
72
+ /** The extraction version a facet must reach. Defaults to session-read's `EXTRACT_VERSION`. */
73
+ version?: number;
74
+ }
75
+ interface BackfillSummary {
76
+ backfilled: number;
77
+ /** Stale transcripts left for later passes. */
78
+ backlog: number;
79
+ /** Sessions whose audit rows were rebuilt, for the caller's rollup. */
80
+ sessionIds: string[];
81
+ }
82
+ /**
83
+ * Bring the audit facet of already-indexed transcripts up to `version` without a
84
+ * full rebuild: re-read each one up to its watermark and replace only its audit
85
+ * rows. Legacy tables are never written, so their accumulating counts cannot double.
86
+ */
87
+ declare function backfillFacets(graph: SessionGraph, transcripts: readonly DiscoveredTranscript[], options?: BackfillOptions): Promise<BackfillSummary>;
88
+
45
89
  /**
46
90
  * Apply one transcript chunk's delta in a single transaction, so a crash can
47
91
  * never leave rows committed behind a stale watermark or ahead of their rows.
48
- * Append-only tables are idempotent through unique indexes; session and usage
49
- * rows accumulate; ref-keyed assets insert-if-absent with COALESCE merges that
50
- * mirror `EventFolder`'s rules.
92
+ * Append-only tables are idempotent through unique indexes; session rows
93
+ * accumulate; usage is left to the rollup, which recomputes it from `request`;
94
+ * ref-keyed assets insert-if-absent with COALESCE merges that mirror
95
+ * `EventFolder`'s rules.
51
96
  */
52
- declare function applyDelta(graph: SessionGraph, transcriptId: number, delta: TranscriptDelta): void;
97
+ declare function applyDelta(graph: SessionGraph, transcriptId: number, delta: TranscriptDelta, source?: DeltaSource): void;
98
+ /** What discovery knows about the file that the lines themselves do not carry. */
99
+ interface DeltaSource {
100
+ /** The Claude config dir the transcript was found under; `null` when discovery did not say. */
101
+ account?: string | null;
102
+ }
53
103
 
54
104
  /**
55
105
  * Drop every derived row one transcript produced, so re-reading it from byte 0
@@ -68,10 +118,13 @@ declare function applyDelta(graph: SessionGraph, transcriptId: number, delta: Tr
68
118
  * inserts are insert-if-absent; a single transcript cannot know whether it was
69
119
  * the only source. The two observation tables are left for the same reason —
70
120
  * `reconcile` folds them, and re-reading re-asserts the same rows.
121
+ *
122
+ * Audit rows and facet state carry `transcript_id` themselves, so they go by it
123
+ * directly rather than through `session`.
71
124
  */
72
125
  declare function purgeTranscript(graph: SessionGraph, transcriptId: number): void;
73
126
 
74
- /** Recompute turn aggregates for the given sessions in one transaction. */
127
+ /** Recompute turn aggregates and the audit rollup for the given sessions in one transaction. */
75
128
  declare function rollupSessions(graph: SessionGraph, sessionIds: readonly string[]): number;
76
129
  interface ReconcileCounts {
77
130
  prMerges: number;
@@ -81,6 +134,74 @@ interface ReconcileCounts {
81
134
  /** Whole-table reconciliations, run once at the end of a refresh pass after the rollup. */
82
135
  declare function reconcile(graph: SessionGraph): ReconcileCounts;
83
136
 
137
+ /**
138
+ * Who started a session and why, as a launcher such as agent-chat recorded it.
139
+ * A transcript cannot state this: a `claude -p` worker and a headless miner look
140
+ * the same from inside. Every field but `originSystem` is optional.
141
+ */
142
+ interface ResolvedOrigin {
143
+ originSystem: string;
144
+ agentId?: string | null;
145
+ agentName?: string | null;
146
+ parentName?: string | null;
147
+ parentSessionId?: string | null;
148
+ profile?: string | null;
149
+ modelAlias?: string | null;
150
+ surface?: string | null;
151
+ isolation?: string | null;
152
+ depth?: number | null;
153
+ /** `spawned`, `adopted`, `inherited` or `human`. */
154
+ originKind?: string | null;
155
+ configDir?: string | null;
156
+ spawnCwd?: string | null;
157
+ spawnedAt?: string | null;
158
+ briefChars?: number | null;
159
+ briefExcerpt?: string | null;
160
+ briefPath?: string | null;
161
+ launchArgs?: string | null;
162
+ }
163
+ /** A lifecycle event the launcher saw outside the transcript: `teleport`, `handoff`, `retired`, `exited`. */
164
+ interface ExternalEvent {
165
+ sessionId: string;
166
+ ts: string;
167
+ kind: string;
168
+ detail?: string | null;
169
+ /** Defaults to the session's resolved origin system. */
170
+ originSystem?: string | null;
171
+ }
172
+ interface OriginResolution {
173
+ /** Keyed by session id. */
174
+ origins: Record<string, ResolvedOrigin>;
175
+ externalEvents?: readonly ExternalEvent[];
176
+ }
177
+ /**
178
+ * Supplied by the caller, never by this package: session-graph is tier 2 and
179
+ * must not learn where a launcher keeps its records. Called once per pass with
180
+ * every session that has no origin row or a stale one.
181
+ */
182
+ type OriginResolver = (sessionIds: readonly string[]) => PromiseLike<OriginResolution> | OriginResolution;
183
+ interface OriginEnrichment {
184
+ /** Session ids handed to the resolver. */
185
+ requested: number;
186
+ /** `session_origin` rows the resolver wrote. */
187
+ applied: number;
188
+ /** `session_external_event` rows the resolver wrote. */
189
+ events: number;
190
+ /** The resolver threw; origin rows stand as the last pass left them. */
191
+ failed: boolean;
192
+ /** Why it failed, for the caller to log. Absent unless `failed`. */
193
+ error?: string;
194
+ }
195
+ declare const NO_ORIGINS: OriginEnrichment;
196
+ declare function sessionsNeedingOrigin(graph: SessionGraph): string[];
197
+ /**
198
+ * Resolve origins for the sessions that lack a current one, then project every
199
+ * stored origin into `spawned` edges and `subagent` rows. Projecting from the
200
+ * table, not from this pass's answer, restores rows a parent transcript's purge
201
+ * removed. A resolver that throws costs this pass its origins and nothing else.
202
+ */
203
+ declare function resolveOrigins(graph: SessionGraph, resolver: OriginResolver | undefined): Promise<OriginEnrichment>;
204
+
84
205
  /**
85
206
  * What a product's task store knows and a transcript cannot state: the task's
86
207
  * present title, its initiative, and the status it holds right now rather than
@@ -129,6 +250,14 @@ interface IndexOptions {
129
250
  withContentHash?: boolean;
130
251
  /** Fill task title, initiative and present status from the caller's store. Absent means transcripts alone. */
131
252
  resolveTasks?: TaskResolver;
253
+ /** Record who launched each session. Runs once per `refreshCorpus` pass, never per transcript. Absent means no origin rows. */
254
+ resolveOrigins?: OriginResolver;
255
+ }
256
+ interface RefreshOptions extends IndexOptions {
257
+ /** Roll up every session, not just the ones this pass touched. */
258
+ full?: boolean;
259
+ /** Stale audit facets re-extracted per pass (default 40). `Infinity` clears the backlog. */
260
+ facetLimit?: number;
132
261
  }
133
262
  type TranscriptOutcome = {
134
263
  status: "unchanged" | "indexed" | "rewound";
@@ -161,22 +290,25 @@ interface RefreshSummary {
161
290
  turnsRolledUp: number;
162
291
  reconciled: ReconcileCounts;
163
292
  tasks: TaskEnrichment;
293
+ origins: OriginEnrichment;
164
294
  markedMissing: number;
295
+ facetsBackfilled: number;
296
+ /** Transcripts whose audit facet is still stale after this pass. */
297
+ facetBacklog: number;
165
298
  }
166
299
  /**
167
- * One pass over a corpus: index every transcript, roll up the sessions that
168
- * changed, reconcile cross-transcript observations, enrich tasks, and mark rows
169
- * whose source file is gone. Idempotent: a second pass over unchanged files
170
- * changes nothing.
300
+ * One pass over a corpus: index every transcript, re-extract a bounded batch of
301
+ * stale audit facets, roll up the sessions that changed, resolve session
302
+ * origins, reconcile cross-transcript observations, enrich tasks, and mark
303
+ * rows whose source file is gone. Idempotent: a second pass over unchanged
304
+ * files changes nothing.
171
305
  *
172
306
  * The resolver is hoisted out of the per-transcript loop and run once over the
173
307
  * whole task table, so a corpus of N transcripts costs one resolver call rather
174
308
  * than N, and a task whose store row changed refreshes even when no transcript
175
309
  * did. That bounds staleness to one pass.
176
310
  */
177
- declare function refreshCorpus(graph: SessionGraph, transcripts: readonly DiscoveredTranscript[], options?: IndexOptions & {
178
- full?: boolean;
179
- }): Promise<RefreshSummary>;
311
+ declare function refreshCorpus(graph: SessionGraph, transcripts: readonly DiscoveredTranscript[], options?: RefreshOptions): Promise<RefreshSummary>;
180
312
 
181
313
  interface NormalizedIndexResult {
182
314
  status: "indexed" | "unchanged" | "missing" | "quarantined";
@@ -222,4 +354,4 @@ declare function normalizedUsage(graph: SessionGraph, ref: string): NormalizedUs
222
354
  /** Ambiguity is explicit; workspace session bodies remain in their original namespace. */
223
355
  declare function resolveConversationAlias(db: Db, legacyRef: string): string | null;
224
356
 
225
- export { type ConversationSummary, DERIVED_TABLES, DOMAIN_DDL, type IndexOptions, type IndexedSpan, KIT, MIGRATIONS, NO_ENRICHMENT, type NormalizedIndexResult, type NormalizedUsageSummary, type OpenSessionGraphOptions, type ReconcileCounts, type RefreshSummary, type ResolvedTask, type SessionGraph, type TaskEnrichment, type TaskResolution, type TaskResolver, type TranscriptOutcome, allSessionIds, allTaskIds, applyDelta, enrichTasks, indexCodexSource, indexTranscript, normalizedSessions, normalizedUsage, openSessionGraph, purgeTranscript, readIndexedText, reconcile, refreshCorpus, resetIndex, resolveConversationAlias, rollupSessions };
357
+ export { AUDIT_DDL, AUDIT_FACET, AUDIT_MIGRATION_NAME, AUDIT_TABLES, type BackfillOptions, type BackfillSummary, type ConversationSummary, DEFAULT_FACET_LIMIT, DERIVED_TABLES, DOMAIN_DDL, type DeltaSource, EPISODE_TABLE, type ExternalEvent, FACET_TABLE, type IndexOptions, type IndexedSpan, KIT, MIGRATIONS, NO_ENRICHMENT, NO_ORIGINS, type NormalizedIndexResult, type NormalizedUsageSummary, ORIGIN_DDL, ORIGIN_MIGRATION_NAME, ORIGIN_TABLES, ORIGIN_VIEWS, type OpenSessionGraphOptions, type OriginEnrichment, type OriginResolution, type OriginResolver, type ReconcileCounts, type RefreshOptions, type RefreshSummary, type ResolvedOrigin, type ResolvedTask, type SessionGraph, type TaskEnrichment, type TaskResolution, type TaskResolver, type TranscriptOutcome, allSessionIds, allTaskIds, applyAudit, applyDelta, backfillFacets, enrichTasks, indexCodexSource, indexTranscript, normalizedSessions, normalizedUsage, openSessionGraph, purgeTranscript, readIndexedText, reconcile, refreshCorpus, resetIndex, resolveConversationAlias, resolveOrigins, rollupSessions, sessionsNeedingOrigin };