@titan-design/session-graph 0.5.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -6
- package/dist/index.d.ts +146 -14
- package/dist/index.js +603 -26
- package/dist/index.js.map +1 -1
- package/package.json +3 -3
package/README.md
CHANGED
|
@@ -14,7 +14,8 @@ import { openSessionGraph, refreshCorpus } from "@titan-design/session-graph";
|
|
|
14
14
|
|
|
15
15
|
const graph = openSessionGraph("~/.local/state/miner/index.sqlite3");
|
|
16
16
|
const summary = await refreshCorpus(graph, await discoverTranscripts());
|
|
17
|
-
// summary.indexed, summary.unchanged, summary.rewound, summary.quarantined, summary.missing
|
|
17
|
+
// summary.indexed, summary.unchanged, summary.rewound, summary.quarantined, summary.missing,
|
|
18
|
+
// summary.facetsBackfilled, summary.facetBacklog
|
|
18
19
|
```
|
|
19
20
|
|
|
20
21
|
`openSessionGraph` takes an optional `schemaVersion`: the highest migration version the
|
|
@@ -29,15 +30,23 @@ passes 1002.
|
|
|
29
30
|
checks the file (`unchanged`, `appended`, `rewritten`, `missing`), reads the delta with
|
|
30
31
|
`extractTranscript`, applies it in one transaction, and advances the watermark with the
|
|
31
32
|
new prefix hash. A malformed line quarantines that transcript only.
|
|
32
|
-
2. `
|
|
33
|
-
|
|
33
|
+
2. `backfillFacets` re-extracts the audit facet of already-indexed transcripts whose
|
|
34
|
+
`transcript_facet` version is below session-read's `EXTRACT_VERSION`: up to
|
|
35
|
+
`facetLimit` of them per pass (default 40), newest `file_mtime` first, each read from
|
|
36
|
+
byte 0 to its watermark. It replaces that transcript's rows in the eight audit tables
|
|
37
|
+
and never writes the legacy tables, so their accumulating counts cannot double.
|
|
38
|
+
Transcripts that are `missing` or `quarantined` are skipped. `summary.facetsBackfilled`
|
|
39
|
+
and `summary.facetBacklog` report progress; pass `facetLimit: Infinity` to clear the
|
|
40
|
+
backlog in one pass. A classifier change is a version bump, not a `resetIndex`.
|
|
41
|
+
3. `rollupSessions` recomputes turn aggregates (index, end, duration, tool calls, thinking
|
|
42
|
+
time) for the sessions that changed, including those the backfill touched. Recompute, never accumulate, so incremental and
|
|
34
43
|
full passes converge.
|
|
35
|
-
|
|
44
|
+
4. `reconcile` folds cross-transcript observations: `gh pr merge` sightings onto PRs,
|
|
36
45
|
complete `gh pr create` sightings into new PR rows, subagent end times and parentage
|
|
37
46
|
from child sessions.
|
|
38
|
-
|
|
47
|
+
5. `enrichTasks` runs if the caller passed a `resolveTasks` resolver, once over the whole
|
|
39
48
|
task table. See below.
|
|
40
|
-
|
|
49
|
+
6. Rows whose source file has vanished are marked `missing`. Their facts stay: surviving
|
|
41
50
|
Claude Code's own pruning is much of the point.
|
|
42
51
|
|
|
43
52
|
`resetIndex` clears every derived table and rewinds watermarks; the next refresh rebuilds
|
|
@@ -100,6 +109,16 @@ files are not needed to migrate. Back up the database before upgrading; restorin
|
|
|
100
109
|
that backup is the rollback path for an older binary. Explicit `resetIndex` remains a
|
|
101
110
|
destructive rebuild and requires the original sources; it is never run by migration.
|
|
102
111
|
|
|
112
|
+
Migration 4, `audit tables`, adds one table per session-read audit event kind
|
|
113
|
+
(`request`, `tool_call`, `inbound`, `context_block`, `compaction`, `queue_op`,
|
|
114
|
+
`session_signal`, `cost_state_observation`) plus `transcript_facet`, and adds
|
|
115
|
+
`session.account` and five rollup or outcome columns. It creates only empty tables and
|
|
116
|
+
nullable columns, so existing rows are untouched; transcripts indexed before it gain
|
|
117
|
+
audit rows through `backfillFacets` on later refreshes. `request` holds one row per `(transcript_id, request_id)`:
|
|
118
|
+
the several assistant lines of one API response collapse to the first line's offset and
|
|
119
|
+
the largest value of each token column. `session.account` comes from discovery, not the
|
|
120
|
+
transcript. `purgeTranscript` and `resetIndex` clear all nine tables.
|
|
121
|
+
|
|
103
122
|
For snapshot-only usage across multiple physical sources, queries select one source
|
|
104
123
|
by latest native usage timestamp, then greatest usage-record coverage and stable
|
|
105
124
|
source ID. Source-local reset epochs cannot safely be summed across copies. This is
|
package/dist/index.d.ts
CHANGED
|
@@ -39,17 +39,67 @@ declare const KIT: {
|
|
|
39
39
|
*/
|
|
40
40
|
declare const DOMAIN_DDL = "\n CREATE TABLE IF NOT EXISTS fact (\n fact_id INTEGER PRIMARY KEY,\n transcript_id INTEGER NOT NULL,\n byte_offset INTEGER NOT NULL,\n byte_length INTEGER NOT NULL,\n event_type TEXT NOT NULL,\n ts TEXT NOT NULL,\n seq INTEGER NOT NULL,\n session_id TEXT NOT NULL,\n prompt_id TEXT,\n tool_use_id TEXT,\n t_indexed TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ','now')),\n UNIQUE (transcript_id, byte_offset)\n );\n CREATE INDEX IF NOT EXISTS idx_fact_session_ts ON fact(session_id, ts);\n CREATE INDEX IF NOT EXISTS idx_fact_prompt ON fact(prompt_id);\n CREATE INDEX IF NOT EXISTS idx_fact_tool_use ON fact(tool_use_id);\n\n CREATE TABLE IF NOT EXISTS session (\n session_id TEXT PRIMARY KEY,\n transcript_id INTEGER,\n started_at TEXT,\n ended_at TEXT,\n start_type TEXT,\n cwd TEXT,\n git_branch TEXT,\n ai_title TEXT,\n seed_prompt TEXT,\n cli_version TEXT,\n turn_count INTEGER NOT NULL DEFAULT 0,\n commit_count INTEGER NOT NULL DEFAULT 0,\n push_count INTEGER NOT NULL DEFAULT 0\n );\n CREATE INDEX IF NOT EXISTS idx_session_started ON session(started_at);\n\n CREATE TABLE IF NOT EXISTS session_model_usage (\n session_id TEXT NOT NULL,\n model TEXT NOT NULL,\n input_tokens INTEGER NOT NULL DEFAULT 0,\n output_tokens INTEGER NOT NULL DEFAULT 0,\n cache_read_tokens INTEGER NOT NULL DEFAULT 0,\n cache_creation_tokens INTEGER NOT NULL DEFAULT 0,\n thinking_tokens INTEGER NOT NULL DEFAULT 0,\n request_count INTEGER NOT NULL DEFAULT 0,\n PRIMARY KEY (session_id, model)\n );\n\n CREATE TABLE IF NOT EXISTS turn (\n prompt_id TEXT PRIMARY KEY,\n session_id TEXT NOT NULL,\n turn_index INTEGER NOT NULL,\n started_at TEXT NOT NULL,\n ended_at TEXT,\n duration_ms INTEGER,\n tool_call_count INTEGER NOT NULL DEFAULT 0,\n thinking_ms INTEGER NOT NULL DEFAULT 0,\n fact_id_start INTEGER\n );\n CREATE INDEX IF NOT EXISTS idx_turn_session ON turn(session_id, turn_index);\n\n CREATE TABLE IF NOT EXISTS permission_phase (\n phase_id INTEGER PRIMARY KEY,\n session_id TEXT NOT NULL,\n from_mode TEXT,\n to_mode TEXT NOT NULL,\n trigger TEXT NOT NULL,\n t_valid TEXT NOT NULL,\n t_invalid TEXT,\n fact_id INTEGER\n );\n CREATE INDEX IF NOT EXISTS idx_phase_session ON permission_phase(session_id, t_valid);\n\n CREATE TABLE IF NOT EXISTS human_edit (\n edit_id INTEGER PRIMARY KEY,\n session_id TEXT NOT NULL,\n file_path TEXT NOT NULL,\n ts TEXT NOT NULL,\n fact_id INTEGER,\n UNIQUE (session_id, file_path, ts)\n );\n\n CREATE TABLE IF NOT EXISTS file_checkpoint (\n checkpoint_id INTEGER PRIMARY KEY,\n session_id TEXT NOT NULL,\n file_path TEXT NOT NULL,\n backup_file_name TEXT NOT NULL,\n version INTEGER NOT NULL,\n backup_time TEXT NOT NULL,\n fact_id INTEGER,\n UNIQUE (session_id, file_path, backup_file_name)\n );\n\n CREATE TABLE IF NOT EXISTS pr (\n pr_ref TEXT PRIMARY KEY, number INTEGER, repo TEXT, title TEXT, state TEXT, url TEXT, merged_at TEXT\n );\n CREATE TABLE IF NOT EXISTS branch (\n branch_ref TEXT PRIMARY KEY, repo TEXT, name TEXT NOT NULL, base TEXT, created_at TEXT, deleted_at TEXT\n );\n CREATE TABLE IF NOT EXISTS file (\n file_ref TEXT PRIMARY KEY, repo TEXT, path TEXT NOT NULL\n );\n CREATE TABLE IF NOT EXISTS task (\n task_ref TEXT PRIMARY KEY, task_id TEXT NOT NULL, initiative TEXT, title TEXT, status TEXT\n );\n CREATE TABLE IF NOT EXISTS subagent (\n agent_ref TEXT PRIMARY KEY,\n session_id TEXT,\n child_session_id TEXT,\n parent_agent_ref TEXT,\n agent_type TEXT,\n label TEXT,\n started_at TEXT,\n ended_at TEXT,\n fact_id INTEGER\n );\n CREATE INDEX IF NOT EXISTS idx_subagent_child ON subagent(child_session_id);\n CREATE TABLE IF NOT EXISTS artifact (\n artifact_ref TEXT PRIMARY KEY, kind TEXT, title TEXT, url TEXT, path TEXT, created_at TEXT\n );\n CREATE TABLE IF NOT EXISTS pr_merge_observation (\n number INTEGER NOT NULL, repo_hint TEXT, merged_at TEXT NOT NULL,\n PRIMARY KEY (number, repo_hint, merged_at)\n );\n CREATE TABLE IF NOT EXISTS pr_create_observation (\n tool_use_id TEXT PRIMARY KEY, title TEXT, number INTEGER, repo TEXT, url TEXT\n );\n";
|
|
41
41
|
/** Every derived table, in an order safe to clear. The watermark table is not derived. */
|
|
42
|
-
declare const DERIVED_TABLES: readonly ["normalized_span", "normalized_event", "normalized_source", "search_span", "edge", "turn", "permission_phase", "human_edit", "file_checkpoint", "subagent", "session_model_usage", "session", "fact", "pr", "pr_merge_observation", "pr_create_observation", "branch", "file", "task", "artifact"];
|
|
42
|
+
declare const DERIVED_TABLES: readonly ["normalized_span", "normalized_event", "normalized_source", "search_span", "edge", "turn", "permission_phase", "human_edit", "file_checkpoint", "subagent", "request", "tool_call", "inbound", "context_block", "compaction", "queue_op", "session_signal", "cost_state_observation", "transcript_facet", "session_origin", "session_external_event", "episode", "session_model_usage", "session", "fact", "pr", "pr_merge_observation", "pr_create_observation", "branch", "file", "task", "artifact"];
|
|
43
43
|
declare const MIGRATIONS: Migration[];
|
|
44
44
|
|
|
45
|
+
/** Recorded in `_migration`; store-sqlite refuses a database whose applied name differs, so never rename it. */
|
|
46
|
+
declare const AUDIT_MIGRATION_NAME = "audit tables";
|
|
47
|
+
/** Tables written per transcript by `applyAudit` and dropped by `purgeTranscript`. */
|
|
48
|
+
declare const AUDIT_TABLES: readonly ["request", "tool_call", "inbound", "context_block", "compaction", "queue_op", "session_signal", "cost_state_observation"];
|
|
49
|
+
/** Per-transcript read state for each audit facet; cleared with the audit rows so a re-read starts clean. */
|
|
50
|
+
declare const FACET_TABLE = "transcript_facet";
|
|
51
|
+
declare const AUDIT_DDL = "\n CREATE TABLE IF NOT EXISTS request (\n transcript_id INTEGER NOT NULL,\n request_id TEXT NOT NULL,\n byte_offset INTEGER NOT NULL, -- first line seen for this request\n session_id TEXT NOT NULL,\n message_id TEXT,\n ts TEXT NOT NULL,\n model TEXT NOT NULL,\n input_tokens INTEGER NOT NULL DEFAULT 0,\n cache_read_tokens INTEGER NOT NULL DEFAULT 0,\n cache_creation_tokens INTEGER NOT NULL DEFAULT 0,\n cache_creation_5m INTEGER NOT NULL DEFAULT 0,\n cache_creation_1h INTEGER NOT NULL DEFAULT 0,\n output_tokens INTEGER NOT NULL DEFAULT 0,\n thinking_tokens INTEGER NOT NULL DEFAULT 0,\n context_tokens INTEGER NOT NULL DEFAULT 0, -- input + cache_read + cache_creation\n service_tier TEXT,\n is_sidechain INTEGER NOT NULL DEFAULT 0,\n -- rollup-owned, recomputed for touched sessions:\n seq_in_session INTEGER,\n gap_ms INTEGER, -- since previous request in the session\n ctx_delta INTEGER, -- context_tokens - (prev.context_tokens + prev.output_tokens)\n wake_cause TEXT,\n wake_delivery TEXT,\n wake_detail TEXT, -- tool name, channel sender, or NULL\n wake_tool_family TEXT,\n wake_mcp_server TEXT,\n wake_offset INTEGER, -- byte_offset of the inbound row\n PRIMARY KEY (transcript_id, request_id)\n ) WITHOUT ROWID;\n CREATE INDEX IF NOT EXISTS idx_request_session_ts ON request(session_id, ts);\n CREATE INDEX IF NOT EXISTS idx_request_ts ON request(ts);\n CREATE INDEX IF NOT EXISTS idx_request_id ON request(request_id);\n\n CREATE TABLE IF NOT EXISTS tool_call (\n transcript_id INTEGER NOT NULL, byte_offset INTEGER NOT NULL, block_index INTEGER NOT NULL,\n session_id TEXT NOT NULL, ts TEXT NOT NULL, tool_use_id TEXT NOT NULL,\n name TEXT NOT NULL, family TEXT NOT NULL, mcp_server TEXT, input_chars INTEGER NOT NULL,\n PRIMARY KEY (transcript_id, byte_offset, block_index)\n ) WITHOUT ROWID;\n CREATE INDEX IF NOT EXISTS idx_tool_call_use ON tool_call(tool_use_id);\n\n CREATE TABLE IF NOT EXISTS inbound (\n transcript_id INTEGER NOT NULL, byte_offset INTEGER NOT NULL, block_index INTEGER NOT NULL,\n session_id TEXT NOT NULL, ts TEXT NOT NULL,\n cause TEXT NOT NULL, delivery TEXT NOT NULL, detail TEXT,\n origin_server TEXT, from_name TEXT, msg_id TEXT, tool_use_id TEXT,\n is_error INTEGER NOT NULL DEFAULT 0, content_hash TEXT NOT NULL, chars INTEGER NOT NULL,\n queued_ms INTEGER, -- rollup-owned\n PRIMARY KEY (transcript_id, byte_offset, block_index)\n ) WITHOUT ROWID;\n CREATE INDEX IF NOT EXISTS idx_inbound_session ON inbound(session_id, byte_offset);\n\n CREATE TABLE IF NOT EXISTS context_block (\n transcript_id INTEGER NOT NULL, byte_offset INTEGER NOT NULL, block_index INTEGER NOT NULL,\n session_id TEXT NOT NULL, ts TEXT NOT NULL, source TEXT NOT NULL,\n tool_use_id TEXT, attachment_type TEXT, chars INTEGER NOT NULL, is_media INTEGER NOT NULL DEFAULT 0,\n PRIMARY KEY (transcript_id, byte_offset, block_index)\n ) WITHOUT ROWID;\n CREATE INDEX IF NOT EXISTS idx_context_block_session ON context_block(session_id, source);\n\n CREATE TABLE IF NOT EXISTS compaction (\n transcript_id INTEGER NOT NULL, byte_offset INTEGER NOT NULL,\n session_id TEXT NOT NULL, ts TEXT NOT NULL, trigger TEXT,\n pre_tokens INTEGER, post_tokens INTEGER, dropped_tokens INTEGER, duration_ms INTEGER,\n mid_loop INTEGER, -- rollup-owned: previous inbound was a tool_result\n PRIMARY KEY (transcript_id, byte_offset)\n ) WITHOUT ROWID;\n\n CREATE TABLE IF NOT EXISTS queue_op (\n transcript_id INTEGER NOT NULL, byte_offset INTEGER NOT NULL,\n session_id TEXT NOT NULL, ts TEXT NOT NULL, operation TEXT NOT NULL,\n content_hash TEXT, origin_server TEXT,\n PRIMARY KEY (transcript_id, byte_offset)\n ) WITHOUT ROWID;\n CREATE INDEX IF NOT EXISTS idx_queue_op_hash ON queue_op(session_id, content_hash);\n\n CREATE TABLE IF NOT EXISTS session_signal (\n transcript_id INTEGER NOT NULL, byte_offset INTEGER NOT NULL, block_index INTEGER NOT NULL,\n session_id TEXT NOT NULL, ts TEXT NOT NULL, signal TEXT NOT NULL, detail TEXT, tool_use_id TEXT,\n -- one Bash block can emit up to three signals (commit, push, pr_create); the signal name disambiguates them\n PRIMARY KEY (transcript_id, byte_offset, block_index, signal)\n ) WITHOUT ROWID;\n CREATE INDEX IF NOT EXISTS idx_signal_session ON session_signal(session_id, ts);\n\n CREATE TABLE IF NOT EXISTS cost_state_observation (\n transcript_id INTEGER NOT NULL, byte_offset INTEGER NOT NULL,\n session_id TEXT NOT NULL, ts TEXT, total_cost_usd REAL NOT NULL, model_usage TEXT NOT NULL,\n PRIMARY KEY (transcript_id, byte_offset)\n ) WITHOUT ROWID;\n\n CREATE TABLE IF NOT EXISTS transcript_facet (\n transcript_id INTEGER NOT NULL, facet TEXT NOT NULL,\n version INTEGER NOT NULL, indexed_to INTEGER NOT NULL,\n PRIMARY KEY (transcript_id, facet)\n ) WITHOUT ROWID;\n";
|
|
52
|
+
|
|
53
|
+
/** Recorded in `_migration`; store-sqlite refuses a database whose applied name differs, so never rename it. */
|
|
54
|
+
declare const ORIGIN_MIGRATION_NAME = "origin, episodes, prices";
|
|
55
|
+
/** Re-derivable from their own sources rather than from transcripts, so `resetIndex` clears them and the next pass refills them. */
|
|
56
|
+
declare const ORIGIN_TABLES: readonly ["session_origin", "session_external_event"];
|
|
57
|
+
/** Written from transcript offsets by `replaceEpisodes`; a rewritten transcript invalidates them, so purge drops them by session. */
|
|
58
|
+
declare const EPISODE_TABLE = "episode";
|
|
59
|
+
declare const ORIGIN_DDL = "\n CREATE TABLE IF NOT EXISTS session_origin ( -- not derived from transcripts\n session_id TEXT PRIMARY KEY,\n origin_system TEXT NOT NULL, -- 'agent-chat'\n agent_id TEXT, agent_name TEXT, parent_name TEXT, parent_session_id TEXT,\n profile TEXT, model_alias TEXT, surface TEXT, isolation TEXT, depth INTEGER,\n origin_kind TEXT, -- spawned | adopted | inherited | human\n config_dir TEXT, spawn_cwd TEXT, spawned_at TEXT,\n brief_chars INTEGER, brief_excerpt TEXT, brief_path TEXT, launch_args TEXT,\n resolved_at TEXT NOT NULL\n );\n CREATE INDEX IF NOT EXISTS idx_origin_parent ON session_origin(parent_session_id);\n\n CREATE TABLE IF NOT EXISTS session_external_event ( -- teleport, handoff, retire, exit\n session_id TEXT NOT NULL, ts TEXT NOT NULL, origin_system TEXT NOT NULL,\n kind TEXT NOT NULL, detail TEXT,\n PRIMARY KEY (session_id, ts, kind)\n ) WITHOUT ROWID;\n\n CREATE TABLE IF NOT EXISTS episode ( -- provisional, see section 9\n session_id TEXT NOT NULL, episode_index INTEGER NOT NULL,\n heuristic TEXT NOT NULL, heuristic_version INTEGER NOT NULL,\n started_at TEXT NOT NULL, ended_at TEXT NOT NULL,\n start_offset INTEGER NOT NULL, end_offset INTEGER NOT NULL,\n opened_by TEXT NOT NULL, -- brief | channel_followup | idle_gap | compaction\n assignment_offset INTEGER, first_deliverable_offset INTEGER, first_deliverable_signal TEXT,\n first_status_offset INTEGER,\n PRIMARY KEY (session_id, heuristic, episode_index)\n ) WITHOUT ROWID;\n\n CREATE TABLE IF NOT EXISTS price (\n model TEXT NOT NULL, effective_from TEXT NOT NULL, table_version INTEGER NOT NULL,\n input_usd_mtok REAL NOT NULL, cache_read_usd_mtok REAL NOT NULL,\n cache_write_5m_usd_mtok REAL NOT NULL, cache_write_1h_usd_mtok REAL NOT NULL,\n output_usd_mtok REAL NOT NULL, source TEXT,\n PRIMARY KEY (model, effective_from)\n ) WITHOUT ROWID;\n";
|
|
60
|
+
declare const ORIGIN_VIEWS: readonly ["\n CREATE VIEW IF NOT EXISTS request_dedup AS\n SELECT transcript_id, request_id, byte_offset, session_id, message_id, ts, model,\n input_tokens, cache_read_tokens, cache_creation_tokens, cache_creation_5m, cache_creation_1h,\n output_tokens, thinking_tokens, context_tokens, service_tier, is_sidechain,\n seq_in_session, gap_ms, ctx_delta, wake_cause, wake_delivery, wake_detail,\n wake_tool_family, wake_mcp_server, wake_offset FROM (\n SELECT r.*, ROW_NUMBER() OVER (PARTITION BY r.request_id ORDER BY r.ts, r.transcript_id) AS copy_rank\n FROM request r\n ) WHERE copy_rank = 1;\n", "\n CREATE VIEW IF NOT EXISTS request_cost AS\n WITH candidate AS (\n SELECT d.request_id, p.model AS price_model, p.effective_from AS price_effective_from,\n p.input_usd_mtok, p.cache_read_usd_mtok, p.cache_write_5m_usd_mtok, p.cache_write_1h_usd_mtok, p.output_usd_mtok,\n ROW_NUMBER() OVER (PARTITION BY d.request_id ORDER BY length(p.model) DESC, p.effective_from DESC) AS match_rank\n FROM request_dedup d\n JOIN price p ON substr(d.model, 1, length(p.model)) = p.model AND p.effective_from <= d.ts\n ), component AS (\n SELECT d.*, c.price_model, c.price_effective_from, c.price_model IS NOT NULL AS priced,\n COALESCE(d.input_tokens * c.input_usd_mtok, 0) / 1e6 AS input_cost_usd,\n COALESCE(d.cache_read_tokens * c.cache_read_usd_mtok, 0) / 1e6 AS cache_read_cost_usd,\n COALESCE(CASE WHEN d.cache_creation_5m + d.cache_creation_1h = 0 THEN d.cache_creation_tokens ELSE d.cache_creation_5m END\n * c.cache_write_5m_usd_mtok, 0) / 1e6 AS cache_write_5m_cost_usd,\n COALESCE(d.cache_creation_1h * c.cache_write_1h_usd_mtok, 0) / 1e6 AS cache_write_1h_cost_usd,\n COALESCE(d.output_tokens * c.output_usd_mtok, 0) / 1e6 AS output_cost_usd\n FROM request_dedup d\n LEFT JOIN candidate c ON c.request_id = d.request_id AND c.match_rank = 1\n )\n SELECT *,\n input_cost_usd + cache_read_cost_usd + cache_write_5m_cost_usd + cache_write_1h_cost_usd + output_cost_usd AS cost_usd,\n (cache_read_tokens < 0.2 * context_tokens AND cache_creation_tokens >= 20000) AS is_cold,\n CASE WHEN context_tokens < 50000 THEN '<50k' WHEN context_tokens < 100000 THEN '50-100k'\n WHEN context_tokens < 200000 THEN '100-200k' ELSE '200k+' END AS context_band,\n CASE WHEN gap_ms IS NULL THEN NULL WHEN gap_ms < 300000 THEN '<5m'\n WHEN gap_ms < 3600000 THEN '5-60m' ELSE '>60m' END AS gap_band\n FROM component;\n", "\n CREATE VIEW IF NOT EXISTS context_contribution AS\n WITH stream AS (\n SELECT transcript_id, byte_offset, 0 AS lane, block_index, NULL AS request_id FROM context_block\n UNION ALL\n SELECT transcript_id, byte_offset, -1 AS lane, 0, request_id FROM request\n ), ordered AS (\n SELECT *, COUNT(request_id) OVER (PARTITION BY transcript_id ORDER BY byte_offset, lane, block_index ROWS UNBOUNDED PRECEDING) AS requests_before\n FROM stream\n ), owner AS (\n SELECT b.transcript_id, b.byte_offset, b.block_index, r.request_id\n FROM ordered b JOIN ordered r ON r.transcript_id = b.transcript_id AND r.lane = -1 AND r.requests_before = b.requests_before + 1\n WHERE b.lane = 0\n ), tool AS (\n SELECT tool_use_id, name, family, mcp_server,\n ROW_NUMBER() OVER (PARTITION BY tool_use_id ORDER BY transcript_id, byte_offset, block_index) AS copy_rank\n FROM tool_call\n )\n SELECT cb.transcript_id, cb.byte_offset, cb.block_index, cb.session_id, cb.ts, cb.source,\n cb.tool_use_id, cb.attachment_type, cb.chars, cb.is_media, o.request_id, d.ctx_delta,\n CASE WHEN cb.source IN ('assistant_text', 'assistant_thinking', 'assistant_tool_input') THEN NULL\n ELSE d.ctx_delta * cb.chars * 1.0 / NULLIF(SUM(CASE WHEN cb.source IN ('assistant_text', 'assistant_thinking', 'assistant_tool_input') THEN 0 ELSE cb.chars END)\n OVER (PARTITION BY o.transcript_id, o.request_id), 0) END AS est_tokens,\n t.name AS tool_name, t.family AS tool_family, t.mcp_server\n FROM context_block cb\n JOIN owner o USING (transcript_id, byte_offset, block_index)\n JOIN request_dedup d ON d.transcript_id = o.transcript_id AND d.request_id = o.request_id\n LEFT JOIN tool t ON t.tool_use_id = cb.tool_use_id AND t.copy_rank = 1;\n"];
|
|
61
|
+
|
|
62
|
+
/** Write one delta's audit events. Runs inside `applyDelta`'s transaction, so it opens none of its own. */
|
|
63
|
+
declare function applyAudit(db: Db, transcriptId: number, delta: TranscriptDelta): void;
|
|
64
|
+
|
|
65
|
+
/** The one facet today: the eight audit tables, all written from the same extraction. */
|
|
66
|
+
declare const AUDIT_FACET = "audit";
|
|
67
|
+
/** Transcripts re-extracted per pass, so a daemon's pass stays short while the backlog drains. */
|
|
68
|
+
declare const DEFAULT_FACET_LIMIT = 40;
|
|
69
|
+
interface BackfillOptions {
|
|
70
|
+
/** Transcripts visited this pass. `Infinity` clears the whole backlog. */
|
|
71
|
+
limit?: number;
|
|
72
|
+
/** The extraction version a facet must reach. Defaults to session-read's `EXTRACT_VERSION`. */
|
|
73
|
+
version?: number;
|
|
74
|
+
}
|
|
75
|
+
interface BackfillSummary {
|
|
76
|
+
backfilled: number;
|
|
77
|
+
/** Stale transcripts left for later passes. */
|
|
78
|
+
backlog: number;
|
|
79
|
+
/** Sessions whose audit rows were rebuilt, for the caller's rollup. */
|
|
80
|
+
sessionIds: string[];
|
|
81
|
+
}
|
|
82
|
+
/**
|
|
83
|
+
* Bring the audit facet of already-indexed transcripts up to `version` without a
|
|
84
|
+
* full rebuild: re-read each one up to its watermark and replace only its audit
|
|
85
|
+
* rows. Legacy tables are never written, so their accumulating counts cannot double.
|
|
86
|
+
*/
|
|
87
|
+
declare function backfillFacets(graph: SessionGraph, transcripts: readonly DiscoveredTranscript[], options?: BackfillOptions): Promise<BackfillSummary>;
|
|
88
|
+
|
|
45
89
|
/**
|
|
46
90
|
* Apply one transcript chunk's delta in a single transaction, so a crash can
|
|
47
91
|
* never leave rows committed behind a stale watermark or ahead of their rows.
|
|
48
|
-
* Append-only tables are idempotent through unique indexes; session
|
|
49
|
-
*
|
|
50
|
-
*
|
|
92
|
+
* Append-only tables are idempotent through unique indexes; session rows
|
|
93
|
+
* accumulate; usage is left to the rollup, which recomputes it from `request`;
|
|
94
|
+
* ref-keyed assets insert-if-absent with COALESCE merges that mirror
|
|
95
|
+
* `EventFolder`'s rules.
|
|
51
96
|
*/
|
|
52
|
-
declare function applyDelta(graph: SessionGraph, transcriptId: number, delta: TranscriptDelta): void;
|
|
97
|
+
declare function applyDelta(graph: SessionGraph, transcriptId: number, delta: TranscriptDelta, source?: DeltaSource): void;
|
|
98
|
+
/** What discovery knows about the file that the lines themselves do not carry. */
|
|
99
|
+
interface DeltaSource {
|
|
100
|
+
/** The Claude config dir the transcript was found under; `null` when discovery did not say. */
|
|
101
|
+
account?: string | null;
|
|
102
|
+
}
|
|
53
103
|
|
|
54
104
|
/**
|
|
55
105
|
* Drop every derived row one transcript produced, so re-reading it from byte 0
|
|
@@ -68,10 +118,13 @@ declare function applyDelta(graph: SessionGraph, transcriptId: number, delta: Tr
|
|
|
68
118
|
* inserts are insert-if-absent; a single transcript cannot know whether it was
|
|
69
119
|
* the only source. The two observation tables are left for the same reason —
|
|
70
120
|
* `reconcile` folds them, and re-reading re-asserts the same rows.
|
|
121
|
+
*
|
|
122
|
+
* Audit rows and facet state carry `transcript_id` themselves, so they go by it
|
|
123
|
+
* directly rather than through `session`.
|
|
71
124
|
*/
|
|
72
125
|
declare function purgeTranscript(graph: SessionGraph, transcriptId: number): void;
|
|
73
126
|
|
|
74
|
-
/** Recompute turn aggregates for the given sessions in one transaction. */
|
|
127
|
+
/** Recompute turn aggregates and the audit rollup for the given sessions in one transaction. */
|
|
75
128
|
declare function rollupSessions(graph: SessionGraph, sessionIds: readonly string[]): number;
|
|
76
129
|
interface ReconcileCounts {
|
|
77
130
|
prMerges: number;
|
|
@@ -81,6 +134,74 @@ interface ReconcileCounts {
|
|
|
81
134
|
/** Whole-table reconciliations, run once at the end of a refresh pass after the rollup. */
|
|
82
135
|
declare function reconcile(graph: SessionGraph): ReconcileCounts;
|
|
83
136
|
|
|
137
|
+
/**
|
|
138
|
+
* Who started a session and why, as a launcher such as agent-chat recorded it.
|
|
139
|
+
* A transcript cannot state this: a `claude -p` worker and a headless miner look
|
|
140
|
+
* the same from inside. Every field but `originSystem` is optional.
|
|
141
|
+
*/
|
|
142
|
+
interface ResolvedOrigin {
|
|
143
|
+
originSystem: string;
|
|
144
|
+
agentId?: string | null;
|
|
145
|
+
agentName?: string | null;
|
|
146
|
+
parentName?: string | null;
|
|
147
|
+
parentSessionId?: string | null;
|
|
148
|
+
profile?: string | null;
|
|
149
|
+
modelAlias?: string | null;
|
|
150
|
+
surface?: string | null;
|
|
151
|
+
isolation?: string | null;
|
|
152
|
+
depth?: number | null;
|
|
153
|
+
/** `spawned`, `adopted`, `inherited` or `human`. */
|
|
154
|
+
originKind?: string | null;
|
|
155
|
+
configDir?: string | null;
|
|
156
|
+
spawnCwd?: string | null;
|
|
157
|
+
spawnedAt?: string | null;
|
|
158
|
+
briefChars?: number | null;
|
|
159
|
+
briefExcerpt?: string | null;
|
|
160
|
+
briefPath?: string | null;
|
|
161
|
+
launchArgs?: string | null;
|
|
162
|
+
}
|
|
163
|
+
/** A lifecycle event the launcher saw outside the transcript: `teleport`, `handoff`, `retired`, `exited`. */
|
|
164
|
+
interface ExternalEvent {
|
|
165
|
+
sessionId: string;
|
|
166
|
+
ts: string;
|
|
167
|
+
kind: string;
|
|
168
|
+
detail?: string | null;
|
|
169
|
+
/** Defaults to the session's resolved origin system. */
|
|
170
|
+
originSystem?: string | null;
|
|
171
|
+
}
|
|
172
|
+
interface OriginResolution {
|
|
173
|
+
/** Keyed by session id. */
|
|
174
|
+
origins: Record<string, ResolvedOrigin>;
|
|
175
|
+
externalEvents?: readonly ExternalEvent[];
|
|
176
|
+
}
|
|
177
|
+
/**
|
|
178
|
+
* Supplied by the caller, never by this package: session-graph is tier 2 and
|
|
179
|
+
* must not learn where a launcher keeps its records. Called once per pass with
|
|
180
|
+
* every session that has no origin row or a stale one.
|
|
181
|
+
*/
|
|
182
|
+
type OriginResolver = (sessionIds: readonly string[]) => PromiseLike<OriginResolution> | OriginResolution;
|
|
183
|
+
interface OriginEnrichment {
|
|
184
|
+
/** Session ids handed to the resolver. */
|
|
185
|
+
requested: number;
|
|
186
|
+
/** `session_origin` rows the resolver wrote. */
|
|
187
|
+
applied: number;
|
|
188
|
+
/** `session_external_event` rows the resolver wrote. */
|
|
189
|
+
events: number;
|
|
190
|
+
/** The resolver threw; origin rows stand as the last pass left them. */
|
|
191
|
+
failed: boolean;
|
|
192
|
+
/** Why it failed, for the caller to log. Absent unless `failed`. */
|
|
193
|
+
error?: string;
|
|
194
|
+
}
|
|
195
|
+
declare const NO_ORIGINS: OriginEnrichment;
|
|
196
|
+
declare function sessionsNeedingOrigin(graph: SessionGraph): string[];
|
|
197
|
+
/**
|
|
198
|
+
* Resolve origins for the sessions that lack a current one, then project every
|
|
199
|
+
* stored origin into `spawned` edges and `subagent` rows. Projecting from the
|
|
200
|
+
* table, not from this pass's answer, restores rows a parent transcript's purge
|
|
201
|
+
* removed. A resolver that throws costs this pass its origins and nothing else.
|
|
202
|
+
*/
|
|
203
|
+
declare function resolveOrigins(graph: SessionGraph, resolver: OriginResolver | undefined): Promise<OriginEnrichment>;
|
|
204
|
+
|
|
84
205
|
/**
|
|
85
206
|
* What a product's task store knows and a transcript cannot state: the task's
|
|
86
207
|
* present title, its initiative, and the status it holds right now rather than
|
|
@@ -129,6 +250,14 @@ interface IndexOptions {
|
|
|
129
250
|
withContentHash?: boolean;
|
|
130
251
|
/** Fill task title, initiative and present status from the caller's store. Absent means transcripts alone. */
|
|
131
252
|
resolveTasks?: TaskResolver;
|
|
253
|
+
/** Record who launched each session. Runs once per `refreshCorpus` pass, never per transcript. Absent means no origin rows. */
|
|
254
|
+
resolveOrigins?: OriginResolver;
|
|
255
|
+
}
|
|
256
|
+
interface RefreshOptions extends IndexOptions {
|
|
257
|
+
/** Roll up every session, not just the ones this pass touched. */
|
|
258
|
+
full?: boolean;
|
|
259
|
+
/** Stale audit facets re-extracted per pass (default 40). `Infinity` clears the backlog. */
|
|
260
|
+
facetLimit?: number;
|
|
132
261
|
}
|
|
133
262
|
type TranscriptOutcome = {
|
|
134
263
|
status: "unchanged" | "indexed" | "rewound";
|
|
@@ -161,22 +290,25 @@ interface RefreshSummary {
|
|
|
161
290
|
turnsRolledUp: number;
|
|
162
291
|
reconciled: ReconcileCounts;
|
|
163
292
|
tasks: TaskEnrichment;
|
|
293
|
+
origins: OriginEnrichment;
|
|
164
294
|
markedMissing: number;
|
|
295
|
+
facetsBackfilled: number;
|
|
296
|
+
/** Transcripts whose audit facet is still stale after this pass. */
|
|
297
|
+
facetBacklog: number;
|
|
165
298
|
}
|
|
166
299
|
/**
|
|
167
|
-
* One pass over a corpus: index every transcript,
|
|
168
|
-
*
|
|
169
|
-
*
|
|
170
|
-
*
|
|
300
|
+
* One pass over a corpus: index every transcript, re-extract a bounded batch of
|
|
301
|
+
* stale audit facets, roll up the sessions that changed, resolve session
|
|
302
|
+
* origins, reconcile cross-transcript observations, enrich tasks, and mark
|
|
303
|
+
* rows whose source file is gone. Idempotent: a second pass over unchanged
|
|
304
|
+
* files changes nothing.
|
|
171
305
|
*
|
|
172
306
|
* The resolver is hoisted out of the per-transcript loop and run once over the
|
|
173
307
|
* whole task table, so a corpus of N transcripts costs one resolver call rather
|
|
174
308
|
* than N, and a task whose store row changed refreshes even when no transcript
|
|
175
309
|
* did. That bounds staleness to one pass.
|
|
176
310
|
*/
|
|
177
|
-
declare function refreshCorpus(graph: SessionGraph, transcripts: readonly DiscoveredTranscript[], options?:
|
|
178
|
-
full?: boolean;
|
|
179
|
-
}): Promise<RefreshSummary>;
|
|
311
|
+
declare function refreshCorpus(graph: SessionGraph, transcripts: readonly DiscoveredTranscript[], options?: RefreshOptions): Promise<RefreshSummary>;
|
|
180
312
|
|
|
181
313
|
interface NormalizedIndexResult {
|
|
182
314
|
status: "indexed" | "unchanged" | "missing" | "quarantined";
|
|
@@ -222,4 +354,4 @@ declare function normalizedUsage(graph: SessionGraph, ref: string): NormalizedUs
|
|
|
222
354
|
/** Ambiguity is explicit; workspace session bodies remain in their original namespace. */
|
|
223
355
|
declare function resolveConversationAlias(db: Db, legacyRef: string): string | null;
|
|
224
356
|
|
|
225
|
-
export { type ConversationSummary, DERIVED_TABLES, DOMAIN_DDL, type IndexOptions, type IndexedSpan, KIT, MIGRATIONS, NO_ENRICHMENT, type NormalizedIndexResult, type NormalizedUsageSummary, type OpenSessionGraphOptions, type ReconcileCounts, type RefreshSummary, type ResolvedTask, type SessionGraph, type TaskEnrichment, type TaskResolution, type TaskResolver, type TranscriptOutcome, allSessionIds, allTaskIds, applyDelta, enrichTasks, indexCodexSource, indexTranscript, normalizedSessions, normalizedUsage, openSessionGraph, purgeTranscript, readIndexedText, reconcile, refreshCorpus, resetIndex, resolveConversationAlias, rollupSessions };
|
|
357
|
+
export { AUDIT_DDL, AUDIT_FACET, AUDIT_MIGRATION_NAME, AUDIT_TABLES, type BackfillOptions, type BackfillSummary, type ConversationSummary, DEFAULT_FACET_LIMIT, DERIVED_TABLES, DOMAIN_DDL, type DeltaSource, EPISODE_TABLE, type ExternalEvent, FACET_TABLE, type IndexOptions, type IndexedSpan, KIT, MIGRATIONS, NO_ENRICHMENT, NO_ORIGINS, type NormalizedIndexResult, type NormalizedUsageSummary, ORIGIN_DDL, ORIGIN_MIGRATION_NAME, ORIGIN_TABLES, ORIGIN_VIEWS, type OpenSessionGraphOptions, type OriginEnrichment, type OriginResolution, type OriginResolver, type ReconcileCounts, type RefreshOptions, type RefreshSummary, type ResolvedOrigin, type ResolvedTask, type SessionGraph, type TaskEnrichment, type TaskResolution, type TaskResolver, type TranscriptOutcome, allSessionIds, allTaskIds, applyAudit, applyDelta, backfillFacets, enrichTasks, indexCodexSource, indexTranscript, normalizedSessions, normalizedUsage, openSessionGraph, purgeTranscript, readIndexedText, reconcile, refreshCorpus, resetIndex, resolveConversationAlias, resolveOrigins, rollupSessions, sessionsNeedingOrigin };
|