eklavya 1.18.2 → 1.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/dashboard.html +559 -14
- package/dist/assets/tutor/references/focus-and-level.md +29 -0
- package/dist/cli.js +648 -20
- package/dist/cli.js.map +1 -1
- package/dist/config-path.js +156 -0
- package/dist/config-path.js.map +1 -0
- package/dist/config.js +231 -2
- package/dist/config.js.map +1 -1
- package/dist/dashboard.js +355 -0
- package/dist/dashboard.js.map +1 -1
- package/dist/eval/retrieval-score.js +77 -0
- package/dist/eval/retrieval-score.js.map +1 -0
- package/dist/hooks/capture-tool.js +107 -0
- package/dist/hooks/capture-tool.js.map +1 -0
- package/dist/hooks/checkpoint-quiz.js +5 -6
- package/dist/hooks/checkpoint-quiz.js.map +1 -1
- package/dist/hooks/lib.js +7 -33
- package/dist/hooks/lib.js.map +1 -1
- package/dist/hooks/memory-lib.js +181 -0
- package/dist/hooks/memory-lib.js.map +1 -0
- package/dist/hooks/prompt-submit-nudge.js +56 -20
- package/dist/hooks/prompt-submit-nudge.js.map +1 -1
- package/dist/hooks/session-start.js +73 -74
- package/dist/hooks/session-start.js.map +1 -1
- package/dist/hooks/stop-quiz-check.js +44 -7
- package/dist/hooks/stop-quiz-check.js.map +1 -1
- package/dist/install.js +37 -0
- package/dist/install.js.map +1 -1
- package/dist/memory/capture.js +123 -0
- package/dist/memory/capture.js.map +1 -0
- package/dist/memory/code.js +186 -0
- package/dist/memory/code.js.map +1 -0
- package/dist/memory/collections.js +87 -0
- package/dist/memory/collections.js.map +1 -0
- package/dist/memory/embed.js +81 -0
- package/dist/memory/embed.js.map +1 -0
- package/dist/memory/hosts.js +71 -0
- package/dist/memory/hosts.js.map +1 -0
- package/dist/memory/identity.js +74 -0
- package/dist/memory/identity.js.map +1 -0
- package/dist/memory/import.js +916 -0
- package/dist/memory/import.js.map +1 -0
- package/dist/memory/learning.js +148 -0
- package/dist/memory/learning.js.map +1 -0
- package/dist/memory/notify.js +220 -0
- package/dist/memory/notify.js.map +1 -0
- package/dist/memory/privacy.js +117 -0
- package/dist/memory/privacy.js.map +1 -0
- package/dist/memory/provider.js +155 -0
- package/dist/memory/provider.js.map +1 -0
- package/dist/memory/recall.js +327 -0
- package/dist/memory/recall.js.map +1 -0
- package/dist/memory/replay.js +174 -0
- package/dist/memory/replay.js.map +1 -0
- package/dist/memory/search.js +143 -0
- package/dist/memory/search.js.map +1 -0
- package/dist/memory/spool.js +93 -0
- package/dist/memory/spool.js.map +1 -0
- package/dist/memory/store.js +399 -0
- package/dist/memory/store.js.map +1 -0
- package/dist/memory/summarize.js +178 -0
- package/dist/memory/summarize.js.map +1 -0
- package/dist/memory/sync.js +595 -0
- package/dist/memory/sync.js.map +1 -0
- package/dist/memory/tokens.js +55 -0
- package/dist/memory/tokens.js.map +1 -0
- package/dist/memory/worker.js +207 -0
- package/dist/memory/worker.js.map +1 -0
- package/dist/migrations/009_memory.sql +226 -0
- package/dist/migrations/010_import.sql +26 -0
- package/dist/migrations/011_sync.sql +76 -0
- package/dist/migrations/012_job_backoff.sql +13 -0
- package/dist/migrations/013_batch_provenance.sql +18 -0
- package/dist/migrations/014_batch_events_index.sql +18 -0
- package/dist/plugin/.claude-plugin/plugin.json +1 -1
- package/dist/plugin/agents/tutor.md +3 -1
- package/dist/plugin/hooks/CLAUDE.md +30 -6
- package/dist/plugin/hooks/hooks.json +10 -0
- package/dist/plugin/skills/CLAUDE.md +17 -7
- package/dist/plugin/skills/memory/SKILL.md +88 -0
- package/dist/plugin/skills/setup/SKILL.md +16 -2
- package/dist/plugin/skills/tutor/references/focus-and-level.md +29 -0
- package/dist/time.js +41 -0
- package/dist/time.js.map +1 -0
- package/dist/tools/code_tools.js +73 -0
- package/dist/tools/code_tools.js.map +1 -0
- package/dist/tools/collection_tools.js +92 -0
- package/dist/tools/collection_tools.js.map +1 -0
- package/dist/tools/config_tools.js +104 -1
- package/dist/tools/config_tools.js.map +1 -1
- package/dist/tools/index.js +15 -0
- package/dist/tools/index.js.map +1 -1
- package/dist/tools/memory_read_tools.js +277 -0
- package/dist/tools/memory_read_tools.js.map +1 -0
- package/dist/tools/memory_write_tools.js +115 -0
- package/dist/tools/memory_write_tools.js.map +1 -0
- package/dist/user-skill/eklavya/SKILL.md +64 -8
- package/package.json +2 -1
|
@@ -0,0 +1,916 @@
|
|
|
1
|
+
import fs from 'node:fs';
|
|
2
|
+
import os from 'node:os';
|
|
3
|
+
import path from 'node:path';
|
|
4
|
+
import Database from 'better-sqlite3';
|
|
5
|
+
import { nowIso } from '../time.js';
|
|
6
|
+
import { entryUid, eventUid } from './identity.js';
|
|
7
|
+
import { addCandidate, indexVector, insertEntry } from './store.js';
|
|
8
|
+
/**
|
|
9
|
+
* The Claude Mem importer (PRD MIG-01/02).
|
|
10
|
+
*
|
|
11
|
+
* Three rules shape everything below, and each one is a failure someone has
|
|
12
|
+
* already had:
|
|
13
|
+
*
|
|
14
|
+
* 1. The source is never written to. It is opened read-only and snapshotted
|
|
15
|
+
* with `VACUUM INTO`, which folds the WAL in -- a bare file copy of the main
|
|
16
|
+
* database silently drops every transaction still sitting in the log.
|
|
17
|
+
* 2. Nothing imported is assessed. Observations become `memory_entries`, the
|
|
18
|
+
* concepts attached to them become `learning_sources` rows with
|
|
19
|
+
* `status = 'candidate'`, and no attempt, mastery row or gate is touched.
|
|
20
|
+
* Exposure in another tool's history is not evidence that anything is known.
|
|
21
|
+
* 3. An unrecognised newer source schema stops. Guessing at a layout that moved
|
|
22
|
+
* produces an import that looks successful and is wrong, which is worse than
|
|
23
|
+
* one that refuses.
|
|
24
|
+
*/
|
|
25
|
+
export const IMPORT_SOURCE = 'claude-mem';
|
|
26
|
+
/**
|
|
27
|
+
* The newest Claude Mem schema this importer has been read against
|
|
28
|
+
* (`schema_versions` in the source). Anything higher stops rather than guesses;
|
|
29
|
+
* anything lower is imported with whatever columns it has, since the layout has
|
|
30
|
+
* only ever gained columns.
|
|
31
|
+
*/
|
|
32
|
+
export const SUPPORTED_SCHEMA_VERSION = 52;
|
|
33
|
+
/** Tables the importer knows about. Anything else in the source is reported. */
|
|
34
|
+
const KNOWN_TABLES = [
|
|
35
|
+
'observations',
|
|
36
|
+
'session_summaries',
|
|
37
|
+
'user_prompts',
|
|
38
|
+
'tool_uses',
|
|
39
|
+
'sdk_sessions',
|
|
40
|
+
'pending_messages',
|
|
41
|
+
'telegram_wrapups',
|
|
42
|
+
'schema_versions',
|
|
43
|
+
'sync_state',
|
|
44
|
+
'sync_outbox',
|
|
45
|
+
'sync_entity_heads',
|
|
46
|
+
'sync_content_outbox',
|
|
47
|
+
'sync_dead_letter',
|
|
48
|
+
'sync_launch_exclusions',
|
|
49
|
+
];
|
|
50
|
+
/**
|
|
51
|
+
* The field disposition report (MIG-01). Written out by hand rather than
|
|
52
|
+
* derived, because "which source fields were deliberately dropped, and why" is
|
|
53
|
+
* exactly the thing a derived list cannot say.
|
|
54
|
+
*/
|
|
55
|
+
const DISPOSITIONS = [
|
|
56
|
+
// observations -> memory_entries
|
|
57
|
+
m('observations', 'id', 'import_id_map.source_id', 'the resume key; Eklavya ids are its own'),
|
|
58
|
+
m('observations', 'memory_session_id', 'memory_entries.session_id', 'the session the work happened in'),
|
|
59
|
+
m('observations', 'project', 'memory_entries.project', 'kept verbatim as the source named it'),
|
|
60
|
+
m('observations', 'merged_into_project', 'memory_entries.project', 'wins over `project` when set: the source already folded a rename'),
|
|
61
|
+
m('observations', 'type', 'memory_entries.type', ''),
|
|
62
|
+
m('observations', 'title', 'memory_entries.title', ''),
|
|
63
|
+
m('observations', 'subtitle', 'memory_entries.narrative', 'prepended to the narrative; Eklavya has no subtitle column'),
|
|
64
|
+
m('observations', 'narrative', 'memory_entries.narrative', ''),
|
|
65
|
+
m('observations', 'text', 'memory_entries.narrative', 'used when there is no narrative'),
|
|
66
|
+
m('observations', 'facts', 'memory_entries.facts', 'JSON array, carried across as-is'),
|
|
67
|
+
m('observations', 'concepts', "learning_sources (status 'candidate')", 'a proposal, never an accepted concept and never a mastery row'),
|
|
68
|
+
m('observations', 'files_read', 'memory_entries.files', 'unioned with files_modified'),
|
|
69
|
+
m('observations', 'files_modified', 'memory_entries.files', 'unioned with files_read'),
|
|
70
|
+
m('observations', 'created_at_epoch', 'memory_entries.occurred_at', 'the ORIGINAL timestamp; an import is not new work'),
|
|
71
|
+
m('observations', 'created_at', 'memory_entries.occurred_at', 'fallback when the epoch is missing'),
|
|
72
|
+
m('observations', 'generated_by_model', 'memory_entries.generator', 'recorded as the generator, prefixed with the importer'),
|
|
73
|
+
m('observations', 'agent_type', 'memory_entry_tags', 'kept as a tag'),
|
|
74
|
+
d('observations', 'agent_id', 'a foreign run identity with nothing on this machine to join to'),
|
|
75
|
+
m('observations', 'prompt_number', 'memory_entry_events.entry_id', 'with memory_session_id, the pair that finds the prompt an observation came from'),
|
|
76
|
+
d('observations', 'discovery_tokens', "the source's own provider accounting, not a saving Eklavya can attest to"),
|
|
77
|
+
d('observations', 'content_hash', "the source's dedupe key; Eklavya keys on its own deterministic entry_uid"),
|
|
78
|
+
d('observations', 'relevance_count', 'a usage counter for the source ranker, meaningless to a different ranker'),
|
|
79
|
+
d('observations', 'metadata', 'an opaque blob with no agreed shape; importing it would import untyped junk'),
|
|
80
|
+
d('observations', 'synced_at', 'cloud sync state for another installation (MIG-01: no foreign sync metadata into runtime)'),
|
|
81
|
+
d('observations', 'origin_device_id', 'foreign device identity — deliberately never copied into active state'),
|
|
82
|
+
d('observations', 'origin_local_id', 'foreign device identity — deliberately never copied into active state'),
|
|
83
|
+
d('observations', 'sync_rev', 'sync bookkeeping for a service Eklavya does not talk to'),
|
|
84
|
+
// session_summaries -> memory_entries (kind = session_summary)
|
|
85
|
+
m('session_summaries', 'id', 'import_id_map.source_id', 'the resume key'),
|
|
86
|
+
m('session_summaries', 'memory_session_id', 'memory_entries.session_id', ''),
|
|
87
|
+
m('session_summaries', 'project', 'memory_entries.project', ''),
|
|
88
|
+
m('session_summaries', 'merged_into_project', 'memory_entries.project', 'wins over `project` when set'),
|
|
89
|
+
m('session_summaries', 'request', 'memory_entries.title', 'the first line becomes the title, the whole of it a narrative section'),
|
|
90
|
+
m('session_summaries', 'investigated', 'memory_entries.narrative', 'a labelled section'),
|
|
91
|
+
m('session_summaries', 'learned', 'memory_entries.narrative', 'a labelled section'),
|
|
92
|
+
m('session_summaries', 'completed', 'memory_entries.narrative', 'a labelled section'),
|
|
93
|
+
m('session_summaries', 'next_steps', 'memory_entries.narrative', 'a labelled section'),
|
|
94
|
+
m('session_summaries', 'notes', 'memory_entries.narrative', 'a labelled section'),
|
|
95
|
+
m('session_summaries', 'files_read', 'memory_entries.files', 'unioned with files_edited'),
|
|
96
|
+
m('session_summaries', 'files_edited', 'memory_entries.files', 'unioned with files_read'),
|
|
97
|
+
m('session_summaries', 'created_at_epoch', 'memory_entries.occurred_at', 'the ORIGINAL timestamp'),
|
|
98
|
+
m('session_summaries', 'created_at', 'memory_entries.occurred_at', 'fallback when the epoch is missing'),
|
|
99
|
+
d('session_summaries', 'prompt_number', 'an ordering hint; occurred_at already orders'),
|
|
100
|
+
d('session_summaries', 'discovery_tokens', "the source's provider accounting"),
|
|
101
|
+
d('session_summaries', 'synced_at', 'cloud sync state for another installation'),
|
|
102
|
+
d('session_summaries', 'origin_device_id', 'foreign device identity'),
|
|
103
|
+
d('session_summaries', 'origin_local_id', 'foreign device identity'),
|
|
104
|
+
d('session_summaries', 'sync_rev', 'sync bookkeeping'),
|
|
105
|
+
// user_prompts -> evidence_events (kind = prompt)
|
|
106
|
+
m('user_prompts', 'id', 'import_id_map.source_id', 'the resume key'),
|
|
107
|
+
m('user_prompts', 'content_session_id', 'evidence_events.session_id', ''),
|
|
108
|
+
m('user_prompts', 'prompt_text', 'evidence_events.body', ''),
|
|
109
|
+
m('user_prompts', 'prompt_number', 'evidence_events.title', 'rendered as "Prompt #n"'),
|
|
110
|
+
m('user_prompts', 'created_at_epoch', 'evidence_events.occurred_at', 'the ORIGINAL timestamp'),
|
|
111
|
+
m('user_prompts', 'created_at', 'evidence_events.occurred_at', 'fallback when the epoch is missing'),
|
|
112
|
+
m('user_prompts', 'session_db_id', 'evidence_events.project', 'resolved through sdk_sessions; prompts carry no project of their own'),
|
|
113
|
+
d('user_prompts', 'synced_at', 'cloud sync state for another installation'),
|
|
114
|
+
d('user_prompts', 'origin_device_id', 'foreign device identity'),
|
|
115
|
+
d('user_prompts', 'origin_local_id', 'foreign device identity'),
|
|
116
|
+
d('user_prompts', 'sync_rev', 'sync bookkeeping'),
|
|
117
|
+
// tool_uses -> evidence_events (kind = tool_use)
|
|
118
|
+
m('tool_uses', 'id', 'import_id_map.source_id', 'the resume key'),
|
|
119
|
+
m('tool_uses', 'tool_use_id', 'evidence_events.event_uid', 'folded into the deterministic uid'),
|
|
120
|
+
m('tool_uses', 'content_session_id', 'evidence_events.session_id', ''),
|
|
121
|
+
m('tool_uses', 'memory_session_id', 'evidence_events.session_id', 'preferred when present'),
|
|
122
|
+
m('tool_uses', 'project', 'evidence_events.project', ''),
|
|
123
|
+
m('tool_uses', 'tool_name', 'evidence_events.tool', ''),
|
|
124
|
+
m('tool_uses', 'tool_input', 'evidence_events.body', 'with the response, as the recorded turn'),
|
|
125
|
+
m('tool_uses', 'tool_response', 'evidence_events.body', 'with the input, as the recorded turn'),
|
|
126
|
+
m('tool_uses', 'cwd', 'evidence_events.checkout', ''),
|
|
127
|
+
m('tool_uses', 'agent_id', 'evidence_events.agent_id', ''),
|
|
128
|
+
m('tool_uses', 'created_at_epoch', 'evidence_events.occurred_at', 'the ORIGINAL timestamp'),
|
|
129
|
+
m('tool_uses', 'created_at', 'evidence_events.occurred_at', 'fallback when the epoch is missing'),
|
|
130
|
+
d('tool_uses', 'agent_type', 'not modelled on an event; the tool and body already say what ran'),
|
|
131
|
+
d('tool_uses', 'session_db_id', 'an internal source row id with no meaning here'),
|
|
132
|
+
d('tool_uses', 'platform_source', "always 'claude' in practice; Eklavya records host on the event"),
|
|
133
|
+
d('tool_uses', 'prompt_number', 'an ordering hint; occurred_at already orders'),
|
|
134
|
+
m('tool_uses', 'observation_id', 'memory_entry_events.event_id', 'the direct key to the observation, remapped through import_id_map rather than carried over'),
|
|
135
|
+
d('tool_uses', 'or_generation_id', 'provider request correlation for a provider Eklavya did not call'),
|
|
136
|
+
d('tool_uses', 'or_session_id', 'provider request correlation'),
|
|
137
|
+
d('tool_uses', 'content_hash', "the source's dedupe key"),
|
|
138
|
+
// Whole tables that are deliberately not imported.
|
|
139
|
+
d('sdk_sessions', '*', 'read for project and session lookup only; a session is not a row Eklavya stores'),
|
|
140
|
+
d('pending_messages', '*', 'an active worker queue — MIG-01 forbids copying live jobs into runtime state'),
|
|
141
|
+
d('telegram_wrapups', '*', 'delivery state for notifications already sent; importing it could re-send or falsely suppress'),
|
|
142
|
+
d('sync_state', '*', 'cloud sync cursors and credentials-adjacent state for another installation'),
|
|
143
|
+
d('sync_outbox', '*', 'undelivered sync operations belonging to the source install'),
|
|
144
|
+
d('sync_content_outbox', '*', 'undelivered sync payloads belonging to the source install'),
|
|
145
|
+
d('sync_entity_heads', '*', 'foreign device revision heads'),
|
|
146
|
+
d('sync_dead_letter', '*', 'failed sync operations belonging to the source install'),
|
|
147
|
+
d('sync_launch_exclusions', '*', 'sync suppression list belonging to the source install'),
|
|
148
|
+
d('schema_versions', '*', 'read to decide whether this importer understands the source'),
|
|
149
|
+
d('observations_fts', '*', 'a derived index; Eklavya rebuilds its own FTS and vectors on insert'),
|
|
150
|
+
];
|
|
151
|
+
function m(table, field, to, reason) {
|
|
152
|
+
return { table, field, to, kind: 'mapped', reason };
|
|
153
|
+
}
|
|
154
|
+
function d(table, field, reason) {
|
|
155
|
+
return { table, field, to: null, kind: 'dropped', reason };
|
|
156
|
+
}
|
|
157
|
+
/** The four source tables that become Eklavya rows. */
|
|
158
|
+
export const IMPORTED_TABLES = ['observations', 'session_summaries', 'user_prompts', 'tool_uses'];
|
|
159
|
+
/** Fails with a message that names the next step, never a bare assertion. */
|
|
160
|
+
export class ImportError extends Error {
|
|
161
|
+
}
|
|
162
|
+
function openSource(file) {
|
|
163
|
+
if (!fs.existsSync(file)) {
|
|
164
|
+
throw new ImportError(`No Claude Mem database at ${file}. Pass the path to claude-mem.db (usually ~/.claude-mem/claude-mem.db).`);
|
|
165
|
+
}
|
|
166
|
+
// Read-only is the guarantee, not a precaution: MIG-02 forbids mutating the
|
|
167
|
+
// source, and `readonly` makes that true even if a query below is wrong.
|
|
168
|
+
return new Database(file, { readonly: true, fileMustExist: true });
|
|
169
|
+
}
|
|
170
|
+
function tableSet(src) {
|
|
171
|
+
return new Set(src.prepare("SELECT name FROM sqlite_master WHERE type='table'").all().map((r) => r.name));
|
|
172
|
+
}
|
|
173
|
+
function count(src, table) {
|
|
174
|
+
try {
|
|
175
|
+
return src.prepare(`SELECT COUNT(*) AS n FROM "${table}"`).get().n;
|
|
176
|
+
}
|
|
177
|
+
catch {
|
|
178
|
+
return 0;
|
|
179
|
+
}
|
|
180
|
+
}
|
|
181
|
+
function sourceSchemaVersion(src, tables) {
|
|
182
|
+
if (!tables.has('schema_versions'))
|
|
183
|
+
return null;
|
|
184
|
+
const row = src.prepare('SELECT MAX(version) AS v FROM schema_versions').get();
|
|
185
|
+
return row.v ?? null;
|
|
186
|
+
}
|
|
187
|
+
/**
|
|
188
|
+
* Epochs in the source are milliseconds, but a fixture or an older row can hold
|
|
189
|
+
* seconds. Distinguishing them by magnitude is safe for any date this century
|
|
190
|
+
* and costs nothing; guessing wrong would file a 2024 observation in 1970.
|
|
191
|
+
*/
|
|
192
|
+
/**
|
|
193
|
+
* The largest instant a JS `Date` can represent. An epoch past it makes an
|
|
194
|
+
* Invalid Date, and `toISOString()` on one *throws*.
|
|
195
|
+
*
|
|
196
|
+
* That matters more than it looks: this runs per row, inside the import loop,
|
|
197
|
+
* and each row commits in its own transaction — so a single absurd timestamp
|
|
198
|
+
* in a source database would abort the run partway through with a raw stack
|
|
199
|
+
* trace, leaving the rows before it committed and nothing saying so. A source
|
|
200
|
+
* database is a file somebody hands us; it does not get to end the process.
|
|
201
|
+
*/
|
|
202
|
+
const MAX_EPOCH_MS = 8.64e15;
|
|
203
|
+
function isoFromEpoch(epoch, fallback) {
|
|
204
|
+
if (typeof epoch === 'number' && Number.isFinite(epoch) && epoch > 0) {
|
|
205
|
+
const ms = epoch < 1e12 ? epoch * 1000 : epoch;
|
|
206
|
+
if (Math.abs(ms) <= MAX_EPOCH_MS)
|
|
207
|
+
return new Date(ms).toISOString();
|
|
208
|
+
// Out of range: fall through to the fallback, then to now. An unusable
|
|
209
|
+
// timestamp is worth losing; the observation it belongs to is not.
|
|
210
|
+
}
|
|
211
|
+
if (fallback) {
|
|
212
|
+
const parsed = new Date(fallback);
|
|
213
|
+
if (!Number.isNaN(parsed.getTime()))
|
|
214
|
+
return parsed.toISOString();
|
|
215
|
+
}
|
|
216
|
+
return nowIso();
|
|
217
|
+
}
|
|
218
|
+
function jsonArray(raw) {
|
|
219
|
+
if (typeof raw !== 'string' || !raw.trim())
|
|
220
|
+
return [];
|
|
221
|
+
try {
|
|
222
|
+
const parsed = JSON.parse(raw);
|
|
223
|
+
if (Array.isArray(parsed))
|
|
224
|
+
return parsed.filter((v) => typeof v === 'string' && v.length > 0);
|
|
225
|
+
}
|
|
226
|
+
catch {
|
|
227
|
+
/* Not JSON: fall through to the comma split below. */
|
|
228
|
+
}
|
|
229
|
+
return raw
|
|
230
|
+
.split(',')
|
|
231
|
+
.map((s) => s.trim())
|
|
232
|
+
.filter(Boolean);
|
|
233
|
+
}
|
|
234
|
+
function slugify(name) {
|
|
235
|
+
return name
|
|
236
|
+
.toLowerCase()
|
|
237
|
+
.replace(/[^a-z0-9]+/g, '-')
|
|
238
|
+
.replace(/^-+|-+$/g, '')
|
|
239
|
+
.slice(0, 80);
|
|
240
|
+
}
|
|
241
|
+
/** The columns this importer knows, per table, for the unrecognised report. */
|
|
242
|
+
const KNOWN_FIELDS = new Map();
|
|
243
|
+
for (const disposition of DISPOSITIONS) {
|
|
244
|
+
if (disposition.field === '*')
|
|
245
|
+
continue;
|
|
246
|
+
const set = KNOWN_FIELDS.get(disposition.table) ?? new Set();
|
|
247
|
+
set.add(disposition.field);
|
|
248
|
+
KNOWN_FIELDS.set(disposition.table, set);
|
|
249
|
+
}
|
|
250
|
+
function unrecognisedFields(src, tables) {
|
|
251
|
+
const out = [];
|
|
252
|
+
for (const [table, known] of KNOWN_FIELDS) {
|
|
253
|
+
if (!tables.has(table))
|
|
254
|
+
continue;
|
|
255
|
+
const cols = src.prepare(`PRAGMA table_info("${table}")`).all();
|
|
256
|
+
for (const col of cols) {
|
|
257
|
+
if (known.has(col.name))
|
|
258
|
+
continue;
|
|
259
|
+
out.push({
|
|
260
|
+
table,
|
|
261
|
+
field: col.name,
|
|
262
|
+
to: null,
|
|
263
|
+
kind: 'unrecognised',
|
|
264
|
+
reason: 'not present when this importer was written — reported rather than dropped silently',
|
|
265
|
+
});
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
return out;
|
|
269
|
+
}
|
|
270
|
+
/**
|
|
271
|
+
* The read-only dry run (MIG-01). Opens the source read-only, counts, and says
|
|
272
|
+
* what would happen. Nothing here writes to either database.
|
|
273
|
+
*/
|
|
274
|
+
export function inventory(sourceDb) {
|
|
275
|
+
const src = openSource(sourceDb);
|
|
276
|
+
try {
|
|
277
|
+
const tables = tableSet(src);
|
|
278
|
+
const version = sourceSchemaVersion(src, tables);
|
|
279
|
+
const supported = version === null ? tables.has('observations') : version <= SUPPORTED_SCHEMA_VERSION;
|
|
280
|
+
const problem = supported
|
|
281
|
+
? null
|
|
282
|
+
: version === null
|
|
283
|
+
? `${sourceDb} has no observations table and no schema_versions row — it does not look like a Claude Mem database.`
|
|
284
|
+
: newerSchemaMessage(version);
|
|
285
|
+
const names = [...tables].filter((n) => !n.startsWith('sqlite_')).sort();
|
|
286
|
+
const counts = names.map((name) => ({
|
|
287
|
+
name,
|
|
288
|
+
rows: count(src, name),
|
|
289
|
+
known: KNOWN_TABLES.includes(name) || name.startsWith('observations_fts'),
|
|
290
|
+
}));
|
|
291
|
+
const projects = tables.has('observations')
|
|
292
|
+
? src
|
|
293
|
+
.prepare(`SELECT COALESCE(merged_into_project, project) AS project, COUNT(*) AS entries
|
|
294
|
+
FROM observations GROUP BY 1 ORDER BY entries DESC`)
|
|
295
|
+
.all()
|
|
296
|
+
: [];
|
|
297
|
+
const range = tables.has('observations')
|
|
298
|
+
? src.prepare('SELECT MIN(created_at_epoch) AS lo, MAX(created_at_epoch) AS hi FROM observations').get()
|
|
299
|
+
: { lo: null, hi: null };
|
|
300
|
+
return {
|
|
301
|
+
sourcePath: sourceDb,
|
|
302
|
+
schemaVersion: version,
|
|
303
|
+
supportedMax: SUPPORTED_SCHEMA_VERSION,
|
|
304
|
+
supported,
|
|
305
|
+
problem,
|
|
306
|
+
tables: counts,
|
|
307
|
+
projects,
|
|
308
|
+
dateRange: {
|
|
309
|
+
from: range.lo ? isoFromEpoch(range.lo, null) : null,
|
|
310
|
+
to: range.hi ? isoFromEpoch(range.hi, null) : null,
|
|
311
|
+
},
|
|
312
|
+
fields: [...DISPOSITIONS, ...unrecognisedFields(src, tables)],
|
|
313
|
+
};
|
|
314
|
+
}
|
|
315
|
+
finally {
|
|
316
|
+
src.close();
|
|
317
|
+
}
|
|
318
|
+
}
|
|
319
|
+
function newerSchemaMessage(version) {
|
|
320
|
+
return (`This Claude Mem database is at schema version ${version}; this importer was written against ${SUPPORTED_SCHEMA_VERSION}. ` +
|
|
321
|
+
'Importing it would mean guessing at columns that have moved, so nothing was read. ' +
|
|
322
|
+
'Upgrade Eklavya, or export from the older Claude Mem version, and run `eklavya memory import --dry-run <path>` again.');
|
|
323
|
+
}
|
|
324
|
+
/**
|
|
325
|
+
* A stable identity for the source file, so two different Claude Mem databases
|
|
326
|
+
* do not share one id map. The realpath is enough and costs nothing: a restored
|
|
327
|
+
* backup opened from a second location is a second source, which is the
|
|
328
|
+
* conservative answer (it re-imports rather than silently skipping).
|
|
329
|
+
*/
|
|
330
|
+
function sourceIdentity(file) {
|
|
331
|
+
try {
|
|
332
|
+
return fs.realpathSync(file);
|
|
333
|
+
}
|
|
334
|
+
catch {
|
|
335
|
+
return path.resolve(file);
|
|
336
|
+
}
|
|
337
|
+
}
|
|
338
|
+
/**
|
|
339
|
+
* Takes a consistent snapshot including WAL state (MIG-01).
|
|
340
|
+
*
|
|
341
|
+
* `VACUUM INTO` on a read-only connection is the whole trick: it reads through
|
|
342
|
+
* the WAL and writes a single self-consistent file, where `cp claude-mem.db`
|
|
343
|
+
* would hand back a main database missing every transaction still in the log.
|
|
344
|
+
*/
|
|
345
|
+
function snapshot(sourceDb, dir, reuse) {
|
|
346
|
+
fs.mkdirSync(dir, { recursive: true });
|
|
347
|
+
const out = path.join(dir, 'claude-mem-snapshot.db');
|
|
348
|
+
if (reuse && fs.existsSync(out))
|
|
349
|
+
return out;
|
|
350
|
+
fs.rmSync(out, { force: true });
|
|
351
|
+
const src = openSource(sourceDb);
|
|
352
|
+
try {
|
|
353
|
+
src.exec(`VACUUM INTO '${out.replace(/'/g, "''")}'`);
|
|
354
|
+
}
|
|
355
|
+
finally {
|
|
356
|
+
src.close();
|
|
357
|
+
}
|
|
358
|
+
return out;
|
|
359
|
+
}
|
|
360
|
+
function mapper(db, sourceId) {
|
|
361
|
+
const read = db.prepare('SELECT target_id FROM import_id_map WHERE source = ? AND source_db = ? AND source_table = ? AND source_id = ?');
|
|
362
|
+
const write = db.prepare(`INSERT OR IGNORE INTO import_id_map
|
|
363
|
+
(source, source_db, source_table, source_id, target_table, target_id, imported_at)
|
|
364
|
+
VALUES (?, ?, ?, ?, ?, ?, ?)`);
|
|
365
|
+
return {
|
|
366
|
+
get(table, id) {
|
|
367
|
+
const row = read.get(IMPORT_SOURCE, sourceId, table, String(id));
|
|
368
|
+
return row?.target_id ?? null;
|
|
369
|
+
},
|
|
370
|
+
set(table, id, targetTable, targetId) {
|
|
371
|
+
write.run(IMPORT_SOURCE, sourceId, table, String(id), targetTable, targetId, nowIso());
|
|
372
|
+
},
|
|
373
|
+
};
|
|
374
|
+
}
|
|
375
|
+
const EMPTY_COUNTS = () => ({
|
|
376
|
+
observations: 0,
|
|
377
|
+
session_summaries: 0,
|
|
378
|
+
user_prompts: 0,
|
|
379
|
+
tool_uses: 0,
|
|
380
|
+
});
|
|
381
|
+
/**
|
|
382
|
+
* Imports a Claude Mem database into Eklavya's memory tables.
|
|
383
|
+
*
|
|
384
|
+
* Only the memory half is written: `memory_entries`, `evidence_events`,
|
|
385
|
+
* `learning_sources` and `import_id_map`. Concepts, attempts, mastery and gates
|
|
386
|
+
* are never read or written here, which is what makes "an import cannot
|
|
387
|
+
* overwrite learning history" a property of the code rather than a promise.
|
|
388
|
+
*/
|
|
389
|
+
/** Applies `projectMap`, recording which names were used so the report can say. */
|
|
390
|
+
/**
|
|
391
|
+
* Says which source project names were rewritten and which were not.
|
|
392
|
+
*
|
|
393
|
+
* The unmapped list is the useful half: a name left alone is history that will
|
|
394
|
+
* only ever surface under `--all-projects`, and somebody reading the report
|
|
395
|
+
* afterwards should not have to work that out for themselves.
|
|
396
|
+
*/
|
|
397
|
+
function recordProjects(report, options, seen, used) {
|
|
398
|
+
const map = options.projectMap ?? {};
|
|
399
|
+
report.projectsMapped = [...used].sort().map((from) => ({ from, to: map[from] }));
|
|
400
|
+
report.projectsKept = [...seen].filter((name) => !used.has(name)).sort();
|
|
401
|
+
}
|
|
402
|
+
function projectMapper(options, used) {
|
|
403
|
+
const map = options.projectMap ?? {};
|
|
404
|
+
return (name) => {
|
|
405
|
+
const mapped = map[name];
|
|
406
|
+
if (mapped === undefined)
|
|
407
|
+
return name;
|
|
408
|
+
used.add(name);
|
|
409
|
+
return mapped;
|
|
410
|
+
};
|
|
411
|
+
}
|
|
412
|
+
export function importFrom(db, sourceDb, opts = {}) {
|
|
413
|
+
const dryRun = opts.dryRun ?? false;
|
|
414
|
+
const found = inventory(sourceDb);
|
|
415
|
+
if (!found.supported)
|
|
416
|
+
throw new ImportError(found.problem ?? 'Unsupported source database.');
|
|
417
|
+
const read = EMPTY_COUNTS();
|
|
418
|
+
for (const table of IMPORTED_TABLES) {
|
|
419
|
+
read[table] = found.tables.find((t) => t.name === table)?.rows ?? 0;
|
|
420
|
+
}
|
|
421
|
+
const mapped = new Set();
|
|
422
|
+
const toProject = projectMapper(opts, mapped);
|
|
423
|
+
const seenProjects = new Set();
|
|
424
|
+
const projectOf = (name) => {
|
|
425
|
+
seenProjects.add(name);
|
|
426
|
+
return toProject(name);
|
|
427
|
+
};
|
|
428
|
+
const report = {
|
|
429
|
+
sourcePath: sourceDb,
|
|
430
|
+
snapshot: null,
|
|
431
|
+
schemaVersion: found.schemaVersion,
|
|
432
|
+
read,
|
|
433
|
+
imported: EMPTY_COUNTS(),
|
|
434
|
+
skipped: EMPTY_COUNTS(),
|
|
435
|
+
candidates: 0,
|
|
436
|
+
links: 0,
|
|
437
|
+
reindexed: 0,
|
|
438
|
+
unsupportedFields: found.fields.filter((f) => f.kind !== 'mapped'),
|
|
439
|
+
validation: { ok: true, notes: [] },
|
|
440
|
+
dryRun,
|
|
441
|
+
projectsMapped: [],
|
|
442
|
+
projectsKept: [],
|
|
443
|
+
};
|
|
444
|
+
if (dryRun) {
|
|
445
|
+
// A dry run reads nothing project-shaped, so the mapping it would apply is
|
|
446
|
+
// reported from the inventory instead of from rows it never touched.
|
|
447
|
+
recordProjects(report, opts, new Set(found.projects.map((p) => p.project)), mapped);
|
|
448
|
+
return report;
|
|
449
|
+
}
|
|
450
|
+
const dir = opts.snapshotDir ?? path.join(os.tmpdir(), 'eklavya-import');
|
|
451
|
+
const snap = snapshot(sourceDb, dir, opts.resume ?? false);
|
|
452
|
+
report.snapshot = snap;
|
|
453
|
+
const src = new Database(snap, { readonly: true, fileMustExist: true });
|
|
454
|
+
try {
|
|
455
|
+
const tables = tableSet(src);
|
|
456
|
+
const ids = mapper(db, sourceIdentity(sourceDb));
|
|
457
|
+
const entryIds = [];
|
|
458
|
+
// sdk_sessions is read for lookup only — a prompt carries no project of its
|
|
459
|
+
// own, and filing it under the wrong repository is worse than not importing
|
|
460
|
+
// it at all.
|
|
461
|
+
const sessionProject = new Map();
|
|
462
|
+
if (tables.has('sdk_sessions')) {
|
|
463
|
+
for (const row of src
|
|
464
|
+
.prepare('SELECT id, project, content_session_id, memory_session_id FROM sdk_sessions')
|
|
465
|
+
.all()) {
|
|
466
|
+
sessionProject.set(row.id, {
|
|
467
|
+
project: projectOf(row.project),
|
|
468
|
+
sessionId: row.memory_session_id ?? row.content_session_id,
|
|
469
|
+
});
|
|
470
|
+
}
|
|
471
|
+
}
|
|
472
|
+
if (tables.has('observations')) {
|
|
473
|
+
for (const row of src.prepare('SELECT * FROM observations ORDER BY id').iterate()) {
|
|
474
|
+
if (ids.get('observations', row.id) !== null) {
|
|
475
|
+
report.skipped.observations++;
|
|
476
|
+
continue;
|
|
477
|
+
}
|
|
478
|
+
const occurredAt = isoFromEpoch(row.created_at_epoch, row.created_at);
|
|
479
|
+
const entryProject = projectOf(row.merged_into_project ?? row.project);
|
|
480
|
+
const title = (row.title ?? row.subtitle ?? row.text ?? 'Imported observation').slice(0, 200);
|
|
481
|
+
const narrative = [row.subtitle, row.narrative ?? row.text].filter(Boolean).join('\n\n');
|
|
482
|
+
const concepts = jsonArray(row.concepts);
|
|
483
|
+
const files = [...new Set([...jsonArray(row.files_read), ...jsonArray(row.files_modified)])];
|
|
484
|
+
const entryId = db.transaction(() => {
|
|
485
|
+
const id = insertEntry(db, {
|
|
486
|
+
project: entryProject,
|
|
487
|
+
sessionId: row.memory_session_id,
|
|
488
|
+
kind: 'observation',
|
|
489
|
+
type: row.type,
|
|
490
|
+
title,
|
|
491
|
+
narrative,
|
|
492
|
+
facts: jsonArray(row.facts),
|
|
493
|
+
files,
|
|
494
|
+
tags: [...concepts.map((c) => c.toLowerCase()), ...(row.agent_type ? [row.agent_type.toLowerCase()] : [])],
|
|
495
|
+
generator: row.generated_by_model ? `${IMPORT_SOURCE}:${row.generated_by_model}` : IMPORT_SOURCE,
|
|
496
|
+
occurredAt,
|
|
497
|
+
importSource: IMPORT_SOURCE,
|
|
498
|
+
entryUid: entryUid({ project: entryProject, title, occurredAt, salt: `${IMPORT_SOURCE}:observations:${row.id}` }),
|
|
499
|
+
});
|
|
500
|
+
ids.set('observations', row.id, 'memory_entries', id);
|
|
501
|
+
// A concept the source attached is a *proposal*. It lands as a
|
|
502
|
+
// candidate and nothing more: no mastery, no attempt, no gate.
|
|
503
|
+
for (const concept of concepts) {
|
|
504
|
+
const slug = slugify(concept);
|
|
505
|
+
if (!slug)
|
|
506
|
+
continue;
|
|
507
|
+
addCandidate(db, {
|
|
508
|
+
entryId: id,
|
|
509
|
+
slug,
|
|
510
|
+
name: concept,
|
|
511
|
+
domain: 'imported',
|
|
512
|
+
confidence: 0,
|
|
513
|
+
project: entryProject,
|
|
514
|
+
});
|
|
515
|
+
report.candidates++;
|
|
516
|
+
}
|
|
517
|
+
return id;
|
|
518
|
+
})();
|
|
519
|
+
entryIds.push(entryId);
|
|
520
|
+
report.imported.observations++;
|
|
521
|
+
}
|
|
522
|
+
}
|
|
523
|
+
if (tables.has('session_summaries')) {
|
|
524
|
+
for (const row of src
|
|
525
|
+
.prepare('SELECT * FROM session_summaries ORDER BY id')
|
|
526
|
+
.iterate()) {
|
|
527
|
+
if (ids.get('session_summaries', row.id) !== null) {
|
|
528
|
+
report.skipped.session_summaries++;
|
|
529
|
+
continue;
|
|
530
|
+
}
|
|
531
|
+
const occurredAt = isoFromEpoch(row.created_at_epoch, row.created_at);
|
|
532
|
+
const entryProject = projectOf(row.merged_into_project ?? row.project);
|
|
533
|
+
const title = (row.request?.split('\n')[0] ?? 'Session summary').slice(0, 200);
|
|
534
|
+
const narrative = [
|
|
535
|
+
['Request', row.request],
|
|
536
|
+
['Investigated', row.investigated],
|
|
537
|
+
['Learned', row.learned],
|
|
538
|
+
['Completed', row.completed],
|
|
539
|
+
['Next steps', row.next_steps],
|
|
540
|
+
['Notes', row.notes],
|
|
541
|
+
]
|
|
542
|
+
.filter(([, value]) => Boolean(value))
|
|
543
|
+
.map(([label, value]) => `${label}: ${value}`)
|
|
544
|
+
.join('\n\n');
|
|
545
|
+
const id = insertEntry(db, {
|
|
546
|
+
project: entryProject,
|
|
547
|
+
sessionId: row.memory_session_id,
|
|
548
|
+
kind: 'session_summary',
|
|
549
|
+
title,
|
|
550
|
+
narrative,
|
|
551
|
+
files: [...new Set([...jsonArray(row.files_read), ...jsonArray(row.files_edited)])],
|
|
552
|
+
generator: IMPORT_SOURCE,
|
|
553
|
+
occurredAt,
|
|
554
|
+
importSource: IMPORT_SOURCE,
|
|
555
|
+
entryUid: entryUid({ project: entryProject, title, occurredAt, salt: `${IMPORT_SOURCE}:session_summaries:${row.id}` }),
|
|
556
|
+
});
|
|
557
|
+
ids.set('session_summaries', row.id, 'memory_entries', id);
|
|
558
|
+
entryIds.push(id);
|
|
559
|
+
report.imported.session_summaries++;
|
|
560
|
+
}
|
|
561
|
+
}
|
|
562
|
+
if (tables.has('user_prompts')) {
|
|
563
|
+
for (const row of src.prepare('SELECT * FROM user_prompts ORDER BY id').iterate()) {
|
|
564
|
+
if (ids.get('user_prompts', row.id) !== null) {
|
|
565
|
+
report.skipped.user_prompts++;
|
|
566
|
+
continue;
|
|
567
|
+
}
|
|
568
|
+
const occurredAt = isoFromEpoch(row.created_at_epoch, row.created_at);
|
|
569
|
+
const session = row.session_db_id === null ? undefined : sessionProject.get(row.session_db_id);
|
|
570
|
+
const id = appendImportedEvent(db, {
|
|
571
|
+
project: session?.project ?? 'unknown',
|
|
572
|
+
sessionId: session?.sessionId ?? row.content_session_id,
|
|
573
|
+
kind: 'prompt',
|
|
574
|
+
title: row.prompt_number === null ? null : `Prompt #${row.prompt_number}`,
|
|
575
|
+
body: row.prompt_text,
|
|
576
|
+
occurredAt,
|
|
577
|
+
});
|
|
578
|
+
ids.set('user_prompts', row.id, 'evidence_events', id);
|
|
579
|
+
report.imported.user_prompts++;
|
|
580
|
+
}
|
|
581
|
+
}
|
|
582
|
+
if (tables.has('tool_uses')) {
|
|
583
|
+
for (const row of src.prepare('SELECT * FROM tool_uses ORDER BY id').iterate()) {
|
|
584
|
+
if (ids.get('tool_uses', row.id) !== null) {
|
|
585
|
+
report.skipped.tool_uses++;
|
|
586
|
+
continue;
|
|
587
|
+
}
|
|
588
|
+
const occurredAt = isoFromEpoch(row.created_at_epoch, row.created_at);
|
|
589
|
+
const body = [row.tool_input, row.tool_response].filter(Boolean).join('\n---\n');
|
|
590
|
+
const id = appendImportedEvent(db, {
|
|
591
|
+
project: projectOf(row.project),
|
|
592
|
+
checkout: row.cwd,
|
|
593
|
+
sessionId: row.memory_session_id ?? row.content_session_id,
|
|
594
|
+
agentId: row.agent_id,
|
|
595
|
+
kind: 'tool_use',
|
|
596
|
+
tool: row.tool_name,
|
|
597
|
+
title: row.tool_name,
|
|
598
|
+
body: `${row.tool_use_id}\n${body}`,
|
|
599
|
+
occurredAt,
|
|
600
|
+
});
|
|
601
|
+
ids.set('tool_uses', row.id, 'evidence_events', id);
|
|
602
|
+
report.imported.tool_uses++;
|
|
603
|
+
}
|
|
604
|
+
}
|
|
605
|
+
// An entry with no evidence behind it is an entry nobody can check:
|
|
606
|
+
// `memory_get --include-evidence` returns nothing and the dashboard's raw
|
|
607
|
+
// evidence block is empty. Both sides are in the id map by now, so the
|
|
608
|
+
// link is rebuilt from the source's own joins rather than guessed.
|
|
609
|
+
report.links = linkEvidence(db, src, tables, ids);
|
|
610
|
+
// Rebuild the vectors for everything imported (MIG-02). `insertEntry`
|
|
611
|
+
// indexes as it goes, so this is a backstop for a run resumed after a crash
|
|
612
|
+
// between the insert and its vector -- and it is what makes "the index was
|
|
613
|
+
// rebuilt" a checked fact rather than an assumption.
|
|
614
|
+
report.reindexed = reindex(db, entryIds);
|
|
615
|
+
report.validation = validate(db, report);
|
|
616
|
+
recordProjects(report, opts, seenProjects, mapped);
|
|
617
|
+
return report;
|
|
618
|
+
}
|
|
619
|
+
finally {
|
|
620
|
+
src.close();
|
|
621
|
+
// The snapshot is a complete copy of the developer's history. It exists so
|
|
622
|
+
// an interrupted run can resume; once the run has validated there is no
|
|
623
|
+
// reason to leave one sitting in a temp directory (PRD SEC-02).
|
|
624
|
+
if (report.validation.ok)
|
|
625
|
+
fs.rmSync(snap, { force: true });
|
|
626
|
+
}
|
|
627
|
+
}
|
|
628
|
+
function appendImportedEvent(db, input) {
|
|
629
|
+
// `host: 'claude-mem'` and `source: 'import'` together are the provenance: an
|
|
630
|
+
// imported event must never be mistaken for something this machine watched.
|
|
631
|
+
const uid = eventUid({
|
|
632
|
+
host: IMPORT_SOURCE,
|
|
633
|
+
sessionId: input.sessionId,
|
|
634
|
+
agentId: input.agentId ?? null,
|
|
635
|
+
kind: input.kind,
|
|
636
|
+
tool: input.tool ?? null,
|
|
637
|
+
occurredAt: input.occurredAt,
|
|
638
|
+
body: input.body,
|
|
639
|
+
});
|
|
640
|
+
const existing = db.prepare('SELECT id FROM evidence_events WHERE event_uid = ?').get(uid);
|
|
641
|
+
if (existing)
|
|
642
|
+
return existing.id;
|
|
643
|
+
return Number(db
|
|
644
|
+
.prepare(`INSERT INTO evidence_events
|
|
645
|
+
(event_uid, project, checkout, session_id, agent_id, host, source, kind, tool,
|
|
646
|
+
title, body, files, occurred_at, received_at, redacted, status)
|
|
647
|
+
VALUES (?, ?, ?, ?, ?, ?, 'import', ?, ?, ?, ?, NULL, ?, ?, 0, 'summarized')`)
|
|
648
|
+
.run(uid, input.project, input.checkout ?? null, input.sessionId, input.agentId ?? null, IMPORT_SOURCE, input.kind, input.tool ?? null, input.title ?? null, input.body, input.occurredAt, nowIso()).lastInsertRowid);
|
|
649
|
+
}
|
|
650
|
+
/**
|
|
651
|
+
* Rebuilds `memory_entry_events` for imported rows.
|
|
652
|
+
*
|
|
653
|
+
* Two joins, and which one is worth running was measured rather than assumed
|
|
654
|
+
* (a real source: 4036 observations, 1146 tool uses, 545 prompts):
|
|
655
|
+
*
|
|
656
|
+
* - `tool_uses.observation_id` is a direct key and covered 842 of 1146 tool
|
|
657
|
+
* uses. The remaining 304 carry no `prompt_number` either, so the pair
|
|
658
|
+
* fallback recovers *none* of them — it is not run for tool uses, where it
|
|
659
|
+
* would only fan one tool use out across every observation of its turn.
|
|
660
|
+
* - `user_prompts` has no `observation_id` column at all, so the pair
|
|
661
|
+
* `(memory_session_id, prompt_number)` is the only join there is. It matched
|
|
662
|
+
* every one of the 4036 observations to the prompt that produced it, which is
|
|
663
|
+
* what makes the drill-down useful at all: the direct key alone reached only
|
|
664
|
+
* 220 distinct observations.
|
|
665
|
+
*
|
|
666
|
+
* A prompt carries no `memory_session_id`, so it is resolved through
|
|
667
|
+
* `sdk_sessions` the same way its project is.
|
|
668
|
+
*/
|
|
669
|
+
function linkEvidence(db, src, tables, ids) {
|
|
670
|
+
if (!tables.has('observations'))
|
|
671
|
+
return 0;
|
|
672
|
+
const joins = [];
|
|
673
|
+
if (tables.has('tool_uses')) {
|
|
674
|
+
joins.push({
|
|
675
|
+
table: 'tool_uses',
|
|
676
|
+
sql: 'SELECT id AS event_src, observation_id AS obs FROM tool_uses WHERE observation_id IS NOT NULL',
|
|
677
|
+
});
|
|
678
|
+
}
|
|
679
|
+
if (tables.has('user_prompts') && tables.has('sdk_sessions')) {
|
|
680
|
+
joins.push({
|
|
681
|
+
table: 'user_prompts',
|
|
682
|
+
sql: `SELECT p.id AS event_src, o.id AS obs
|
|
683
|
+
FROM user_prompts p
|
|
684
|
+
JOIN sdk_sessions s ON s.id = p.session_db_id
|
|
685
|
+
JOIN observations o
|
|
686
|
+
ON o.memory_session_id = COALESCE(s.memory_session_id, s.content_session_id)
|
|
687
|
+
AND o.prompt_number = p.prompt_number
|
|
688
|
+
WHERE p.prompt_number IS NOT NULL`,
|
|
689
|
+
});
|
|
690
|
+
}
|
|
691
|
+
const link = db.prepare('INSERT OR IGNORE INTO memory_entry_events (entry_id, event_id) VALUES (?, ?)');
|
|
692
|
+
return db.transaction(() => {
|
|
693
|
+
let written = 0;
|
|
694
|
+
for (const { table, sql } of joins) {
|
|
695
|
+
for (const row of src.prepare(sql).iterate()) {
|
|
696
|
+
// Only where both sides were imported: a source row the importer
|
|
697
|
+
// skipped has no target to point at, and a dangling link is worse
|
|
698
|
+
// than a missing one.
|
|
699
|
+
const entryId = ids.get('observations', row.obs);
|
|
700
|
+
const eventId = ids.get(table, row.event_src);
|
|
701
|
+
if (entryId !== null && eventId !== null)
|
|
702
|
+
written += link.run(entryId, eventId).changes;
|
|
703
|
+
}
|
|
704
|
+
}
|
|
705
|
+
return written;
|
|
706
|
+
})();
|
|
707
|
+
}
|
|
708
|
+
function reindex(db, entryIds) {
|
|
709
|
+
let done = 0;
|
|
710
|
+
const get = db.prepare('SELECT title, narrative, facts, files FROM memory_entries WHERE id = ?');
|
|
711
|
+
for (const id of entryIds) {
|
|
712
|
+
const row = get.get(id);
|
|
713
|
+
if (!row)
|
|
714
|
+
continue;
|
|
715
|
+
indexVector(db, id, [row.title, row.narrative, row.facts ?? '', row.files ?? ''].join('\n'));
|
|
716
|
+
done++;
|
|
717
|
+
}
|
|
718
|
+
return done;
|
|
719
|
+
}
|
|
720
|
+
/**
|
|
721
|
+
* Count validation (MIG-02). Every source row must be accounted for as either
|
|
722
|
+
* imported or already present; anything else means rows were dropped silently,
|
|
723
|
+
* which is the failure mode an import report exists to catch.
|
|
724
|
+
*/
|
|
725
|
+
function validate(db, report) {
|
|
726
|
+
const notes = [];
|
|
727
|
+
for (const table of IMPORTED_TABLES) {
|
|
728
|
+
const seen = report.imported[table] + report.skipped[table];
|
|
729
|
+
if (seen !== report.read[table]) {
|
|
730
|
+
notes.push(`${table}: read ${report.read[table]}, accounted for ${seen}`);
|
|
731
|
+
}
|
|
732
|
+
}
|
|
733
|
+
const vectorless = db
|
|
734
|
+
.prepare(`SELECT COUNT(*) AS n FROM memory_entries e
|
|
735
|
+
WHERE e.import_source IS NOT NULL AND e.deleted_at IS NULL
|
|
736
|
+
AND NOT EXISTS (SELECT 1 FROM memory_vectors v WHERE v.entry_id = e.id)`)
|
|
737
|
+
.get().n;
|
|
738
|
+
if (vectorless > 0)
|
|
739
|
+
notes.push(`${vectorless} imported entries have no vector`);
|
|
740
|
+
return { ok: notes.length === 0, notes };
|
|
741
|
+
}
|
|
742
|
+
/**
|
|
743
|
+
* The export format version `eklavya memory export` stamps and
|
|
744
|
+
* `restoreExport` refuses to guess past. Bump it when the payload shape below
|
|
745
|
+
* changes.
|
|
746
|
+
*/
|
|
747
|
+
export const EXPORT_SCHEMA_VERSION = 1;
|
|
748
|
+
/** Columns are listed rather than spread: an export carries ids and a
|
|
749
|
+
* `batch_id` that mean nothing in the destination database, and binding them
|
|
750
|
+
* would either collide with a live row or point a foreign key at nothing. */
|
|
751
|
+
const EVENT_COLUMNS = [
|
|
752
|
+
'event_uid', 'project', 'checkout', 'session_id', 'agent_id', 'host', 'source', 'kind', 'tool',
|
|
753
|
+
'title', 'body', 'files', 'occurred_at', 'received_at', 'redacted', 'status',
|
|
754
|
+
];
|
|
755
|
+
const ENTRY_COLUMNS = [
|
|
756
|
+
'entry_uid', 'project', 'session_id', 'kind', 'type', 'title', 'narrative', 'facts', 'files',
|
|
757
|
+
'generator', 'confidence', 'occurred_at', 'created_at', 'deleted_at', 'import_source',
|
|
758
|
+
];
|
|
759
|
+
const RECEIPT_COLUMNS = [
|
|
760
|
+
'receipt_uid', 'project', 'session_id', 'scope', 'method', 'base_tokens', 'delivered_tokens',
|
|
761
|
+
'item_count', 'delivery', 'created_at',
|
|
762
|
+
];
|
|
763
|
+
function values(row, columns) {
|
|
764
|
+
return columns.map((c) => row[c] ?? null);
|
|
765
|
+
}
|
|
766
|
+
function rows(payload, key) {
|
|
767
|
+
const value = payload[key];
|
|
768
|
+
return Array.isArray(value) ? value : [];
|
|
769
|
+
}
|
|
770
|
+
/**
|
|
771
|
+
* Restores a `eklavya memory export` file (the other half of the backup pair).
|
|
772
|
+
*
|
|
773
|
+
* Shares the importer's three rules next door, because the invariants are the
|
|
774
|
+
* same: nothing outside the memory tables is written, so attempts, mastery and
|
|
775
|
+
* gates are untouched; a schema version this build does not know stops with
|
|
776
|
+
* both versions named rather than guessing; and identity is the `*_uid`
|
|
777
|
+
* columns, so a second restore of the same file adds nothing.
|
|
778
|
+
*
|
|
779
|
+
* Ids are remapped rather than reused. An export's primary keys belong to the
|
|
780
|
+
* database it came from, and the destination may already have rows at those
|
|
781
|
+
* numbers — restoring them verbatim would either collide or, worse, silently
|
|
782
|
+
* attach one machine's evidence to another machine's observation.
|
|
783
|
+
*/
|
|
784
|
+
export function restoreExport(db, file) {
|
|
785
|
+
if (!fs.existsSync(file)) {
|
|
786
|
+
throw new ImportError(`No export file at ${file}. Pass the file \`eklavya memory export\` wrote.`);
|
|
787
|
+
}
|
|
788
|
+
let payload;
|
|
789
|
+
try {
|
|
790
|
+
payload = JSON.parse(fs.readFileSync(file, 'utf8'));
|
|
791
|
+
}
|
|
792
|
+
catch {
|
|
793
|
+
throw new ImportError(`${file} is not readable JSON. Pass the file \`eklavya memory export\` wrote.`);
|
|
794
|
+
}
|
|
795
|
+
const found = payload?.schema_version;
|
|
796
|
+
if (found !== EXPORT_SCHEMA_VERSION) {
|
|
797
|
+
throw new ImportError(`${file} is export schema version ${found === undefined ? 'unstated' : String(found)}; ` +
|
|
798
|
+
`this build understands version ${EXPORT_SCHEMA_VERSION}. ` +
|
|
799
|
+
'Restore it with the Eklavya version that wrote it, or export again from that version.');
|
|
800
|
+
}
|
|
801
|
+
const report = {
|
|
802
|
+
file,
|
|
803
|
+
schemaVersion: EXPORT_SCHEMA_VERSION,
|
|
804
|
+
entries: { restored: 0, skipped: 0 },
|
|
805
|
+
evidence: { restored: 0, skipped: 0 },
|
|
806
|
+
receipts: { restored: 0, skipped: 0 },
|
|
807
|
+
tags: 0,
|
|
808
|
+
links: 0,
|
|
809
|
+
receiptItems: 0,
|
|
810
|
+
reindexed: 0,
|
|
811
|
+
};
|
|
812
|
+
const restoredEntryIds = [];
|
|
813
|
+
db.transaction(() => {
|
|
814
|
+
const insert = (table, columns) => db.prepare(`INSERT INTO ${table} (${columns.join(', ')}) VALUES (${columns.map(() => '?').join(', ')})`);
|
|
815
|
+
// Evidence first: entries link to it, and the link table needs both sides
|
|
816
|
+
// mapped before it can be written.
|
|
817
|
+
const eventIds = new Map();
|
|
818
|
+
const findEvent = db.prepare('SELECT id FROM evidence_events WHERE event_uid = ?');
|
|
819
|
+
const insEvent = insert('evidence_events', EVENT_COLUMNS);
|
|
820
|
+
for (const row of rows(payload, 'evidence')) {
|
|
821
|
+
const existing = findEvent.get(row.event_uid);
|
|
822
|
+
if (existing) {
|
|
823
|
+
eventIds.set(Number(row.id), existing.id);
|
|
824
|
+
report.evidence.skipped++;
|
|
825
|
+
continue;
|
|
826
|
+
}
|
|
827
|
+
eventIds.set(Number(row.id), Number(insEvent.run(...values(row, EVENT_COLUMNS)).lastInsertRowid));
|
|
828
|
+
report.evidence.restored++;
|
|
829
|
+
}
|
|
830
|
+
const entryIds = new Map();
|
|
831
|
+
const findEntry = db.prepare('SELECT id FROM memory_entries WHERE entry_uid = ?');
|
|
832
|
+
const insEntry = insert('memory_entries', ENTRY_COLUMNS);
|
|
833
|
+
for (const row of rows(payload, 'entries')) {
|
|
834
|
+
const existing = findEntry.get(row.entry_uid);
|
|
835
|
+
if (existing) {
|
|
836
|
+
entryIds.set(Number(row.id), existing.id);
|
|
837
|
+
report.entries.skipped++;
|
|
838
|
+
continue;
|
|
839
|
+
}
|
|
840
|
+
const id = Number(insEntry.run(...values(row, ENTRY_COLUMNS)).lastInsertRowid);
|
|
841
|
+
entryIds.set(Number(row.id), id);
|
|
842
|
+
restoredEntryIds.push(id);
|
|
843
|
+
report.entries.restored++;
|
|
844
|
+
}
|
|
845
|
+
// Second pass, because a correction can supersede a row that had not been
|
|
846
|
+
// inserted yet when its own turn came.
|
|
847
|
+
const supersede = db.prepare('UPDATE memory_entries SET superseded_by = ? WHERE id = ?');
|
|
848
|
+
for (const row of rows(payload, 'entries')) {
|
|
849
|
+
const target = entryIds.get(Number(row.superseded_by));
|
|
850
|
+
const self = entryIds.get(Number(row.id));
|
|
851
|
+
if (row.superseded_by != null && target && self)
|
|
852
|
+
supersede.run(target, self);
|
|
853
|
+
}
|
|
854
|
+
const insTag = db.prepare('INSERT OR IGNORE INTO memory_entry_tags (entry_id, tag) VALUES (?, ?)');
|
|
855
|
+
for (const row of rows(payload, 'tags')) {
|
|
856
|
+
const entryId = entryIds.get(Number(row.entry_id));
|
|
857
|
+
if (entryId)
|
|
858
|
+
report.tags += insTag.run(entryId, row.tag).changes;
|
|
859
|
+
}
|
|
860
|
+
const insLink = db.prepare('INSERT OR IGNORE INTO memory_entry_events (entry_id, event_id) VALUES (?, ?)');
|
|
861
|
+
for (const row of rows(payload, 'entry_events')) {
|
|
862
|
+
const entryId = entryIds.get(Number(row.entry_id));
|
|
863
|
+
const eventId = eventIds.get(Number(row.event_id));
|
|
864
|
+
if (entryId && eventId)
|
|
865
|
+
report.links += insLink.run(entryId, eventId).changes;
|
|
866
|
+
}
|
|
867
|
+
const receiptIds = new Map();
|
|
868
|
+
const findReceipt = db.prepare('SELECT id FROM context_receipts WHERE receipt_uid = ?');
|
|
869
|
+
const insReceipt = insert('context_receipts', RECEIPT_COLUMNS);
|
|
870
|
+
for (const row of rows(payload, 'receipts')) {
|
|
871
|
+
const existing = findReceipt.get(row.receipt_uid);
|
|
872
|
+
if (existing) {
|
|
873
|
+
receiptIds.set(Number(row.id), existing.id);
|
|
874
|
+
report.receipts.skipped++;
|
|
875
|
+
continue;
|
|
876
|
+
}
|
|
877
|
+
receiptIds.set(Number(row.id), Number(insReceipt.run(...values(row, RECEIPT_COLUMNS)).lastInsertRowid));
|
|
878
|
+
report.receipts.restored++;
|
|
879
|
+
}
|
|
880
|
+
const insItem = db.prepare(`INSERT OR IGNORE INTO context_receipt_items (receipt_id, entry_id, source_tokens, sent_tokens, stage)
|
|
881
|
+
VALUES (?, ?, ?, ?, ?)`);
|
|
882
|
+
for (const row of rows(payload, 'receipt_items')) {
|
|
883
|
+
const receiptId = receiptIds.get(Number(row.receipt_id));
|
|
884
|
+
const entryId = entryIds.get(Number(row.entry_id));
|
|
885
|
+
if (receiptId && entryId) {
|
|
886
|
+
report.receiptItems += insItem.run(receiptId, entryId, row.source_tokens ?? 0, row.sent_tokens ?? 0, row.stage ?? 'index').changes;
|
|
887
|
+
}
|
|
888
|
+
}
|
|
889
|
+
// FTS keeps itself in step through the insert trigger; the vectors do not,
|
|
890
|
+
// so a restore without this is a restore nothing can find semantically.
|
|
891
|
+
report.reindexed = reindex(db, restoredEntryIds);
|
|
892
|
+
})();
|
|
893
|
+
return report;
|
|
894
|
+
}
|
|
895
|
+
/**
|
|
896
|
+
* The payload `eklavya memory export` writes and `restoreExport` reads.
|
|
897
|
+
*
|
|
898
|
+
* Here rather than in the CLI so the two halves of the backup pair cannot
|
|
899
|
+
* drift: a column added to the export and not to the restore is a column that
|
|
900
|
+
* silently does not survive a round trip.
|
|
901
|
+
*/
|
|
902
|
+
export function exportPayload(db) {
|
|
903
|
+
const all = (sql) => db.prepare(sql).all();
|
|
904
|
+
return {
|
|
905
|
+
schema_version: EXPORT_SCHEMA_VERSION,
|
|
906
|
+
exported_at: nowIso(),
|
|
907
|
+
db_schema_version: db.prepare("SELECT value FROM meta WHERE key = 'schema_version'").get()?.value,
|
|
908
|
+
entries: all('SELECT * FROM memory_entries ORDER BY id'),
|
|
909
|
+
tags: all('SELECT * FROM memory_entry_tags ORDER BY entry_id, tag'),
|
|
910
|
+
entry_events: all('SELECT * FROM memory_entry_events ORDER BY entry_id, event_id'),
|
|
911
|
+
evidence: all('SELECT * FROM evidence_events ORDER BY id'),
|
|
912
|
+
receipts: all('SELECT * FROM context_receipts ORDER BY id'),
|
|
913
|
+
receipt_items: all('SELECT * FROM context_receipt_items ORDER BY receipt_id, entry_id, stage'),
|
|
914
|
+
};
|
|
915
|
+
}
|
|
916
|
+
//# sourceMappingURL=import.js.map
|