akm-cli 0.9.17-alpha.8 → 0.9.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +277 -1694
- package/STABILITY.md +9 -8
- package/dist/assets/hints/cli-hints-full.md +6 -7
- package/dist/assets/improve-strategies/catchup.json +0 -3
- package/dist/assets/improve-strategies/consolidate.json +0 -1
- package/dist/assets/improve-strategies/default.json +1 -2
- package/dist/assets/improve-strategies/proactive-maintenance.json +1 -2
- package/dist/assets/improve-strategies/quick.json +1 -2
- package/dist/assets/improve-strategies/reflect-distill.json +1 -2
- package/dist/assets/improve-strategies/thorough.json +0 -3
- package/dist/assets/prompts/consolidate-pair.md +20 -0
- package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +17 -19
- package/dist/assets/stash-skeleton/facts/conventions/domains.md +2 -2
- package/dist/assets/templates/html/health.html +3 -5
- package/dist/cli/retired-commands.js +1 -1
- package/dist/cli/unknown-flags.js +24 -1
- package/dist/cli.js +46 -1
- package/dist/commands/health/archive-usage.js +92 -0
- package/dist/commands/health/data-dir-usage.js +25 -13
- package/dist/commands/health/html-report.js +1 -4
- package/dist/commands/health/improve-metrics.js +25 -37
- package/dist/commands/health/md-report.js +1 -6
- package/dist/commands/health/report-view-model.js +4 -14
- package/dist/commands/health/windows.js +0 -1
- package/dist/commands/health.js +13 -0
- package/dist/commands/improve/consolidate/continuity-check.js +137 -0
- package/dist/commands/improve/consolidate/pair-pass.js +791 -0
- package/dist/commands/improve/consolidate.js +38 -63
- package/dist/commands/improve/extract-prompt.js +1 -2
- package/dist/commands/improve/improve-cli.js +1 -1
- package/dist/commands/improve/improve-strategies.js +23 -5
- package/dist/commands/improve/improve.js +19 -30
- package/dist/commands/improve/ledger.js +3 -2
- package/dist/commands/improve/loop-stages.js +5 -85
- package/dist/commands/improve/memory/memory-belief.js +3 -1
- package/dist/commands/improve/memory/memory-improve.js +262 -11
- package/dist/commands/improve/planner.js +0 -5
- package/dist/commands/improve/preparation.js +20 -135
- package/dist/commands/improve/retrieval-scope.js +19 -4
- package/dist/commands/improve/salience.js +1 -14
- package/dist/commands/improve/stage.js +0 -1
- package/dist/commands/lint/base-linter.js +19 -11
- package/dist/commands/proposal/drain.js +8 -1
- package/dist/commands/proposal/proposal-cli.js +16 -2
- package/dist/commands/proposal/proposal-types.js +7 -0
- package/dist/commands/proposal/proposal.js +37 -6
- package/dist/commands/proposal/repository.js +613 -4
- package/dist/commands/proposal/validators/proposals.js +9 -0
- package/dist/commands/read/knowledge.js +3 -2
- package/dist/commands/read/show.js +0 -14
- package/dist/commands/sources/info.js +122 -18
- package/dist/commands/sources/stash-cli.js +23 -3
- package/dist/core/bundle-rename.js +1 -7
- package/dist/core/config/config-schema.js +8 -1
- package/dist/core/config/config.js +23 -48
- package/dist/core/config/engine-semantics.js +0 -2
- package/dist/core/config/schema/improve-processes.js +17 -42
- package/dist/core/config/schema/index-config.js +5 -25
- package/dist/core/file-change.js +13 -5
- package/dist/core/improve-result.js +22 -6
- package/dist/core/improve-types.js +0 -1
- package/dist/core/loopback.js +7 -12
- package/dist/core/parse.js +13 -16
- package/dist/core/state/migrations.js +15 -0
- package/dist/core/time.js +0 -20
- package/dist/indexer/db/llm-cache.js +2 -2
- package/dist/indexer/ensure-index.js +2 -2
- package/dist/indexer/index-written-assets.js +2 -3
- package/dist/indexer/indexer.js +18 -418
- package/dist/indexer/passes/metadata.js +0 -19
- package/dist/indexer/walk/walker.js +3 -4
- package/dist/llm/client.js +8 -10
- package/dist/llm/embedders/remote.js +1 -2
- package/dist/llm/feature-gate.js +0 -5
- package/dist/output/shapes/helpers.js +20 -4
- package/dist/output/text/command-format.js +9 -8
- package/dist/output/text/proposal-format.js +47 -1
- package/dist/output/text/show-format.js +0 -20
- package/dist/scripts/akm-migrate-node.js +923 -950
- package/dist/scripts/akm-migrate.js +923 -950
- package/dist/setup/steps/connection.js +5 -6
- package/dist/setup/steps/platforms.js +2 -2
- package/dist/sources/providers/git-stash.js +83 -4
- package/dist/storage/repositories/improve-ledger-repository.js +48 -7
- package/dist/storage/repositories/index-connection.js +5 -2
- package/dist/storage/repositories/index-entries-repository.js +4 -7
- package/dist/storage/repositories/index-entry-schema.js +4 -2
- package/dist/storage/repositories/index-llm-cache-repository.js +7 -26
- package/dist/storage/repositories/index-schema.js +55 -104
- package/dist/storage/repositories/proposals-repository.js +61 -0
- package/dist/storage/repositories/salience-repository.js +1 -19
- package/docs/migration/README.md +1 -1
- package/docs/migration/release-notes/0.9.17.md +130 -41
- package/docs/migration/release-notes/README.md +7 -0
- package/docs/reference/cli.md +27 -21
- package/docs/reference/configuration.md +21 -12
- package/docs/reference/data-and-telemetry.md +0 -1
- package/package.json +1 -1
- package/schemas/akm-config.json +0 -342
- package/dist/assets/improve-strategies/graph-refresh.json +0 -15
- package/dist/assets/prompts/contradiction-judge.md +0 -33
- package/dist/assets/prompts/graph-extract-system.md +0 -1
- package/dist/assets/prompts/graph-extract-user-prompt.md +0 -35
- package/dist/assets/prompts/metadata-enhance-system.md +0 -1
- package/dist/assets/tasks/improve/akm-graph-refresh-weekly.yml +0 -4
- package/dist/indexer/db/graph-db.js +0 -399
- package/dist/indexer/graph/graph-extraction.js +0 -809
- package/dist/indexer/graph/graph-related.js +0 -131
- package/dist/indexer/graph/graph-types.js +0 -4
- package/dist/llm/graph-extract.js +0 -892
- package/dist/llm/metadata-enhance.js +0 -95
|
@@ -122,7 +122,8 @@ const LLM_PRESETS = [
|
|
|
122
122
|
},
|
|
123
123
|
];
|
|
124
124
|
/**
|
|
125
|
-
* Step 3a: pick an LLM provider. Used for
|
|
125
|
+
* Step 3a: pick an LLM provider. Used for LLM-gated background features
|
|
126
|
+
* (e.g. memory inference).
|
|
126
127
|
*
|
|
127
128
|
* @internal Exported for testing only.
|
|
128
129
|
*/
|
|
@@ -151,14 +152,14 @@ export async function stepLlm(current, ollamaEndpoint, ollamaChatModels, lmStudi
|
|
|
151
152
|
}
|
|
152
153
|
options.push({ value: "lmstudio", label: "LM Studio / local server", hint: lmStudioOptionHint(lmStudio) });
|
|
153
154
|
options.push({ value: "custom", label: "Custom OpenAI-compatible endpoint" });
|
|
154
|
-
options.push({ value: "none", label: "Skip LLM", hint: "
|
|
155
|
+
options.push({ value: "none", label: "Skip LLM", hint: "LLM-gated background features stay disabled" });
|
|
155
156
|
const currentLlm = readCurrentLlmEngine(current);
|
|
156
157
|
if (currentLlm) {
|
|
157
158
|
options.push(keepCurrentOption(currentLlm));
|
|
158
159
|
}
|
|
159
160
|
const initialValue = currentLlm ? "keep" : ollamaAvailable ? "ollama" : (LLM_PRESETS[0]?.value ?? "none");
|
|
160
161
|
const choice = await prompt(() => p.select({
|
|
161
|
-
message: "Configure an LLM for
|
|
162
|
+
message: "Configure an LLM for background features:",
|
|
162
163
|
options,
|
|
163
164
|
initialValue,
|
|
164
165
|
}));
|
|
@@ -251,7 +252,7 @@ export async function stepLlm(current, ollamaEndpoint, ollamaChatModels, lmStudi
|
|
|
251
252
|
return llm;
|
|
252
253
|
}
|
|
253
254
|
/**
|
|
254
|
-
* Step 1/2: Configure the small model connection used for
|
|
255
|
+
* Step 1/2: Configure the small model connection used for bounded LLM features.
|
|
255
256
|
*
|
|
256
257
|
* Detects Ollama automatically and pre-selects it. The user may also choose
|
|
257
258
|
* OpenAI, LM Studio, a custom endpoint, or skip the step entirely.
|
|
@@ -260,7 +261,6 @@ export async function stepSmallModelConnection(current) {
|
|
|
260
261
|
p.log.step("Step 1/2: Configure your small model connection");
|
|
261
262
|
p.note([
|
|
262
263
|
"This connection is used for background processing:",
|
|
263
|
-
" • akm index (metadata enhancement)",
|
|
264
264
|
" • akm improve (lesson distillation)",
|
|
265
265
|
" • akm remember --enrich (memory compression)",
|
|
266
266
|
].join("\n"));
|
|
@@ -301,7 +301,6 @@ export async function stepSmallModelConnection(current) {
|
|
|
301
301
|
if (providerChoice === "skip") {
|
|
302
302
|
p.note([
|
|
303
303
|
"Enrichment features disabled:",
|
|
304
|
-
" • akm index — metadata enhancement disabled",
|
|
305
304
|
" • akm improve — lesson generation",
|
|
306
305
|
" • akm remember --enrich",
|
|
307
306
|
"",
|
|
@@ -53,10 +53,10 @@ export function printCapabilitySummary(smallModelSkipped, agentConfigured) {
|
|
|
53
53
|
const lines = ["Setup complete. Here's what's enabled:", ""];
|
|
54
54
|
lines.push(" ✓ akm search, akm curate, akm show, akm index, akm remember — always available");
|
|
55
55
|
if (!smallModelSkipped) {
|
|
56
|
-
lines.push(" ✓
|
|
56
|
+
lines.push(" ✓ akm improve, akm remember --enrich — small model configured");
|
|
57
57
|
}
|
|
58
58
|
else {
|
|
59
|
-
lines.push(" ✗
|
|
59
|
+
lines.push(" ✗ akm improve, akm remember --enrich — run `akm setup` to enable");
|
|
60
60
|
}
|
|
61
61
|
if (agentConfigured) {
|
|
62
62
|
lines.push(" ✓ akm proposal new, akm improve, akm task — agent configured");
|
|
@@ -25,11 +25,18 @@ import { getCachePaths, parseGitRepoUrl } from "./git-provider.js";
|
|
|
25
25
|
export function isGitBackedStash(stashDir) {
|
|
26
26
|
return fs.existsSync(path.join(stashDir, ".git"));
|
|
27
27
|
}
|
|
28
|
-
/**
|
|
29
|
-
|
|
28
|
+
/**
|
|
29
|
+
* Return repo-relative dirty/staged paths without changing the index, and
|
|
30
|
+
* whether `git status` itself succeeded. A broken submodule, a detached
|
|
31
|
+
* `GIT_DIR`, or git simply not being on `PATH` all exit nonzero here — a
|
|
32
|
+
* caller that would otherwise read the empty `paths` as "nothing is dirty"
|
|
33
|
+
* must check `ok` first (see {@link listGitChangedPaths}'s callers that
|
|
34
|
+
* cannot, and the archive purge sweep, which can and does).
|
|
35
|
+
*/
|
|
36
|
+
export function tryListGitChangedPaths(repoDir) {
|
|
30
37
|
const result = runGit(["-C", repoDir, "status", "--porcelain", "-z", "--untracked-files=all"]);
|
|
31
38
|
if (result.status !== 0)
|
|
32
|
-
return [];
|
|
39
|
+
return { paths: [], ok: false };
|
|
33
40
|
const records = result.stdout.split("\0");
|
|
34
41
|
const paths = [];
|
|
35
42
|
for (let i = 0; i < records.length; i++) {
|
|
@@ -44,7 +51,79 @@ export function listGitChangedPaths(repoDir) {
|
|
|
44
51
|
paths.push(previousPath);
|
|
45
52
|
}
|
|
46
53
|
}
|
|
47
|
-
return paths;
|
|
54
|
+
return { paths, ok: true };
|
|
55
|
+
}
|
|
56
|
+
/** Return repo-relative dirty/staged paths without changing the index. `[]` on any git failure — see {@link tryListGitChangedPaths} for a caller that must tell that apart from "nothing is dirty". */
|
|
57
|
+
export function listGitChangedPaths(repoDir) {
|
|
58
|
+
return tryListGitChangedPaths(repoDir).paths;
|
|
59
|
+
}
|
|
60
|
+
/**
|
|
61
|
+
* Return repo-relative paths git tracks at HEAD/index under `pathspec` (or
|
|
62
|
+
* the whole repo when omitted), and whether `git ls-files` itself
|
|
63
|
+
* succeeded — see {@link tryListGitChangedPaths}, the same contract.
|
|
64
|
+
*/
|
|
65
|
+
export function tryListGitTrackedPaths(repoDir, pathspec) {
|
|
66
|
+
const args = ["-C", repoDir, "ls-files", "-z"];
|
|
67
|
+
if (pathspec)
|
|
68
|
+
args.push("--", pathspec);
|
|
69
|
+
const result = runGit(args);
|
|
70
|
+
if (result.status !== 0)
|
|
71
|
+
return { paths: [], ok: false };
|
|
72
|
+
return { paths: result.stdout.split("\0").filter((record) => record.length > 0), ok: true };
|
|
73
|
+
}
|
|
74
|
+
/**
|
|
75
|
+
* Return repo-relative paths under `pathspec` that `git ls-files -v` tags as
|
|
76
|
+
* NOT verifiable against the worktree: `assume-unchanged` (a lowercase tag —
|
|
77
|
+
* `ls-files -v` lowercases a file's normal tag letter when that bit is set)
|
|
78
|
+
* or `skip-worktree` (the literal `S`). `git status` silently omits an edit
|
|
79
|
+
* to either kind of file — the index is telling git not to compare it — so a
|
|
80
|
+
* caller that trusts {@link tryListGitChangedPaths} alone would read a
|
|
81
|
+
* genuinely modified file as clean.
|
|
82
|
+
*/
|
|
83
|
+
export function tryListGitUnverifiablePaths(repoDir, pathspec) {
|
|
84
|
+
const args = ["-C", repoDir, "ls-files", "-v", "-z"];
|
|
85
|
+
if (pathspec)
|
|
86
|
+
args.push("--", pathspec);
|
|
87
|
+
const result = runGit(args);
|
|
88
|
+
if (result.status !== 0)
|
|
89
|
+
return { paths: [], ok: false };
|
|
90
|
+
const paths = [];
|
|
91
|
+
for (const record of result.stdout.split("\0")) {
|
|
92
|
+
if (!record)
|
|
93
|
+
continue;
|
|
94
|
+
const tag = record.slice(0, 1);
|
|
95
|
+
if (tag === "S" || (tag >= "a" && tag <= "z"))
|
|
96
|
+
paths.push(record.slice(2));
|
|
97
|
+
}
|
|
98
|
+
return { paths, ok: true };
|
|
99
|
+
}
|
|
100
|
+
/**
|
|
101
|
+
* The three-part git cleanliness check shared by the archive purge sweep
|
|
102
|
+
* (`purgeGracedArchive` in `commands/improve/memory/memory-improve.ts`) and
|
|
103
|
+
* the `memory-cleanup-archive` health advisory (`collectArchiveUsageAdvisory`
|
|
104
|
+
* in `commands/health/archive-usage.ts`): a path is safe to treat as
|
|
105
|
+
* committed only if it is tracked ({@link tryListGitTrackedPaths}), not dirty
|
|
106
|
+
* ({@link tryListGitChangedPaths}), and not assume-unchanged/skip-worktree
|
|
107
|
+
* ({@link tryListGitUnverifiablePaths} — those hide their own edits from
|
|
108
|
+
* `git status`, so an unverifiable file is never trusted as clean either).
|
|
109
|
+
*
|
|
110
|
+
* Each of the three git calls can fail independently (a broken submodule, or
|
|
111
|
+
* git missing from `PATH`); `ok` is `false` if any one does, and `isSafe`
|
|
112
|
+
* then returns `false` for every path rather than guessing — callers that
|
|
113
|
+
* need to short-circuit before doing other work still check `ok` themselves.
|
|
114
|
+
*/
|
|
115
|
+
export function checkGitPathSafety(repoDir, pathspec) {
|
|
116
|
+
const dirtyQuery = tryListGitChangedPaths(repoDir);
|
|
117
|
+
const trackedQuery = tryListGitTrackedPaths(repoDir, pathspec);
|
|
118
|
+
const unverifiableQuery = tryListGitUnverifiablePaths(repoDir, pathspec);
|
|
119
|
+
const ok = dirtyQuery.ok && trackedQuery.ok && unverifiableQuery.ok;
|
|
120
|
+
const dirty = new Set(dirtyQuery.paths);
|
|
121
|
+
const tracked = new Set(trackedQuery.paths);
|
|
122
|
+
const unverifiable = new Set(unverifiableQuery.paths);
|
|
123
|
+
return {
|
|
124
|
+
ok,
|
|
125
|
+
isSafe: (repoRelativePath) => ok && tracked.has(repoRelativePath) && !dirty.has(repoRelativePath) && !unverifiable.has(repoRelativePath),
|
|
126
|
+
};
|
|
48
127
|
}
|
|
49
128
|
export class GitStashPushError extends Error {
|
|
50
129
|
commit;
|
|
@@ -27,6 +27,17 @@ export const LEDGER_DEFAULT_REJECTION_WINDOW_DAYS = 7;
|
|
|
27
27
|
export const LEDGER_EXPIRED_GRACE_DAYS = 1;
|
|
28
28
|
/** Revisit cadence for a ref the stage looked at and had nothing to do (or is still pending). */
|
|
29
29
|
export const LEDGER_REVISIT_CADENCE_DAYS = 7;
|
|
30
|
+
/**
|
|
31
|
+
* The consolidate pair pass's own ledger source (alpha.9): kept apart from
|
|
32
|
+
* the promote pass's `consolidate` rows so the two candidate-selection
|
|
33
|
+
* cadences never collide on the same `(stash, ref, source)` key. Its
|
|
34
|
+
* eligibility is entirely content-driven (`content_hash` above, compared by
|
|
35
|
+
* `selectInitiators` in `src/commands/improve/consolidate/pair-pass.ts`) —
|
|
36
|
+
* {@link windowDays} below gives it no `next_eligible_at` timer at all, so a
|
|
37
|
+
* row never "expires" on its own; only a content change makes the ref
|
|
38
|
+
* eligible again.
|
|
39
|
+
*/
|
|
40
|
+
export const PAIR_PASS_LEDGER_SOURCE = "consolidate-pair";
|
|
30
41
|
/**
|
|
31
42
|
* Outcomes whose window a fresh signal on the asset (new feedback, a content
|
|
32
43
|
* change) cannot lift. Every other window is a revisit cadence that a signal
|
|
@@ -38,6 +49,12 @@ export const LEDGER_HARD_OUTCOMES = new Set([
|
|
|
38
49
|
"expired",
|
|
39
50
|
]);
|
|
40
51
|
function windowDays(source, outcome) {
|
|
52
|
+
// The pair pass's own eligibility never reads next_eligible_at (it compares
|
|
53
|
+
// content_hash instead — selectInitiators in pair-pass.ts) — recording a
|
|
54
|
+
// window here would be a number nothing enforces, so every row it writes
|
|
55
|
+
// stays "eligible now" regardless of outcome.
|
|
56
|
+
if (source === PAIR_PASS_LEDGER_SOURCE)
|
|
57
|
+
return null;
|
|
41
58
|
switch (outcome) {
|
|
42
59
|
case "rejected":
|
|
43
60
|
case "quality_rejected":
|
|
@@ -91,6 +108,7 @@ function toRow(row) {
|
|
|
91
108
|
nextEligibleAt: row.next_eligible_at,
|
|
92
109
|
proposalId: row.proposal_id,
|
|
93
110
|
detail: row.detail,
|
|
111
|
+
contentHash: row.content_hash,
|
|
94
112
|
};
|
|
95
113
|
}
|
|
96
114
|
function trimDetail(detail) {
|
|
@@ -112,16 +130,18 @@ export function recordImproveLedger(db, input) {
|
|
|
112
130
|
nextEligibleAt: nextEligibleAt(input.source, input.outcome, input.at),
|
|
113
131
|
proposalId: input.proposalId ?? null,
|
|
114
132
|
detail: trimDetail(input.detail),
|
|
133
|
+
contentHash: input.contentHash ?? null,
|
|
115
134
|
};
|
|
116
135
|
db.prepare(`INSERT INTO improve_ledger
|
|
117
|
-
(stash_dir, ref, source, last_attempt_at, outcome, next_eligible_at, proposal_id, detail)
|
|
118
|
-
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
|
136
|
+
(stash_dir, ref, source, last_attempt_at, outcome, next_eligible_at, proposal_id, detail, content_hash)
|
|
137
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
119
138
|
ON CONFLICT(stash_dir, ref, source) DO UPDATE SET
|
|
120
139
|
last_attempt_at = excluded.last_attempt_at,
|
|
121
140
|
outcome = excluded.outcome,
|
|
122
141
|
next_eligible_at = excluded.next_eligible_at,
|
|
123
142
|
proposal_id = excluded.proposal_id,
|
|
124
|
-
detail = excluded.detail
|
|
143
|
+
detail = excluded.detail,
|
|
144
|
+
content_hash = excluded.content_hash`).run(row.stashDir, row.ref, row.source, row.lastAttemptAt, row.outcome, row.nextEligibleAt, row.proposalId, row.detail, row.contentHash);
|
|
125
145
|
return row;
|
|
126
146
|
}
|
|
127
147
|
/**
|
|
@@ -150,19 +170,40 @@ export function recordImproveLedgerDecision(db, input) {
|
|
|
150
170
|
...(input.detail !== undefined ? { detail: input.detail } : {}),
|
|
151
171
|
});
|
|
152
172
|
}
|
|
173
|
+
/**
|
|
174
|
+
* Should-fix 6 (second review round): a read-only or dry-run open never
|
|
175
|
+
* migrates, so it can land on a state.db from before migration 029 added
|
|
176
|
+
* `content_hash` — reading it there threw "no such column", which
|
|
177
|
+
* `loadRetrievalScope`'s own catch then reported as "usage history
|
|
178
|
+
* unreadable", making the WHOLE scope `undefined` (every asset eligible) on
|
|
179
|
+
* every read-only/dry-run call against an as-yet-unmigrated database. A
|
|
180
|
+
* per-connection cache, since a real `Database` handle's schema does not
|
|
181
|
+
* change mid-lifetime and this is checked on every ledger read.
|
|
182
|
+
*/
|
|
183
|
+
const hasContentHashColumnCache = new WeakMap();
|
|
184
|
+
function hasContentHashColumn(db) {
|
|
185
|
+
const cached = hasContentHashColumnCache.get(db);
|
|
186
|
+
if (cached !== undefined)
|
|
187
|
+
return cached;
|
|
188
|
+
const has = db.prepare("PRAGMA table_info(improve_ledger)").all().some((c) => c.name === "content_hash");
|
|
189
|
+
hasContentHashColumnCache.set(db, has);
|
|
190
|
+
return has;
|
|
191
|
+
}
|
|
153
192
|
export function getImproveLedgerRow(db, stashDir, ref, source) {
|
|
193
|
+
const withHash = hasContentHashColumn(db);
|
|
154
194
|
const row = db
|
|
155
|
-
.prepare(`SELECT stash_dir, ref, source, last_attempt_at, outcome, next_eligible_at, proposal_id, detail
|
|
195
|
+
.prepare(`SELECT stash_dir, ref, source, last_attempt_at, outcome, next_eligible_at, proposal_id, detail${withHash ? ", content_hash" : ""}
|
|
156
196
|
FROM improve_ledger WHERE stash_dir = ? AND ref = ? AND source = ?`)
|
|
157
197
|
.get(stashDir, ref, source);
|
|
158
|
-
return row ? toRow(row) : undefined;
|
|
198
|
+
return row ? toRow(withHash ? row : { ...row, content_hash: null }) : undefined;
|
|
159
199
|
}
|
|
160
200
|
/** Every row for one stash, optionally narrowed to `sources`. */
|
|
161
201
|
export function listImproveLedgerRows(db, stashDir, sources) {
|
|
202
|
+
const withHash = hasContentHashColumn(db);
|
|
162
203
|
const sourceFilter = sources && sources.length > 0 ? ` AND source IN (${sources.map(() => "?").join(", ")})` : "";
|
|
163
204
|
const rows = db
|
|
164
|
-
.prepare(`SELECT stash_dir, ref, source, last_attempt_at, outcome, next_eligible_at, proposal_id, detail
|
|
205
|
+
.prepare(`SELECT stash_dir, ref, source, last_attempt_at, outcome, next_eligible_at, proposal_id, detail${withHash ? ", content_hash" : ""}
|
|
165
206
|
FROM improve_ledger WHERE stash_dir = ?${sourceFilter} ORDER BY ref ASC, source ASC`)
|
|
166
207
|
.all(stashDir, ...(sources && sources.length > 0 ? sources : []));
|
|
167
|
-
return rows.map(toRow);
|
|
208
|
+
return rows.map((row) => toRow(withHash ? row : { ...row, content_hash: null }));
|
|
168
209
|
}
|
|
@@ -180,9 +180,12 @@ export function openReadonlyExistingDatabase(dbPath, options) {
|
|
|
180
180
|
// never block — but in the DELETE/TRUNCATE modes the network-FS fallback and
|
|
181
181
|
// AKM_SQLITE_JOURNAL_MODE can select, a concurrent writer makes every read
|
|
182
182
|
// fail instantly with SQLITE_BUSY. busy_timeout is legal on a read-only
|
|
183
|
-
// connection, so apply just that one.
|
|
183
|
+
// connection, so apply just that one. `busyTimeoutMs` defaults to the
|
|
184
|
+
// shared 30s constant; a caller that must never sit behind another akm
|
|
185
|
+
// process's write lock for long (e.g. `akm info`) can pass a much shorter
|
|
186
|
+
// bound instead.
|
|
184
187
|
try {
|
|
185
|
-
db.exec(`PRAGMA busy_timeout = ${SQLITE_BUSY_TIMEOUT_MS}`);
|
|
188
|
+
db.exec(`PRAGMA busy_timeout = ${options?.busyTimeoutMs ?? SQLITE_BUSY_TIMEOUT_MS}`);
|
|
186
189
|
checkIndexLayout(db, resolvedPath);
|
|
187
190
|
return db;
|
|
188
191
|
}
|
|
@@ -50,8 +50,8 @@ export function upsertEntry(db, filePath, entry, provenance, contentHash) {
|
|
|
50
50
|
// does not have to scan + JSON-decode every memory row.
|
|
51
51
|
const derivedFrom = typeof entry.derivedFrom === "string" && entry.derivedFrom.trim() ? entry.derivedFrom.trim() : null;
|
|
52
52
|
const hash = embedHash(entry);
|
|
53
|
-
// `content_hash` is optional
|
|
54
|
-
//
|
|
53
|
+
// `content_hash` is optional; an upsert that omits it preserves the scan
|
|
54
|
+
// writer's current value.
|
|
55
55
|
const apply = () => {
|
|
56
56
|
const previous = stmts.findByItemRef.get(provenance.itemRef);
|
|
57
57
|
const result = stmts.upsert.get(provenance.itemRef, provenance.bundleId, provenance.componentId, provenance.conceptId, provenance.adapterId, entry.type, filePath, contentHash ?? null, JSON.stringify(entry), derivedFrom, hash);
|
|
@@ -70,8 +70,8 @@ export function upsertEntry(db, filePath, entry, provenance, contentHash) {
|
|
|
70
70
|
return db.transaction(apply)();
|
|
71
71
|
}
|
|
72
72
|
const upsertStmtsByDb = new WeakMap();
|
|
73
|
-
// item_ref is the sole durable conflict target. `content_hash` COALESCEs so
|
|
74
|
-
//
|
|
73
|
+
// item_ref is the sole durable conflict target. `content_hash` COALESCEs so
|
|
74
|
+
// an upsert that omits it (`contentHash` is optional) cannot wipe a scan hash.
|
|
75
75
|
const UPSERT_SET_CLAUSE = `SET
|
|
76
76
|
bundle_id = excluded.bundle_id,
|
|
77
77
|
component_id = excluded.component_id,
|
|
@@ -418,9 +418,6 @@ function deleteRelatedRows(db, ids, options = {}) {
|
|
|
418
418
|
// commits; standalone delete callers retain the immediate behavior.
|
|
419
419
|
if (options.cleanupUsageEvents !== false)
|
|
420
420
|
deleteUsageEventsByEntryIds(numericIds);
|
|
421
|
-
// graph_files is keyed by its own stash_root/file_path/body_hash identity,
|
|
422
|
-
// so deleting an entry row intentionally leaves extracted graph data intact,
|
|
423
|
-
// and with it the graph_meta counts the graph writer derives from those rows.
|
|
424
421
|
}
|
|
425
422
|
export function deleteUsageEventsByEntryIds(entryIds) {
|
|
426
423
|
if (entryIds.length === 0 || !fs.existsSync(getStateDbPath()))
|
|
@@ -8,8 +8,10 @@
|
|
|
8
8
|
* readers serve what is there, and the writable opener (`ensureSchema`,
|
|
9
9
|
* `index-schema.ts`) brings it up to date in place — added and dropped
|
|
10
10
|
* columns, retired tables dropped, a one-time full-text rebuild — without
|
|
11
|
-
* touching embeddings, utility scores,
|
|
12
|
-
*
|
|
11
|
+
* touching embeddings, utility scores, or the LLM enrichment cache (the one
|
|
12
|
+
* exception is the LLM entity graph, unconditionally dropped since its
|
|
13
|
+
* retirement in 0.9.17-alpha.9). A newer layout is refused, naming the
|
|
14
|
+
* upgrade.
|
|
13
15
|
*/
|
|
14
16
|
// 26: declared links (#935) live in `asset_links`, one row per link, owned by
|
|
15
17
|
// the entry that declares it. The writable opener derives them in place from
|
|
@@ -4,11 +4,11 @@
|
|
|
4
4
|
/**
|
|
5
5
|
* `index.db` LLM enrichment-cache repository.
|
|
6
6
|
*
|
|
7
|
-
* Owns the raw SQL for `llm_enrichment_cache` — the body-hash-keyed cache
|
|
8
|
-
*
|
|
7
|
+
* Owns the raw SQL for `llm_enrichment_cache` — the body-hash-keyed cache the
|
|
8
|
+
* graph-extraction and memory-inference passes use to skip an LLM call when
|
|
9
|
+
* a file's body is unchanged.
|
|
9
10
|
*/
|
|
10
11
|
import { sha256Hex } from "../../runtime.js";
|
|
11
|
-
import { escapeLikePattern } from "../like-pattern.js";
|
|
12
12
|
import { SQLITE_CHUNK_SIZE } from "./index-sql.js";
|
|
13
13
|
/**
|
|
14
14
|
* Look up a cached LLM result for the given asset_ref.
|
|
@@ -18,7 +18,7 @@ import { SQLITE_CHUNK_SIZE } from "./index-sql.js";
|
|
|
18
18
|
* cached). In both cases the caller should invoke the LLM and write a new
|
|
19
19
|
* cache entry.
|
|
20
20
|
*/
|
|
21
|
-
export function getLlmCacheEntry(db, assetRef, currentBodyHash, cacheVariant
|
|
21
|
+
export function getLlmCacheEntry(db, assetRef, currentBodyHash, cacheVariant) {
|
|
22
22
|
const row = db
|
|
23
23
|
.prepare("SELECT asset_ref, cache_variant, body_hash, result_json, updated_at FROM llm_enrichment_cache WHERE asset_ref = ? AND cache_variant = ?")
|
|
24
24
|
.get(assetRef, cacheVariant);
|
|
@@ -44,7 +44,7 @@ export function getLlmCacheEntry(db, assetRef, currentBodyHash, cacheVariant = "
|
|
|
44
44
|
* compare `entry.bodyHash` against the current body hash themselves. This lets
|
|
45
45
|
* the batch path issue one DB query per chunk instead of one per file.
|
|
46
46
|
*/
|
|
47
|
-
export function getLlmCacheEntriesByRefs(db, refs, cacheVariant
|
|
47
|
+
export function getLlmCacheEntriesByRefs(db, refs, cacheVariant) {
|
|
48
48
|
const result = new Map();
|
|
49
49
|
if (refs.length === 0)
|
|
50
50
|
return result;
|
|
@@ -70,7 +70,7 @@ export function getLlmCacheEntriesByRefs(db, refs, cacheVariant = "") {
|
|
|
70
70
|
/**
|
|
71
71
|
* Insert or update a cached LLM result for the given asset_ref.
|
|
72
72
|
*/
|
|
73
|
-
export function upsertLlmCacheEntry(db, assetRef, bodyHash, resultJson, cacheVariant
|
|
73
|
+
export function upsertLlmCacheEntry(db, assetRef, bodyHash, resultJson, cacheVariant) {
|
|
74
74
|
db.prepare(`INSERT INTO llm_enrichment_cache (asset_ref, cache_variant, body_hash, result_json, updated_at)
|
|
75
75
|
VALUES (?, ?, ?, ?, ?)
|
|
76
76
|
ON CONFLICT(asset_ref, cache_variant) DO UPDATE SET
|
|
@@ -83,33 +83,14 @@ export function upsertLlmCacheEntry(db, assetRef, bodyHash, resultJson, cacheVar
|
|
|
83
83
|
* `entries` table. Should be called during the cleanup phase of each index
|
|
84
84
|
* run to prevent the cache from growing unboundedly as assets are removed.
|
|
85
85
|
*
|
|
86
|
-
*
|
|
87
|
-
* refs use canonical `item_ref`; preserve a cache row that matches either
|
|
88
|
-
* current identity.
|
|
86
|
+
* Cache refs are absolute file paths (memory inference).
|
|
89
87
|
*/
|
|
90
88
|
export function clearStaleCacheEntries(db) {
|
|
91
89
|
db.exec(`
|
|
92
90
|
DELETE FROM llm_enrichment_cache
|
|
93
91
|
WHERE asset_ref NOT IN (SELECT file_path FROM entries)
|
|
94
|
-
AND asset_ref NOT IN (SELECT item_ref FROM entries)
|
|
95
92
|
`);
|
|
96
93
|
}
|
|
97
|
-
/**
|
|
98
|
-
* Rewrite every `llm_enrichment_cache.asset_ref` naming `oldBundleId` (a
|
|
99
|
-
* metadata-enrichment cache key, the canonical `<bundle>//conceptId` form of
|
|
100
|
-
* `entries.item_ref`) to `newBundleId` (`akm bundle rename`, D6). Must run in
|
|
101
|
-
* the same `index.db` write as `renameEntriesBundleId` — otherwise the next
|
|
102
|
-
* `akm index`'s {@link clearStaleCacheEntries} deletes every row still keyed
|
|
103
|
-
* to the old bundle, forcing a full LLM re-enrichment. A graph/memory cache
|
|
104
|
-
* row (keyed by absolute file path, not `item_ref`) never matches the `//`
|
|
105
|
-
* prefix and is left alone. Returns the number of rows rewritten.
|
|
106
|
-
*/
|
|
107
|
-
export function renameLlmCacheAssetRefs(db, oldBundleId, newBundleId) {
|
|
108
|
-
const prefix = `${escapeLikePattern(oldBundleId)}//`;
|
|
109
|
-
return Number(db
|
|
110
|
-
.prepare(`UPDATE llm_enrichment_cache SET asset_ref = ? || substr(asset_ref, ?) WHERE asset_ref LIKE ? ESCAPE '\\'`)
|
|
111
|
-
.run(newBundleId, oldBundleId.length + 1, `${prefix}%`).changes);
|
|
112
|
-
}
|
|
113
94
|
/**
|
|
114
95
|
* Compute a stable SHA-256 hex digest of a UTF-8 string. Used as the body_hash
|
|
115
96
|
* key in `llm_enrichment_cache`. Routed through the runtime boundary so the
|
|
@@ -10,12 +10,15 @@
|
|
|
10
10
|
* for columns added after a table first shipped, drops of retired derived
|
|
11
11
|
* tables and columns, and one in-place rebuild of the (derived, cheap) FTS
|
|
12
12
|
* table when its layout is older than this release's. It never drops
|
|
13
|
-
* `entries`, `embeddings`, `utility_scores`,
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
*
|
|
13
|
+
* `entries`, `embeddings`, `utility_scores`, or `llm_enrichment_cache` to
|
|
14
|
+
* cross a version boundary; the only from-scratch rebuild is the
|
|
15
|
+
* SQLITE_CORRUPT path in `index-connection.ts`. A layout newer than this
|
|
16
|
+
* release's is refused, naming the upgrade ({@link newerIndexLayoutError}).
|
|
17
|
+
* The one exception is the LLM entity graph (`graph_meta`, `graph_files`,
|
|
18
|
+
* `graph_file_*`), retired in 0.9.17-alpha.9: those tables are dropped
|
|
19
|
+
* unconditionally below (index.db is a regenerable cache, and declared links
|
|
20
|
+
* — `asset_links` — now back `akm show`'s `links` field, which replaced the
|
|
21
|
+
* graph's `related` list).
|
|
19
22
|
*/
|
|
20
23
|
import { createRequire } from "node:module";
|
|
21
24
|
import path from "node:path";
|
|
@@ -30,12 +33,6 @@ import { getMeta, setMeta } from "./index-meta-repository.js";
|
|
|
30
33
|
export const DB_VERSION = CANONICAL_INDEX_DB_VERSION;
|
|
31
34
|
/** `index_meta` key set when the writable opener migrated the layout; cleared once `akm index` VACUUMs. */
|
|
32
35
|
export const VACUUM_PENDING_META = "vacuumPending";
|
|
33
|
-
/**
|
|
34
|
-
* The value written to `graph_meta.schema_version`, a NOT NULL column in every
|
|
35
|
-
* released layout. Releases up to 0.9.17-alpha.5 write 4 and nothing ever
|
|
36
|
-
* compares it; the index layout version gates the graph tables' shape.
|
|
37
|
-
*/
|
|
38
|
-
export const GRAPH_SCHEMA_VERSION = 4;
|
|
39
36
|
/** The layout that added declared links (`asset_links`, #935). */
|
|
40
37
|
const DECLARED_LINKS_LAYOUT = 26;
|
|
41
38
|
/**
|
|
@@ -69,97 +66,15 @@ const REGISTRY_INDEX_CACHE_DDL = `
|
|
|
69
66
|
CREATE INDEX IF NOT EXISTS idx_registry_cache_fetched
|
|
70
67
|
ON registry_index_cache(fetched_at);
|
|
71
68
|
`;
|
|
72
|
-
/**
|
|
73
|
-
* Create the graph-extraction tables (`graph_meta`/`graph_files`/`graph_file_entities`/
|
|
74
|
-
* `graph_file_relations`).
|
|
75
|
-
*
|
|
76
|
-
* graph_files is self-keyed on (stash_root, file_path, body_hash) and is not
|
|
77
|
-
* tied to entries.id (#624-P1): re-upserting an entries row never disturbs the
|
|
78
|
-
* extracted graph, and a content change yields a distinct key. A UNIQUE index
|
|
79
|
-
* on (stash_root, file_path) still enforces one graph_files row per path.
|
|
80
|
-
*/
|
|
81
|
-
function ensureGraphTables(db) {
|
|
82
|
-
db.exec(`
|
|
83
|
-
CREATE TABLE IF NOT EXISTS graph_meta (
|
|
84
|
-
stash_root TEXT PRIMARY KEY,
|
|
85
|
-
schema_version INTEGER NOT NULL,
|
|
86
|
-
generated_at TEXT NOT NULL,
|
|
87
|
-
considered_files INTEGER NOT NULL DEFAULT 0,
|
|
88
|
-
extracted_files INTEGER NOT NULL DEFAULT 0,
|
|
89
|
-
entity_count INTEGER NOT NULL DEFAULT 0,
|
|
90
|
-
relation_count INTEGER NOT NULL DEFAULT 0,
|
|
91
|
-
extraction_coverage REAL NOT NULL DEFAULT 0,
|
|
92
|
-
density REAL NOT NULL DEFAULT 0,
|
|
93
|
-
extractor_id TEXT,
|
|
94
|
-
extraction_run_id TEXT,
|
|
95
|
-
model TEXT,
|
|
96
|
-
prompt_version TEXT,
|
|
97
|
-
batch_size INTEGER,
|
|
98
|
-
cache_hits INTEGER NOT NULL DEFAULT 0,
|
|
99
|
-
cache_misses INTEGER NOT NULL DEFAULT 0,
|
|
100
|
-
truncation_count INTEGER NOT NULL DEFAULT 0,
|
|
101
|
-
failure_count INTEGER NOT NULL DEFAULT 0
|
|
102
|
-
);
|
|
103
|
-
|
|
104
|
-
CREATE TABLE IF NOT EXISTS graph_files (
|
|
105
|
-
stash_root TEXT NOT NULL,
|
|
106
|
-
file_path TEXT NOT NULL,
|
|
107
|
-
file_order INTEGER NOT NULL,
|
|
108
|
-
file_type TEXT NOT NULL,
|
|
109
|
-
body_hash TEXT NOT NULL,
|
|
110
|
-
confidence REAL,
|
|
111
|
-
status TEXT NOT NULL DEFAULT 'extracted',
|
|
112
|
-
reason TEXT,
|
|
113
|
-
extraction_run_id TEXT,
|
|
114
|
-
PRIMARY KEY (stash_root, file_path, body_hash)
|
|
115
|
-
);
|
|
116
|
-
|
|
117
|
-
CREATE UNIQUE INDEX IF NOT EXISTS idx_graph_files_path
|
|
118
|
-
ON graph_files(stash_root, file_path);
|
|
119
|
-
|
|
120
|
-
CREATE INDEX IF NOT EXISTS idx_graph_files_stash_order
|
|
121
|
-
ON graph_files(stash_root, file_order);
|
|
122
|
-
|
|
123
|
-
CREATE TABLE IF NOT EXISTS graph_file_entities (
|
|
124
|
-
stash_root TEXT NOT NULL,
|
|
125
|
-
file_path TEXT NOT NULL,
|
|
126
|
-
body_hash TEXT NOT NULL,
|
|
127
|
-
entity_order INTEGER NOT NULL,
|
|
128
|
-
entity_norm TEXT NOT NULL,
|
|
129
|
-
entity TEXT NOT NULL,
|
|
130
|
-
PRIMARY KEY (stash_root, file_path, body_hash, entity_order),
|
|
131
|
-
FOREIGN KEY (stash_root, file_path, body_hash)
|
|
132
|
-
REFERENCES graph_files(stash_root, file_path, body_hash) ON DELETE CASCADE
|
|
133
|
-
);
|
|
134
|
-
|
|
135
|
-
CREATE INDEX IF NOT EXISTS idx_graph_file_entities_entity_norm
|
|
136
|
-
ON graph_file_entities(stash_root, entity_norm);
|
|
137
|
-
|
|
138
|
-
CREATE TABLE IF NOT EXISTS graph_file_relations (
|
|
139
|
-
stash_root TEXT NOT NULL,
|
|
140
|
-
file_path TEXT NOT NULL,
|
|
141
|
-
body_hash TEXT NOT NULL,
|
|
142
|
-
relation_order INTEGER NOT NULL,
|
|
143
|
-
from_entity_norm TEXT NOT NULL,
|
|
144
|
-
from_entity TEXT NOT NULL,
|
|
145
|
-
to_entity_norm TEXT NOT NULL,
|
|
146
|
-
to_entity TEXT NOT NULL,
|
|
147
|
-
relation_type TEXT,
|
|
148
|
-
confidence REAL,
|
|
149
|
-
PRIMARY KEY (stash_root, file_path, body_hash, relation_order),
|
|
150
|
-
FOREIGN KEY (stash_root, file_path, body_hash)
|
|
151
|
-
REFERENCES graph_files(stash_root, file_path, body_hash) ON DELETE CASCADE
|
|
152
|
-
);
|
|
153
|
-
`);
|
|
154
|
-
}
|
|
155
69
|
/**
|
|
156
70
|
* An `entries` table missing a required column cannot be read or written by
|
|
157
71
|
* this release (the last such change was v20→v21, which removed the
|
|
158
72
|
* transitional `entry_key`/`dir_path`/... columns and made `item_ref` the
|
|
159
73
|
* key). Recreate only the tables keyed by `entries.id` — their ids are about
|
|
160
|
-
* to be re-minted, so the rows would dangle anyway.
|
|
161
|
-
*
|
|
162
|
-
*
|
|
74
|
+
* to be re-minted, so the rows would dangle anyway. The LLM enrichment cache
|
|
75
|
+
* (keyed by ref) is kept. The LLM entity-graph tables are unconditionally
|
|
76
|
+
* dropped elsewhere in this file regardless of this recreation (retired
|
|
77
|
+
* 0.9.17-alpha.9), not kept. The next index run re-walks every source.
|
|
163
78
|
*/
|
|
164
79
|
function ensureEntriesLayout(db) {
|
|
165
80
|
if (!tableExists(db, "entries"))
|
|
@@ -168,8 +83,8 @@ function ensureEntriesLayout(db) {
|
|
|
168
83
|
if (missing.length === 0)
|
|
169
84
|
return;
|
|
170
85
|
warn(`Index database entries table predates the ${missing.join(", ")} column${missing.length === 1 ? "" : "s"} — ` +
|
|
171
|
-
"recreating the entries-keyed tables (entries, full-text, embeddings, utility scores);
|
|
172
|
-
"LLM enrichment cache
|
|
86
|
+
"recreating the entries-keyed tables (entries, full-text, embeddings, utility scores); the " +
|
|
87
|
+
"LLM enrichment cache is kept. The next index run re-walks every source.");
|
|
173
88
|
db.transaction(() => {
|
|
174
89
|
for (const table of [
|
|
175
90
|
"entries_fts",
|
|
@@ -246,7 +161,7 @@ function ensureFtsLayout(db) {
|
|
|
246
161
|
const entryCount = Number(db.prepare("SELECT COUNT(*) AS n FROM entries").get().n);
|
|
247
162
|
if (entryCount > 0) {
|
|
248
163
|
warn(`Rebuilding the full-text index for ${entryCount} entr${entryCount === 1 ? "y" : "ies"} ` +
|
|
249
|
-
"(embeddings, utility scores,
|
|
164
|
+
"(embeddings, utility scores, and the LLM enrichment cache are kept).");
|
|
250
165
|
}
|
|
251
166
|
db.transaction(() => {
|
|
252
167
|
db.exec("DROP TABLE IF EXISTS entries_fts");
|
|
@@ -311,6 +226,25 @@ export function ensureSchema(db) {
|
|
|
311
226
|
db.exec("DROP TABLE IF EXISTS entry_fragments_fts");
|
|
312
227
|
db.exec("DROP TABLE IF EXISTS utility_scores_scoped");
|
|
313
228
|
db.exec("DROP TABLE IF EXISTS graph_extraction_queue");
|
|
229
|
+
// The LLM entity graph, retired in 0.9.17-alpha.9: declared links
|
|
230
|
+
// (`asset_links`) now back `akm show`'s `links` field (which replaced the
|
|
231
|
+
// graph's `related` list) and curate's support refs (#935), and the
|
|
232
|
+
// navigation eval measured vector kNN beating the graph's `related` list
|
|
233
|
+
// by 0.157 P@5. `graph_files` stands in for the whole set — all four
|
|
234
|
+
// tables are only ever created and dropped together. Gated on it (rather
|
|
235
|
+
// than the unconditional `DROP TABLE IF EXISTS` pattern used above) so
|
|
236
|
+
// this reclaim runs once: after the first writable open drops these
|
|
237
|
+
// tables, every later open finds `graph_files` already gone and skips the
|
|
238
|
+
// no-op DROPs and the repeat VACUUM flag below. An older release's
|
|
239
|
+
// `CREATE TABLE IF NOT EXISTS` still recreates them (empty) if it ever
|
|
240
|
+
// opens this index again — a later open here would then drop them again.
|
|
241
|
+
const hadGraphTables = tableExists(db, "graph_files");
|
|
242
|
+
if (hadGraphTables) {
|
|
243
|
+
db.exec("DROP TABLE IF EXISTS graph_meta");
|
|
244
|
+
db.exec("DROP TABLE IF EXISTS graph_files");
|
|
245
|
+
db.exec("DROP TABLE IF EXISTS graph_file_entities");
|
|
246
|
+
db.exec("DROP TABLE IF EXISTS graph_file_relations");
|
|
247
|
+
}
|
|
314
248
|
// One float32 BLOB per entry, searched by an exact scan
|
|
315
249
|
// (index-vec-repository.ts). `model` is the provider fingerprint the vector was generated under
|
|
316
250
|
// (`deriveSemanticProviderFingerprint`); the embedding pass re-embeds only
|
|
@@ -363,8 +297,8 @@ export function ensureSchema(db) {
|
|
|
363
297
|
ensureColumn(db, "index_dir_state", "row_count", "INTEGER");
|
|
364
298
|
ensureColumn(db, "index_dir_state", "index_variant", "TEXT");
|
|
365
299
|
// LLM enrichment result cache, keyed by a stable asset_ref string (the
|
|
366
|
-
// absolute file path
|
|
367
|
-
//
|
|
300
|
+
// absolute file path of the memory-inference pass) plus the body hash the
|
|
301
|
+
// result was produced for.
|
|
368
302
|
db.exec(`
|
|
369
303
|
CREATE TABLE IF NOT EXISTS llm_enrichment_cache (
|
|
370
304
|
asset_ref TEXT NOT NULL,
|
|
@@ -378,7 +312,24 @@ export function ensureSchema(db) {
|
|
|
378
312
|
CREATE INDEX IF NOT EXISTS idx_llm_cache_updated
|
|
379
313
|
ON llm_enrichment_cache(updated_at);
|
|
380
314
|
`);
|
|
381
|
-
|
|
315
|
+
// Metadata-enhance retired (RS-D, 0.9.17-alpha.9): its rows were the only
|
|
316
|
+
// ones keyed by the default empty cache_variant (memory inference writes
|
|
317
|
+
// `memory-inference-v2`), so this is safe to run unconditionally on every
|
|
318
|
+
// writable open. The table
|
|
319
|
+
// itself stays — memory inference still reads it.
|
|
320
|
+
db.exec("DELETE FROM llm_enrichment_cache WHERE cache_variant = ''");
|
|
321
|
+
// The graph-extraction cache variant is retired along with the tables
|
|
322
|
+
// above; its rows would otherwise sit unread forever. Gated the same way,
|
|
323
|
+
// on the same one-time flag, so a rerun does not re-scan the cache table
|
|
324
|
+
// for rows that are already gone.
|
|
325
|
+
if (hadGraphTables) {
|
|
326
|
+
db.exec("DELETE FROM llm_enrichment_cache WHERE cache_variant LIKE 'graph-extraction:%'");
|
|
327
|
+
// The drops and delete above freed real space (measured ~68MB on a
|
|
328
|
+
// representative index): flag it the same way a version-gated layout
|
|
329
|
+
// migration does, since this reclaim is unconditional-on-version but
|
|
330
|
+
// still one-time-per-index (guarded by hadGraphTables above).
|
|
331
|
+
setMeta(db, VACUUM_PENDING_META, "1");
|
|
332
|
+
}
|
|
382
333
|
dropVecMirror(db);
|
|
383
334
|
// Meta keys only the sqlite-vec mirror read.
|
|
384
335
|
db.exec("DELETE FROM index_meta WHERE key IN ('embeddingDim', 'vecFastPathReady')");
|