akm-cli 0.9.16-alpha.1 → 0.9.16-alpha.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +40 -132
- package/dist/assets/hints/cli-hints-full.md +13 -6
- package/dist/assets/tasks/core/index-refresh.yml +1 -1
- package/dist/assets/tasks/improve/akm-improve-catchup.yml +3 -6
- package/dist/cli/retired-commands.js +0 -4
- package/dist/cli/unknown-flags.js +3 -36
- package/dist/commands/env/env-binding.js +4 -4
- package/dist/commands/env/env-cli.js +3 -3
- package/dist/commands/improve/collapse-detector.js +2 -2
- package/dist/commands/improve/consolidate.js +4 -6
- package/dist/commands/improve/improve-cli.js +20 -15
- package/dist/commands/improve/reflect.js +23 -2
- package/dist/commands/lint/base-linter.js +9 -0
- package/dist/commands/lint/env-key-rules.js +2 -2
- package/dist/commands/proposal/propose.js +15 -1
- package/dist/commands/proposal/repository.js +3 -12
- package/dist/commands/proposal/validators/proposal-quality-validators.js +40 -3
- package/dist/commands/proposal/validators/proposal-validators.js +5 -4
- package/dist/commands/read/curate.js +44 -34
- package/dist/commands/read/search.js +35 -54
- package/dist/commands/read/show.js +21 -2
- package/dist/commands/registry-cli.js +5 -5
- package/dist/commands/sources/add-cli.js +59 -16
- package/dist/commands/sources/bundle-cli.js +35 -11
- package/dist/commands/sources/bundle-config-ops.js +30 -0
- package/dist/commands/sources/dangerous-env-audit.js +4 -4
- package/dist/commands/sources/info.js +8 -8
- package/dist/commands/sources/installed-stashes.js +55 -61
- package/dist/commands/sources/source-add.js +39 -38
- package/dist/commands/sources/source-manage.js +34 -12
- package/dist/commands/sources/stash-cli.js +111 -119
- package/dist/commands/sources/stash-skeleton.js +6 -3
- package/dist/commands/tasks/explain.js +4 -1
- package/dist/commands/tasks/tasks-cli.js +31 -9
- package/dist/commands/tasks/tasks.js +239 -194
- package/dist/commands/tasks/validate.js +20 -32
- package/dist/core/activation-policy.js +4 -4
- package/dist/core/adapter/adapters/akm-adapter.js +8 -35
- package/dist/core/adapter/adapters/akm-metadata.js +1 -11
- package/dist/core/adapter/execution-source.js +10 -29
- package/dist/core/asset/asset-placement.js +0 -35
- package/dist/core/config/config-schema.js +64 -8
- package/dist/core/config/config-sources.js +96 -2
- package/dist/core/config/config.js +190 -24
- package/dist/core/config/legacy-source-shape-shim.js +9 -0
- package/dist/core/config/schema/embedding.js +30 -7
- package/dist/core/config/schema/execution.js +23 -0
- package/dist/core/config/schema/experimental.js +1 -1
- package/dist/core/config/schema/scheduler.js +20 -0
- package/dist/core/config/schema/search.js +10 -12
- package/dist/core/config/schema/sources-bundles.js +32 -1
- package/dist/core/content-safety.js +52 -0
- package/dist/core/errors.js +2 -5
- package/dist/core/maintenance-barrier.js +11 -13
- package/dist/core/paths.js +11 -0
- package/dist/core/run-lock.js +2 -5
- package/dist/core/state/migrations.js +1 -26
- package/dist/core/state-db.js +27 -63
- package/dist/core/type-presentation.js +1 -1
- package/dist/core/write-source.js +13 -8
- package/dist/indexer/bundle-identity-guard.js +45 -8
- package/dist/indexer/ensure-index.js +0 -5
- package/dist/indexer/index-db-contention.js +56 -0
- package/dist/indexer/index-rebuild-lock.js +73 -0
- package/dist/indexer/index-written-assets.js +171 -133
- package/dist/indexer/indexer.js +1621 -458
- package/dist/indexer/lookup/adapter-concept-owner.js +5 -19
- package/dist/indexer/materialize-embeddings.js +785 -0
- package/dist/indexer/passes/dir-staleness.js +161 -0
- package/dist/indexer/passes/metadata.js +1 -18
- package/dist/indexer/scan/drain-dir.js +70 -27
- package/dist/indexer/search/db-search.js +89 -373
- package/dist/indexer/search/ranking-contributors.js +16 -21
- package/dist/indexer/search/ranking.js +57 -135
- package/dist/indexer/search/search-source.js +29 -11
- package/dist/integrations/agent/execution-lowering.js +3 -2
- package/dist/integrations/agent/execution-preparation.js +32 -1
- package/dist/integrations/agent/prompts.js +1 -1
- package/dist/integrations/agent/request-lowering.js +3 -2
- package/dist/llm/client.js +3 -11
- package/dist/llm/embedder.js +3 -10
- package/dist/llm/embedders/remote.js +104 -133
- package/dist/llm/feature-gate.js +2 -4
- package/dist/llm/rerank-client.js +3 -3
- package/dist/output/shapes/passthrough.js +2 -1
- package/dist/output/text/command-format.js +13 -19
- package/dist/output/text/helpers.js +1 -1
- package/dist/output/text/index.js +2 -5
- package/dist/registry/resolve.js +37 -10
- package/dist/scripts/akm-migrate-node.js +15197 -11351
- package/dist/scripts/akm-migrate.js +15514 -11668
- package/dist/setup/semantic-assets.js +2 -2
- package/dist/setup/setup.js +3 -3
- package/dist/setup/steps/connection.js +2 -3
- package/dist/setup/steps/tasks.js +29 -36
- package/dist/sources/providers/git-install.js +17 -11
- package/dist/sources/providers/git-provider.js +12 -5
- package/dist/sources/providers/git-stash.js +38 -16
- package/dist/sources/snapshot-fetchers/website-ingest.js +3 -3
- package/dist/storage/repositories/embedding-salvage-repository.js +184 -0
- package/dist/storage/repositories/index-connection.js +3 -1
- package/dist/storage/repositories/index-entries-repository.js +68 -77
- package/dist/storage/repositories/index-entry-schema.js +25 -16
- package/dist/storage/repositories/index-fts-repository.js +263 -29
- package/dist/storage/repositories/index-meta-repository.js +29 -0
- package/dist/storage/repositories/index-schema.js +122 -115
- package/dist/storage/repositories/index-utility-repository.js +1 -1
- package/dist/storage/repositories/index-vec-repository.js +435 -22
- package/dist/tasks/activation-config.js +90 -0
- package/dist/tasks/backends/cron.js +9 -0
- package/dist/tasks/backends/launchd.js +1 -0
- package/dist/tasks/backends/schtasks.js +2 -0
- package/dist/tasks/embedded.js +4 -5
- package/dist/tasks/scheduler-binding.js +2 -2
- package/dist/tasks/scheduler-sync-preview.js +8 -1
- package/dist/tasks/scheduler-sync.js +19 -10
- package/dist/tasks/source/parse-task-source.js +10 -113
- package/dist/tasks/source/project-v4.js +2 -2
- package/dist/tasks/source/task-source-v4.js +4 -12
- package/dist/tasks/source/task-to-v3.js +4 -12
- package/dist/tasks/source/task-to-v4.js +40 -7
- package/docs/migration/README.md +1 -0
- package/docs/migration/release-notes/0.9.15.md +36 -34
- package/docs/migration/release-notes/0.9.16.md +60 -98
- package/docs/migration/release-notes/README.md +0 -5
- package/docs/migration/v0.9.1-to-v0.9.2.md +6 -9
- package/docs/reference/cli.md +124 -122
- package/docs/reference/configuration.md +137 -133
- package/docs/reference/data-and-telemetry.md +1 -2
- package/docs/reference/tasks.md +34 -29
- package/package.json +1 -1
- package/schemas/akm-config.json +170 -6
- package/schemas/akm-task.json +1 -2
- package/dist/commands/sources/index-status.js +0 -99
- package/dist/core/hash.js +0 -18
- package/dist/indexer/drain.js +0 -306
- package/dist/indexer/embedding-identity.js +0 -20
- package/dist/indexer/enrich.js +0 -260
- package/dist/indexer/reconcile.js +0 -890
- package/dist/indexer/scan/parse-file.js +0 -66
- package/dist/indexer/units/unit.js +0 -159
- package/dist/llm/embedders/provider-limits.js +0 -288
- package/dist/storage/repositories/files-repository.js +0 -181
- package/dist/storage/repositories/units-repository.js +0 -510
package/dist/indexer/drain.js
DELETED
|
@@ -1,306 +0,0 @@
|
|
|
1
|
-
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
-
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
-
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
-
import { getConfigPath } from "../core/paths.js";
|
|
5
|
-
import { isVerbose } from "../core/warn.js";
|
|
6
|
-
import { embedBatch } from "../llm/embedder.js";
|
|
7
|
-
import { probeProviderLimits } from "../llm/embedders/provider-limits.js";
|
|
8
|
-
import { describeEmbeddingCredential, hasRemoteEndpoint, normalizeEmbeddingEndpoint, } from "../llm/embedders/remote.js";
|
|
9
|
-
import { getMeta, setMeta } from "../storage/repositories/index-meta-repository.js";
|
|
10
|
-
import { SQLITE_CHUNK_SIZE } from "../storage/repositories/index-sql.js";
|
|
11
|
-
import { isVecAvailable } from "../storage/repositories/index-vec-repository.js";
|
|
12
|
-
import { dropOtherIdentities, listMissingHashes, upsertUnitVectors } from "../storage/repositories/units-repository.js";
|
|
13
|
-
import { deriveObservedEmbeddingIdentity } from "./embedding-identity.js";
|
|
14
|
-
/**
|
|
15
|
-
* Failure threshold, within the recent window below, that stops dispatching
|
|
16
|
-
* further provider batches. Mirrors materialize-embeddings.ts's own
|
|
17
|
-
* (unexported) `CIRCUIT_BREAKER_THRESHOLD`, #954 — reimplemented here at the
|
|
18
|
-
* same value rather than imported, since that file is private and slated for
|
|
19
|
-
* deletion by B5; the underlying stop-dispatch MECHANISM (`onSkip` returning
|
|
20
|
-
* `false`) is still the real `RemoteEmbedder`'s, reused unmodified. Two
|
|
21
|
-
* independent streaks share it: single-document failures (a multi-document
|
|
22
|
-
* timeout is not yet evidence of a dead endpoint — `RemoteEmbedder` retries
|
|
23
|
-
* and splits it smaller before ever reporting it this small), or network
|
|
24
|
-
* errors at ANY size (never retried, trusted immediately). Storage-write
|
|
25
|
-
* failures (E5b — `upsertUnitVectors`'s own per-row result) feed the SAME two
|
|
26
|
-
* streaks: a sustained STORAGE failure (contention, permissions, a full
|
|
27
|
-
* disk) must stop paying for provider requests just as surely as a
|
|
28
|
-
* sustained PROVIDER failure, even while the provider itself keeps
|
|
29
|
-
* succeeding.
|
|
30
|
-
*/
|
|
31
|
-
const CIRCUIT_BREAKER_THRESHOLD = 3;
|
|
32
|
-
/**
|
|
33
|
-
* Recent-history window (in settled batch-starts) the two streaks above are
|
|
34
|
-
* evaluated over, in place of a plain "reset to zero on any success" counter
|
|
35
|
-
* (round-2 field finding): with concurrent dispatch (default 2, up to 16)
|
|
36
|
-
* outcomes settle out of dispatch order, so a degraded endpoint failing MOST
|
|
37
|
-
* requests never tripped the breaker as long as occasional successes
|
|
38
|
-
* interleaved — reproduced with a 67% failure rate dispatching the entire
|
|
39
|
-
* pending set. `CIRCUIT_BREAKER_THRESHOLD` failures within the last
|
|
40
|
-
* `CIRCUIT_BREAKER_WINDOW` settled batch-starts of a streak's own kind (see
|
|
41
|
-
* {@link pushBreakerOutcome}) trips it: a genuinely dead endpoint (no
|
|
42
|
-
* successes at all) still trips in exactly `CIRCUIT_BREAKER_THRESHOLD`
|
|
43
|
-
* batches, same as before; a single success now only AGES a failure out of
|
|
44
|
-
* the window over time rather than erasing the whole run's evidence at once.
|
|
45
|
-
*/
|
|
46
|
-
const CIRCUIT_BREAKER_WINDOW = CIRCUIT_BREAKER_THRESHOLD * 2;
|
|
47
|
-
/** Record one settled batch-start's outcome into a breaker streak's window, capped at {@link CIRCUIT_BREAKER_WINDOW}. */
|
|
48
|
-
function pushBreakerOutcome(window, isFailure) {
|
|
49
|
-
window.push(isFailure);
|
|
50
|
-
if (window.length > CIRCUIT_BREAKER_WINDOW)
|
|
51
|
-
window.shift();
|
|
52
|
-
}
|
|
53
|
-
/** Failures currently recorded in a breaker streak's window. */
|
|
54
|
-
function breakerFailureCount(window) {
|
|
55
|
-
return window.reduce((n, isFailure) => n + (isFailure ? 1 : 0), 0);
|
|
56
|
-
}
|
|
57
|
-
/**
|
|
58
|
-
* Prefix of the per-committed-batch progress line (`"${DRAIN_BATCH_PROGRESS_PREFIX}N: …"`,
|
|
59
|
-
* emitted once per provider batch this call commits). Exported so a caller
|
|
60
|
-
* juggling several `onProgress` sources (`stash-cli.ts`'s `akm index`) can
|
|
61
|
-
* recognize — and, outside `--verbose`, suppress — this specific
|
|
62
|
-
* high-frequency line by prefix rather than re-deriving its own copy of the
|
|
63
|
-
* pattern (#954).
|
|
64
|
-
*/
|
|
65
|
-
export const DRAIN_BATCH_PROGRESS_PREFIX = "[drain] batch ";
|
|
66
|
-
function throwIfAborted(signal) {
|
|
67
|
-
if (signal?.aborted) {
|
|
68
|
-
throw signal.reason instanceof Error ? signal.reason : new Error("drain interrupted");
|
|
69
|
-
}
|
|
70
|
-
}
|
|
71
|
-
/** Every distinct unit hash the index currently knows about, regardless of identity. */
|
|
72
|
-
function selectAllUnitHashes(db) {
|
|
73
|
-
return db.prepare("SELECT unit_hash FROM unit_texts ORDER BY unit_hash").all().map((row) => row.unit_hash);
|
|
74
|
-
}
|
|
75
|
-
/** `unit_hash -> text` for exactly `hashes`, chunked to respect SQLite's bound-parameter limit. */
|
|
76
|
-
function fetchUnitTexts(db, hashes) {
|
|
77
|
-
const texts = new Map();
|
|
78
|
-
for (let offset = 0; offset < hashes.length; offset += SQLITE_CHUNK_SIZE) {
|
|
79
|
-
const chunk = hashes.slice(offset, offset + SQLITE_CHUNK_SIZE);
|
|
80
|
-
const placeholders = chunk.map(() => "?").join(",");
|
|
81
|
-
const rows = db
|
|
82
|
-
.prepare(`SELECT unit_hash, text FROM unit_texts WHERE unit_hash IN (${placeholders})`)
|
|
83
|
-
.all(...chunk);
|
|
84
|
-
for (const row of rows)
|
|
85
|
-
texts.set(row.unit_hash, row.text);
|
|
86
|
-
}
|
|
87
|
-
return texts;
|
|
88
|
-
}
|
|
89
|
-
/**
|
|
90
|
-
* Effective embedding config and request packing for this drain (index
|
|
91
|
-
* redesign, B5 — replaces the retired `embedding.maxTokens`/`batchSize`/
|
|
92
|
-
* `contextLength` config keys): `concurrency` (in-flight requests) defaults
|
|
93
|
-
* to the provider's OWN observed slot count when `embedding.concurrency`
|
|
94
|
-
* itself leaves it unset (`probeProviderLimits` already applies that same
|
|
95
|
-
* override to `slots`). The request TOKEN WINDOW, chars-per-token ratio, and
|
|
96
|
-
* Ollama `num_ctx` are no longer config fields at all — they are threaded
|
|
97
|
-
* into `RemoteEmbedder.embedBatch` as `packing`, sourced straight from the
|
|
98
|
-
* same probe: `windowTokens` for the per-request budget, `charsPerToken` for
|
|
99
|
-
* the calibrated per-text token estimate, and `windowTokens` again for
|
|
100
|
-
* Ollama's `num_ctx` when `source === "ollama"`. `windowIsKnown`
|
|
101
|
-
* (`source !== "default"`) gates `RemoteEmbedder`'s same-run adaptive
|
|
102
|
-
* shrink: a provider that reports nothing about its own context size still
|
|
103
|
-
* gets that corrective, but a probed, authoritative window does not need it
|
|
104
|
-
* second-guessed.
|
|
105
|
-
*/
|
|
106
|
-
async function resolveEmbeddingPacking(config, signal) {
|
|
107
|
-
const base = config.embedding ?? {};
|
|
108
|
-
const limits = await probeProviderLimits(base, { signal });
|
|
109
|
-
return {
|
|
110
|
-
embeddingConfig: { ...base, concurrency: base.concurrency ?? limits.slots },
|
|
111
|
-
packing: {
|
|
112
|
-
tokenBudget: limits.windowTokens,
|
|
113
|
-
charsPerToken: limits.charsPerToken,
|
|
114
|
-
windowIsKnown: limits.source !== "default",
|
|
115
|
-
ollamaNumCtx: limits.source === "ollama" ? limits.windowTokens : undefined,
|
|
116
|
-
},
|
|
117
|
-
};
|
|
118
|
-
}
|
|
119
|
-
/**
|
|
120
|
-
* #953 field gap, ported from the deleted `materialize-embeddings.ts`
|
|
121
|
-
* (`git show fc711fd6^:src/indexer/materialize-embeddings.ts`): a keyless
|
|
122
|
-
* request against a remote embedding endpoint could not be reproduced in the
|
|
123
|
-
* lab — every `RemoteEmbedder` path already resolves `secret://` through one
|
|
124
|
-
* boundary, so a keyless request can only mean `embedding.apiKey` was absent
|
|
125
|
-
* from the config THIS run loaded. The actionable outcome is a
|
|
126
|
-
* self-diagnosing run, not a fix: one default-level line, emitted once
|
|
127
|
-
* before the first provider request this call makes, naming the endpoint,
|
|
128
|
-
* model, and credential SOURCE (never the value) so a field run can compare
|
|
129
|
-
* it against what the gateway actually saw. A local (non-remote) endpoint,
|
|
130
|
-
* or a call with nothing pending, has nothing to diagnose and stays silent.
|
|
131
|
-
*/
|
|
132
|
-
function emitCredentialDiagnostic(config, onProgress) {
|
|
133
|
-
if (!onProgress || !hasRemoteEndpoint(config.embedding ?? {}))
|
|
134
|
-
return;
|
|
135
|
-
const endpoint = normalizeEmbeddingEndpoint(config.embedding?.endpoint ?? "");
|
|
136
|
-
const credential = describeEmbeddingCredential(config.embedding?.apiKey);
|
|
137
|
-
const configFileSuffix = isVerbose() ? `; config: ${getConfigPath()}` : "";
|
|
138
|
-
onProgress(`[embed] endpoint ${endpoint}, model ${config.embedding?.model ?? "unknown"}; credential: ${credential}${configFileSuffix}`);
|
|
139
|
-
}
|
|
140
|
-
function formatDoneLine(counts) {
|
|
141
|
-
return (`[drain] done: ${counts.pending} pending, ${counts.embedded} embedded, ${counts.failed} failed, ` +
|
|
142
|
-
`${counts.skipped} skipped (identity: ${counts.identity ?? "unknown"})`);
|
|
143
|
-
}
|
|
144
|
-
export async function drainEmbeddingQueue(db, config, opts = {}) {
|
|
145
|
-
throwIfAborted(opts.signal);
|
|
146
|
-
let identity = getMeta(db, "embeddingIdentity") ?? null;
|
|
147
|
-
if (config.semanticSearchMode === "off") {
|
|
148
|
-
return { pending: 0, embedded: 0, failed: 0, skipped: 0, identity };
|
|
149
|
-
}
|
|
150
|
-
const candidateHashes = opts.onlyHashes ? [...new Set(opts.onlyHashes)] : selectAllUnitHashes(db);
|
|
151
|
-
const missingHashes = identity ? listMissingHashes(db, candidateHashes, identity) : candidateHashes;
|
|
152
|
-
const pending = missingHashes.length;
|
|
153
|
-
const emitDone = (counts) => {
|
|
154
|
-
opts.onProgress?.(formatDoneLine(counts));
|
|
155
|
-
return counts;
|
|
156
|
-
};
|
|
157
|
-
// upsertUnitVectors is a no-op without sqlite-vec (units-repository.ts), so
|
|
158
|
-
// embedding the pending set here would just throw every vector away and
|
|
159
|
-
// leave it "missing" again for the next call — pure wasted provider
|
|
160
|
-
// traffic. `akmIndex`'s verification reports the missing extension as
|
|
161
|
-
// blocked; this stays silent. `pending` is computed above so the done line and `akm
|
|
162
|
-
// index status` stay truthful even though nothing was attempted.
|
|
163
|
-
if (!isVecAvailable(db)) {
|
|
164
|
-
return emitDone({ pending, embedded: 0, failed: 0, skipped: pending, identity });
|
|
165
|
-
}
|
|
166
|
-
if (pending === 0) {
|
|
167
|
-
return emitDone({ pending: 0, embedded: 0, failed: 0, skipped: 0, identity });
|
|
168
|
-
}
|
|
169
|
-
const boundedHashes = opts.limit !== undefined ? missingHashes.slice(0, opts.limit) : missingHashes;
|
|
170
|
-
const textByHash = fetchUnitTexts(db, boundedHashes);
|
|
171
|
-
// A hash in `unit_texts` should always resolve to a row (it was just read
|
|
172
|
-
// from that same table above), but a missing row is dropped rather than
|
|
173
|
-
// sent to the provider as `undefined` text.
|
|
174
|
-
const orderedHashes = boundedHashes.filter((hash) => textByHash.has(hash));
|
|
175
|
-
const texts = orderedHashes.map((hash) => textByHash.get(hash));
|
|
176
|
-
if (texts.length === 0) {
|
|
177
|
-
return emitDone({ pending, embedded: 0, failed: 0, skipped: 0, identity });
|
|
178
|
-
}
|
|
179
|
-
emitCredentialDiagnostic(config, opts.onProgress);
|
|
180
|
-
const { embeddingConfig, packing } = await resolveEmbeddingPacking(config, opts.signal);
|
|
181
|
-
let embedded = 0;
|
|
182
|
-
let failed = 0;
|
|
183
|
-
let batchNumber = 0;
|
|
184
|
-
// Two independent circuit-breaker streaks (single-document failures,
|
|
185
|
-
// network errors at any size) — see CIRCUIT_BREAKER_WINDOW above.
|
|
186
|
-
const singleDocFailureWindow = [];
|
|
187
|
-
const networkErrorFailureWindow = [];
|
|
188
|
-
// Whether this CALL has already decided the identity its first committed
|
|
189
|
-
// row observed (E1) — adoption happens at most once per call; see the
|
|
190
|
-
// module doc comment and the identity block in `onBatch` below.
|
|
191
|
-
let identityDecidedThisCall = false;
|
|
192
|
-
const onSkip = (skip) => {
|
|
193
|
-
failed++;
|
|
194
|
-
if (!skip.batchStart)
|
|
195
|
-
return undefined;
|
|
196
|
-
if (skip.reason === "context-window-exceeded") {
|
|
197
|
-
// Proves the provider IS reachable; not evidence of a dead endpoint.
|
|
198
|
-
singleDocFailureWindow.length = 0;
|
|
199
|
-
networkErrorFailureWindow.length = 0;
|
|
200
|
-
return undefined;
|
|
201
|
-
}
|
|
202
|
-
pushBreakerOutcome(singleDocFailureWindow, skip.batchSize === 1);
|
|
203
|
-
pushBreakerOutcome(networkErrorFailureWindow, skip.failureKind === "network-error");
|
|
204
|
-
if (breakerFailureCount(singleDocFailureWindow) >= CIRCUIT_BREAKER_THRESHOLD ||
|
|
205
|
-
breakerFailureCount(networkErrorFailureWindow) >= CIRCUIT_BREAKER_THRESHOLD) {
|
|
206
|
-
return false;
|
|
207
|
-
}
|
|
208
|
-
return undefined;
|
|
209
|
-
};
|
|
210
|
-
const onBatch = (indices, embeddings, model, outcome) => {
|
|
211
|
-
// "retrying"/"budget-lowered" are in-flight notices for a batch that has
|
|
212
|
-
// not settled yet (see EmbeddingBatchOutcome) — nothing to commit or
|
|
213
|
-
// count, and not a distinct "batch" for the one-line-per-batch contract.
|
|
214
|
-
if (outcome?.outcome === "retrying" || outcome?.outcome === "budget-lowered")
|
|
215
|
-
return;
|
|
216
|
-
const rows = [];
|
|
217
|
-
for (let k = 0; k < indices.length; k++) {
|
|
218
|
-
const embedding = embeddings[k];
|
|
219
|
-
if (!embedding)
|
|
220
|
-
continue;
|
|
221
|
-
const learned = deriveObservedEmbeddingIdentity(config.embedding, model, embedding.length);
|
|
222
|
-
if (!identityDecidedThisCall) {
|
|
223
|
-
// This call's FIRST committed row decides the identity it adopts
|
|
224
|
-
// (E1) — learned once, not re-derived per batch: a provider whose
|
|
225
|
-
// responses alternate between models WITHIN one call (a
|
|
226
|
-
// load-balanced gateway, a blue/green rollout behind one endpoint)
|
|
227
|
-
// must not thrash the store between them (embed, delete, re-embed,
|
|
228
|
-
// never converging). A genuine model change is still caught, just
|
|
229
|
-
// not until the NEXT call's own first batch observes it and purges
|
|
230
|
-
// whatever this call left behind.
|
|
231
|
-
identityDecidedThisCall = true;
|
|
232
|
-
if (learned && learned !== identity) {
|
|
233
|
-
identity = learned;
|
|
234
|
-
setMeta(db, "embeddingIdentity", identity);
|
|
235
|
-
dropOtherIdentities(db, identity, embedding.length);
|
|
236
|
-
}
|
|
237
|
-
}
|
|
238
|
-
const currentIdentity = identity;
|
|
239
|
-
if (currentIdentity === null || learned !== currentIdentity) {
|
|
240
|
-
// Either nothing has ever been learned, or a LATER batch this same
|
|
241
|
-
// call reported an identity different from the one already adopted
|
|
242
|
-
// — left missing rather than switched to; it becomes "missing"
|
|
243
|
-
// again under whatever identity this call is using, and a later
|
|
244
|
-
// call, whose own first batch observes it, picks it up. Counted in
|
|
245
|
-
// `skipped` below (attempted minus embedded minus failed), not
|
|
246
|
-
// `embedded`.
|
|
247
|
-
continue;
|
|
248
|
-
}
|
|
249
|
-
const hash = orderedHashes[indices[k]];
|
|
250
|
-
if (hash)
|
|
251
|
-
rows.push({ hash, identity: currentIdentity, vector: embedding });
|
|
252
|
-
}
|
|
253
|
-
let storageBreakerTripped = false;
|
|
254
|
-
if (rows.length > 0) {
|
|
255
|
-
// upsertUnitVectors commits each row in its own transaction — this IS
|
|
256
|
-
// "each provider batch commits durably" (a wrapping db.transaction()
|
|
257
|
-
// here would only nest as an unobservable SAVEPOINT inside it, per the
|
|
258
|
-
// ambient-transaction hazard materialize-embeddings.ts's own drift
|
|
259
|
-
// guard documents), now made even finer-grained so one malformed
|
|
260
|
-
// vector in a batch (e.g. a width mismatch) can't roll back the rest
|
|
261
|
-
// of an otherwise-good response.
|
|
262
|
-
const result = upsertUnitVectors(db, rows);
|
|
263
|
-
embedded += result.inserted;
|
|
264
|
-
failed += result.failed;
|
|
265
|
-
if (result.failed > 0) {
|
|
266
|
-
// E5b: a write failure is just as much evidence of a broken run as
|
|
267
|
-
// a provider failure — a sustained STORAGE failure (contention,
|
|
268
|
-
// permissions, a full disk) must not keep dispatching every
|
|
269
|
-
// remaining batch to a perfectly healthy provider at full cost
|
|
270
|
-
// while every write silently fails. One event per committed batch
|
|
271
|
-
// (the "count batch starts, not documents" rule onSkip already
|
|
272
|
-
// applies to provider failures), fed into the SAME two streaks.
|
|
273
|
-
pushBreakerOutcome(singleDocFailureWindow, true);
|
|
274
|
-
pushBreakerOutcome(networkErrorFailureWindow, true);
|
|
275
|
-
if (breakerFailureCount(singleDocFailureWindow) >= CIRCUIT_BREAKER_THRESHOLD ||
|
|
276
|
-
breakerFailureCount(networkErrorFailureWindow) >= CIRCUIT_BREAKER_THRESHOLD) {
|
|
277
|
-
storageBreakerTripped = true;
|
|
278
|
-
}
|
|
279
|
-
}
|
|
280
|
-
}
|
|
281
|
-
if (embeddings.some((embedding) => embedding !== undefined)) {
|
|
282
|
-
pushBreakerOutcome(singleDocFailureWindow, false);
|
|
283
|
-
pushBreakerOutcome(networkErrorFailureWindow, false);
|
|
284
|
-
}
|
|
285
|
-
batchNumber++;
|
|
286
|
-
if (opts.onProgress) {
|
|
287
|
-
const docCount = outcome?.docCount ?? indices.length;
|
|
288
|
-
const label = outcome && outcome.outcome !== "stored" ? `failed: ${outcome.reason ?? "unknown"}` : `${rows.length} stored`;
|
|
289
|
-
opts.onProgress(`${DRAIN_BATCH_PROGRESS_PREFIX}${batchNumber}: ${docCount} docs → ${label}`);
|
|
290
|
-
}
|
|
291
|
-
if (storageBreakerTripped) {
|
|
292
|
-
// onBatch has no `false`-return stop-dispatch contract the way onSkip
|
|
293
|
-
// does (a storage failure can trip this even when the provider itself
|
|
294
|
-
// keeps succeeding, so onSkip is never called at all) — this reuses
|
|
295
|
-
// RemoteEmbedder's own documented mechanism instead: a throw from
|
|
296
|
-
// onBatch stops the pool from dispatching any further provider
|
|
297
|
-
// request, and is rethrown once every in-flight batch has settled.
|
|
298
|
-
throw new Error(`Circuit breaker: ${CIRCUIT_BREAKER_THRESHOLD} storage write failures while embedding; stopping further provider requests this call.`);
|
|
299
|
-
}
|
|
300
|
-
};
|
|
301
|
-
await embedBatch(texts, embeddingConfig, opts.signal, onSkip, onBatch, packing);
|
|
302
|
-
throwIfAborted(opts.signal);
|
|
303
|
-
const attempted = texts.length;
|
|
304
|
-
const skipped = Math.max(0, attempted - embedded - failed);
|
|
305
|
-
return emitDone({ pending, embedded, failed, skipped, identity });
|
|
306
|
-
}
|
|
@@ -1,20 +0,0 @@
|
|
|
1
|
-
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
-
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
-
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
-
import { DETERMINISTIC_EMBED_MODEL_ID, isDeterministicEmbedEnabled } from "../llm/embedders/deterministic.js";
|
|
5
|
-
import { DEFAULT_LOCAL_MODEL } from "../llm/embedders/local.js";
|
|
6
|
-
/**
|
|
7
|
-
* Returns `undefined` when nothing was actually observed this call (no
|
|
8
|
-
* vector to measure yet) — there is nothing to key an identity on.
|
|
9
|
-
*/
|
|
10
|
-
export function deriveObservedEmbeddingIdentity(embedding, observedModel, observedVectorLen) {
|
|
11
|
-
if (isDeterministicEmbedEnabled()) {
|
|
12
|
-
return `deterministic:${DETERMINISTIC_EMBED_MODEL_ID}`;
|
|
13
|
-
}
|
|
14
|
-
if (observedVectorLen === undefined)
|
|
15
|
-
return undefined;
|
|
16
|
-
if (embedding?.endpoint) {
|
|
17
|
-
return `remote:${observedModel ?? embedding.model ?? "unknown"}|${observedVectorLen}`;
|
|
18
|
-
}
|
|
19
|
-
return `local:${embedding?.localModel ?? DEFAULT_LOCAL_MODEL}|${observedVectorLen}`;
|
|
20
|
-
}
|
package/dist/indexer/enrich.js
DELETED
|
@@ -1,260 +0,0 @@
|
|
|
1
|
-
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
-
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
-
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
-
/**
|
|
5
|
-
* LLM metadata-enrichment pass, restored on the reconcile path
|
|
6
|
-
* (docs/plans/index-redesign-contract.md, B5e).
|
|
7
|
-
*
|
|
8
|
-
* Before the index redesign, `akm index` ran a config-driven metadata
|
|
9
|
-
* enhancement pass over every "generated"-quality entry, keyed by a
|
|
10
|
-
* `(assetRef, cacheVariant)` cache row whose `body_hash` column happened to
|
|
11
|
-
* gate freshness. The reconcile rewrite (`reconcile.ts`, B1) dropped the
|
|
12
|
-
* call site along with the rest of the old phase pipeline. This module
|
|
13
|
-
* restores the feature on the NEW path, content-addressed throughout:
|
|
14
|
-
* `reconcileRoots` collects one {@link MetadataEnrichmentCandidate} per file
|
|
15
|
-
* it just upserted (added or changed) and hands the batch to
|
|
16
|
-
* {@link enrichReconciledEntries} once, AFTER every per-file transaction in
|
|
17
|
-
* this run has already committed — a provider call must never run inside
|
|
18
|
-
* `applyChange`'s `BEGIN IMMEDIATE` transaction (docs/plans/index-redesign.md
|
|
19
|
-
* rule 5: every index write stays a short, idempotent transaction; an LLM
|
|
20
|
-
* call can take seconds to minutes and must not hold one open).
|
|
21
|
-
*
|
|
22
|
-
* **Content-addressed cache** — `llm_enrichment_cache` (still shared with
|
|
23
|
-
* graph-extraction and memory-inference, which key it by absolute file path)
|
|
24
|
-
* is used here with `asset_ref = body_hash = candidate.blobHash`: the cache
|
|
25
|
-
* row IS the content address, so a cache hit means "this exact byte content
|
|
26
|
-
* has already been enriched" regardless of which entry or how many entries
|
|
27
|
-
* currently carry it, and a rename or an unrelated field edit elsewhere in
|
|
28
|
-
* the same file never invalidates it. `withLlmCache` (`./db/llm-cache.ts`)
|
|
29
|
-
* already implements exactly this hash-gated lookup/call/write shape, so this
|
|
30
|
-
* module reuses it rather than duplicating the pattern a third time.
|
|
31
|
-
*
|
|
32
|
-
* **Fail-soft** — `enhanceMetadata`'s `EnhanceMetadataOutcome` distinguishes
|
|
33
|
-
* `enriched` (real success — cache it) from `skipped` (feature gate closed)
|
|
34
|
-
* and `failed` (provider/network error): `withLlmCache`'s "only cache a
|
|
35
|
-
* defined result" contract means a `skipped`/`failed` outcome (mapped to
|
|
36
|
-
* `undefined` below) writes no cache row and leaves the entry's `quality`
|
|
37
|
-
* untouched, so a transient provider outage can never poison an entry into a
|
|
38
|
-
* permanent enrichment skip.
|
|
39
|
-
*
|
|
40
|
-
* **`--full` re-applies without a new provider call** — `reconcileRoots`'s
|
|
41
|
-
* `forceReparse` re-parses every file, so an unchanged file's fresh
|
|
42
|
-
* `IndexDocument` is `quality: "generated"` again (enrichment only ever
|
|
43
|
-
* updated the DB row, never the source file) and becomes an enrichment
|
|
44
|
-
* candidate again on every `--full` run. Its `blobHash` is unchanged, so the
|
|
45
|
-
* content-addressed cache lookup above hits and re-applies the SAME cached
|
|
46
|
-
* fields with no new provider call — this falls out of content-addressing
|
|
47
|
-
* for free and needs no `--full`-specific branch here.
|
|
48
|
-
*/
|
|
49
|
-
import fs from "node:fs";
|
|
50
|
-
import { concurrentMap } from "../core/concurrent.js";
|
|
51
|
-
import { ConfigError } from "../core/errors.js";
|
|
52
|
-
import { defaultConcurrencyForEndpoint } from "../core/loopback.js";
|
|
53
|
-
import { withImmediateTransaction } from "../core/state-db.js";
|
|
54
|
-
import { warn } from "../core/warn.js";
|
|
55
|
-
import { resolveIndexPassExecution } from "../llm/index-passes.js";
|
|
56
|
-
import { enhanceMetadata } from "../llm/metadata-enhance.js";
|
|
57
|
-
import { insertNewUnitTexts } from "../storage/repositories/files-repository.js";
|
|
58
|
-
import { upsertEntry } from "../storage/repositories/index-entries-repository.js";
|
|
59
|
-
import { replaceEntryUnits } from "../storage/repositories/units-repository.js";
|
|
60
|
-
import { withLlmCache } from "./db/llm-cache.js";
|
|
61
|
-
import { getMarkdownFragmentContent, hasMarkdownFragmentContent, isEnrichmentComplete, setMarkdownFragmentContent, } from "./passes/metadata.js";
|
|
62
|
-
import { buildSearchText } from "./search/search-fields.js";
|
|
63
|
-
import { deriveUnits, toUnitSource } from "./units/unit.js";
|
|
64
|
-
/**
|
|
65
|
-
* Namespaces this pass's `llm_enrichment_cache` rows away from
|
|
66
|
-
* graph-extraction's and memory-inference's own `cacheVariant` values, which
|
|
67
|
-
* key the SAME shared table by absolute file path rather than content hash.
|
|
68
|
-
*/
|
|
69
|
-
const METADATA_ENRICHMENT_CACHE_VARIANT = "metadata-enhance-v1";
|
|
70
|
-
function emptyCounts() {
|
|
71
|
-
return { attempted: 0, cacheHits: 0, enriched: 0, failed: 0, skipped: 0 };
|
|
72
|
-
}
|
|
73
|
-
/** Only "generated"-quality entries missing description/tags/searchHints are worth an LLM call — see `isEnrichmentComplete`. */
|
|
74
|
-
function isEligibleForEnrichment(entry) {
|
|
75
|
-
return entry.quality === "generated" && !isEnrichmentComplete(entry);
|
|
76
|
-
}
|
|
77
|
-
/**
|
|
78
|
-
* Bounded-pool width for this pass — kept as a direct call to the shared
|
|
79
|
-
* classifier (not `indexer.ts`'s `getDefaultLlmConcurrency` wrapper) to avoid
|
|
80
|
-
* an indexer.ts → enrich.ts → indexer.ts import cycle, exactly like
|
|
81
|
-
* `src/llm/embedders/remote.ts`'s `resolveEmbeddingConcurrency` — see that
|
|
82
|
-
* function's neighboring comment. `tests/indexer/llm-concurrency-default.test.ts`
|
|
83
|
-
* pins `getDefaultLlmConcurrency`'s behavior; this mirrors it exactly.
|
|
84
|
-
*/
|
|
85
|
-
function resolveEnrichmentConcurrency(connection) {
|
|
86
|
-
if (typeof connection?.concurrency === "number")
|
|
87
|
-
return connection.concurrency;
|
|
88
|
-
return defaultConcurrencyForEndpoint(connection?.endpoint);
|
|
89
|
-
}
|
|
90
|
-
/**
|
|
91
|
-
* Run the metadata-enrichment pass over every eligible candidate
|
|
92
|
-
* `reconcileRoots` collected this run, with a bounded concurrency pool
|
|
93
|
-
* (`resolveEnrichmentConcurrency`). Only called when
|
|
94
|
-
* `resolveIndexPassExecution("enrichment", config)` resolves a runner — an
|
|
95
|
-
* unconfigured engine, or `index.enrichment.enabled: false`, is a no-op with
|
|
96
|
-
* zero cache reads and zero provider calls. The separate `metadata_enhance`
|
|
97
|
-
* feature gate (`index.metadataEnhance.enabled`, default `false`) is checked
|
|
98
|
-
* per-call inside `enhanceMetadata` itself, so a closed gate still shows up
|
|
99
|
-
* here as a cheap `skipped` outcome rather than being special-cased twice.
|
|
100
|
-
*
|
|
101
|
-
* A `ConfigError` (a required symbolic credential that resolved to nothing)
|
|
102
|
-
* is not fail-soft like a provider error — `enhanceMetadata` lets it escape
|
|
103
|
-
* `tryLlmFeature`'s normal fallback (`llm/structured-call.ts`'s
|
|
104
|
-
* `callStructured`) precisely so a genuinely broken config surfaces loudly
|
|
105
|
-
* instead of reading as an ordinary per-entry failure. `concurrentMap`
|
|
106
|
-
* itself swallows a thrown callback into an `undefined` slot, so this
|
|
107
|
-
* catches it per-candidate and rethrows the first occurrence once every
|
|
108
|
-
* in-flight candidate has settled.
|
|
109
|
-
*/
|
|
110
|
-
export async function enrichReconciledEntries(db, config, candidates, maxChars, opts) {
|
|
111
|
-
const counts = emptyCounts();
|
|
112
|
-
const eligible = candidates.filter((candidate) => isEligibleForEnrichment(candidate.entry));
|
|
113
|
-
if (eligible.length === 0)
|
|
114
|
-
return counts;
|
|
115
|
-
const runner = resolveIndexPassExecution("enrichment", config).runner;
|
|
116
|
-
if (!runner)
|
|
117
|
-
return counts;
|
|
118
|
-
const concurrency = resolveEnrichmentConcurrency(runner.connection);
|
|
119
|
-
opts?.onProgress?.(`Metadata enrichment starting for ${eligible.length} entr${eligible.length === 1 ? "y" : "ies"} (concurrency ${concurrency}).`);
|
|
120
|
-
let configFailure;
|
|
121
|
-
await concurrentMap(eligible, async (candidate) => {
|
|
122
|
-
if (opts?.signal?.aborted)
|
|
123
|
-
return;
|
|
124
|
-
counts.attempted++;
|
|
125
|
-
try {
|
|
126
|
-
await enrichOneCandidate(db, runner, config, candidate, maxChars, counts, opts?.signal);
|
|
127
|
-
}
|
|
128
|
-
catch (err) {
|
|
129
|
-
if (err instanceof ConfigError) {
|
|
130
|
-
configFailure ??= err;
|
|
131
|
-
return;
|
|
132
|
-
}
|
|
133
|
-
throw err;
|
|
134
|
-
}
|
|
135
|
-
}, concurrency);
|
|
136
|
-
if (configFailure)
|
|
137
|
-
throw configFailure;
|
|
138
|
-
opts?.onProgress?.(`Metadata enrichment finished: ${counts.enriched} enriched (${counts.cacheHits} from cache), ` +
|
|
139
|
-
`${counts.failed} failed, ${counts.skipped} skipped.`);
|
|
140
|
-
if (counts.failed > 0 && counts.enriched === 0 && counts.skipped === 0) {
|
|
141
|
-
warn(`LLM metadata enrichment failed for all ${counts.failed} attempted entr${counts.failed === 1 ? "y" : "ies"} — ` +
|
|
142
|
-
"index built without enrichment. Check the engine selected by index.enrichment.engine (or defaults.llmEngine).");
|
|
143
|
-
}
|
|
144
|
-
return counts;
|
|
145
|
-
}
|
|
146
|
-
async function enrichOneCandidate(db, runner, config, candidate, maxChars, counts, signal) {
|
|
147
|
-
let sawOutcome;
|
|
148
|
-
let cacheHit = false;
|
|
149
|
-
const metadata = await withLlmCache(db, candidate.blobHash, "", false, async () => {
|
|
150
|
-
let fileContent;
|
|
151
|
-
try {
|
|
152
|
-
fileContent = fs.readFileSync(candidate.filePath, "utf8");
|
|
153
|
-
}
|
|
154
|
-
catch {
|
|
155
|
-
// Best-effort context for the prompt only — enhanceMetadata still
|
|
156
|
-
// runs (with less context) when the file cannot be re-read.
|
|
157
|
-
}
|
|
158
|
-
const outcome = await enhanceMetadata(runner, candidate.entry, fileContent, signal, config);
|
|
159
|
-
if (outcome.status !== "enriched") {
|
|
160
|
-
sawOutcome = outcome.status;
|
|
161
|
-
return undefined;
|
|
162
|
-
}
|
|
163
|
-
return outcome.metadata;
|
|
164
|
-
}, (raw) => (raw !== null && typeof raw === "object" ? raw : undefined), candidate.blobHash, METADATA_ENRICHMENT_CACHE_VARIANT, { onCacheHit: () => (cacheHit = true) });
|
|
165
|
-
if (metadata === undefined) {
|
|
166
|
-
if (sawOutcome === "failed")
|
|
167
|
-
counts.failed++;
|
|
168
|
-
else
|
|
169
|
-
counts.skipped++;
|
|
170
|
-
return;
|
|
171
|
-
}
|
|
172
|
-
let applied;
|
|
173
|
-
try {
|
|
174
|
-
applied = applyEnrichmentToEntry(db, candidate, maxChars, metadata);
|
|
175
|
-
}
|
|
176
|
-
catch (err) {
|
|
177
|
-
// A real write failure (not the stale-identity no-op below, which never
|
|
178
|
-
// throws): `concurrentMap` would otherwise swallow this into a silent
|
|
179
|
-
// undefined slot with `counts.enriched` never incremented but no record
|
|
180
|
-
// of the failure either. Surface it the same way a provider failure is
|
|
181
|
-
// already surfaced.
|
|
182
|
-
counts.failed++;
|
|
183
|
-
warn(`[index] Metadata enrichment write failed for ${candidate.filePath}: ` +
|
|
184
|
-
(err instanceof Error ? err.message : String(err)));
|
|
185
|
-
return;
|
|
186
|
-
}
|
|
187
|
-
if (!applied) {
|
|
188
|
-
// The live `entries` row no longer matches the identity this candidate
|
|
189
|
-
// was queued under (a concurrent rename or delete) — see
|
|
190
|
-
// `applyEnrichmentToEntry`. The metadata is real and already cached
|
|
191
|
-
// above, so this is reported as skipped rather than lost, and the next
|
|
192
|
-
// ordinary run re-applies it from cache with no new provider call.
|
|
193
|
-
counts.skipped++;
|
|
194
|
-
return;
|
|
195
|
-
}
|
|
196
|
-
if (cacheHit)
|
|
197
|
-
counts.cacheHits++;
|
|
198
|
-
counts.enriched++;
|
|
199
|
-
}
|
|
200
|
-
/**
|
|
201
|
-
* Merge enrichment fields onto the candidate's entry, re-derive its units,
|
|
202
|
-
* and write both through the SAME canonical entry/FTS mutation and
|
|
203
|
-
* `unit_texts`/`entry_units` maintenance `applyChange` (`reconcile.ts`) uses
|
|
204
|
-
* — one short `BEGIN IMMEDIATE` transaction, no `content_hash` argument so
|
|
205
|
-
* `upsertEntry`'s `COALESCE` preserves the scan-derived blob hash untouched.
|
|
206
|
-
*
|
|
207
|
-
* Fragment units are unaffected: only `description`/`tags`/`searchHints`
|
|
208
|
-
* change, which feeds solely unit ordinal 0 (`structuredFieldsText`,
|
|
209
|
-
* `units/unit.ts`); `replaceEntryUnits` is still a full delete-then-insert
|
|
210
|
-
* for the entry, so `getMarkdownFragmentContent`/`setMarkdownFragmentContent`
|
|
211
|
-
* re-tag the merged copy — otherwise `deriveUnits` would see no markdown
|
|
212
|
-
* body at all and silently drop every fragment unit `applyChange` already
|
|
213
|
-
* derived for this entry.
|
|
214
|
-
*
|
|
215
|
-
* **Stale-identity guard** — `candidate` was captured before the LLM round
|
|
216
|
-
* trip above, which can take seconds to minutes. A concurrent reconcile can
|
|
217
|
-
* rename (`repointEntry` updates the SAME `entries.id` in place with a new
|
|
218
|
-
* `item_ref`/`content_hash`/`file_path`) or delete this row while that call
|
|
219
|
-
* was in flight. Writing the captured values regardless would either (a)
|
|
220
|
-
* `upsertEntry` under the stale `item_ref`, which no longer conflicts with
|
|
221
|
-
* anything and INSERTs a ghost row pointing at an identity that no longer
|
|
222
|
-
* exists, or (b) `replaceEntryUnits(candidate.entryId)` overwriting a live
|
|
223
|
-
* renamed row's units with the old identity's hashes — or, if the row was
|
|
224
|
-
* deleted outright, throw a foreign-key error. So this re-reads the live row
|
|
225
|
-
* BY ID inside the same transaction and no-ops (returns `false`, counted as
|
|
226
|
-
* skipped by the caller) unless its `content_hash` and `item_ref` still
|
|
227
|
-
* match what this candidate was queued under; only then is `candidate.entryId`
|
|
228
|
-
* — now confirmed live, never a captured id that may no longer exist — used
|
|
229
|
-
* to write.
|
|
230
|
-
*/
|
|
231
|
-
function applyEnrichmentToEntry(db, candidate, maxChars, metadata) {
|
|
232
|
-
const merged = { ...candidate.entry, quality: "enriched" };
|
|
233
|
-
if (metadata.description)
|
|
234
|
-
merged.description = metadata.description;
|
|
235
|
-
if (metadata.tags?.length)
|
|
236
|
-
merged.tags = metadata.tags;
|
|
237
|
-
if (metadata.searchHints?.length)
|
|
238
|
-
merged.searchHints = metadata.searchHints;
|
|
239
|
-
if (hasMarkdownFragmentContent(candidate.entry)) {
|
|
240
|
-
setMarkdownFragmentContent(merged, getMarkdownFragmentContent(candidate.entry));
|
|
241
|
-
}
|
|
242
|
-
const searchText = buildSearchText(merged);
|
|
243
|
-
return withImmediateTransaction(db, () => {
|
|
244
|
-
const live = db
|
|
245
|
-
.prepare("SELECT content_hash AS contentHash, item_ref AS itemRef FROM entries WHERE id = ?")
|
|
246
|
-
.get(candidate.entryId);
|
|
247
|
-
if (!live || live.contentHash !== candidate.blobHash || live.itemRef !== candidate.provenance.itemRef) {
|
|
248
|
-
return false;
|
|
249
|
-
}
|
|
250
|
-
upsertEntry(db, candidate.filePath, merged, searchText, candidate.provenance);
|
|
251
|
-
const units = deriveUnits(toUnitSource(candidate.entryId, merged), maxChars);
|
|
252
|
-
insertNewUnitTexts(db, units.map((unit) => ({
|
|
253
|
-
hash: unit.hash,
|
|
254
|
-
kind: unit.fragmentId === null ? "card" : "fragment",
|
|
255
|
-
text: unit.text,
|
|
256
|
-
})));
|
|
257
|
-
replaceEntryUnits(db, candidate.entryId, units.map((unit) => ({ ordinal: unit.ordinal, fragmentId: unit.fragmentId, hash: unit.hash })));
|
|
258
|
-
return true;
|
|
259
|
-
}, "index");
|
|
260
|
-
}
|