akm-cli 0.9.15-beta.1 → 0.9.15-beta.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,184 @@
1
+ // This Source Code Form is subject to the terms of the Mozilla Public
2
+ // License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ /**
5
+ * `index.db` embedding salvage (#955) — a transient, self-emptying table
6
+ * that lets a full rebuild or an index-generation bump reuse vectors instead
7
+ * of re-embedding a corpus whose content did not change.
8
+ *
9
+ * Zero steady-state cost by design: this is NOT a second embedding cache.
10
+ * Rows are copied aside only at the moment they would otherwise be discarded
11
+ * wholesale — a full-index wipe (`persistDirRecords`) or a generation bump
12
+ * (`rebuildIncompatibleIndexGeneration`) — and are consumed by the very next
13
+ * embedding pass (`generateEmbeddingsForDb`). A pass that completes without
14
+ * abort or circuit-break purges whatever is left; an interrupted pass leaves
15
+ * the table for the next attempt to pick up.
16
+ *
17
+ * Reuse is keyed on `sha256(search_text)` plus the fingerprint the vector was
18
+ * generated under — a fingerprint mismatch or a single-byte content change
19
+ * both correctly fall through to a real provider call. `content_hash` is the
20
+ * PRIMARY KEY (not `(content_hash, fingerprint)`) so relabeling a whole
21
+ * generation's fingerprint after a canary "keep" verdict is one UPDATE, and a
22
+ * hash colliding across two discards simply keeps the most recent copy —
23
+ * salvage is a best-effort optimization, not a durable multi-generation
24
+ * archive.
25
+ */
26
+ import { createHash } from "node:crypto";
27
+ import { blobToEmbedding } from "./embeddings-repository.js";
28
+ import { getMeta } from "./index-meta-repository.js";
29
+ import { SQLITE_CHUNK_SIZE } from "./index-sql.js";
30
+ /**
31
+ * Create the salvage table. Additive-only DDL: it carries no bearing on the
32
+ * `entries` generation fingerprint (`hasCanonicalEntrySchema`), so adding it
33
+ * does not require an index-generation bump.
34
+ */
35
+ export function ensureEmbeddingSalvageTable(db) {
36
+ db.exec(`
37
+ CREATE TABLE IF NOT EXISTS embedding_salvage (
38
+ content_hash TEXT PRIMARY KEY,
39
+ fingerprint TEXT NOT NULL,
40
+ embedding BLOB NOT NULL,
41
+ salvaged_at TEXT NOT NULL
42
+ );
43
+ `);
44
+ }
45
+ /** The one hash function salvage writes and reuse lookups must agree on. */
46
+ export function hashEmbeddableText(searchText) {
47
+ return createHash("sha256").update(searchText, "utf8").digest("hex");
48
+ }
49
+ function tableExists(db, name) {
50
+ return db.prepare("SELECT 1 FROM sqlite_master WHERE type='table' AND name=?").get(name) != null;
51
+ }
52
+ function tableHasColumn(db, table, column) {
53
+ const columns = db.prepare(`PRAGMA table_info(${table})`).all();
54
+ return columns.some((c) => c.name === column);
55
+ }
56
+ /**
57
+ * Copy every (hash of search_text, embedding) pair about to be discarded
58
+ * wholesale into `embedding_salvage`, tagged with the `embeddingFingerprint`
59
+ * the discarded vectors were generated under. The caller MUST run this
60
+ * inside the same transaction as the discard that follows it, so the copy
61
+ * and the delete commit or roll back together.
62
+ *
63
+ * Streams `entries JOIN embeddings` in id-ordered pages of
64
+ * {@link SQLITE_CHUNK_SIZE} instead of loading every row into memory before
65
+ * hashing anything — a full rebuild of a large stash otherwise held the
66
+ * entire corpus's search text and vectors in memory at once just to copy
67
+ * them aside (#955, field-report follow-up).
68
+ *
69
+ * A no-op (returns 0) when there is no stored `embeddingFingerprint` to tag
70
+ * rows with (nothing was ever verified against a provider, so there is
71
+ * nothing worth reusing later) or the generation being discarded predates
72
+ * the `entries.search_text` column or has no `embeddings` table at all — an
73
+ * older generation than that has nothing this can safely read.
74
+ */
75
+ export function salvageEmbeddingsBeforeDiscard(db) {
76
+ const fingerprint = getMeta(db, "embeddingFingerprint");
77
+ if (!fingerprint)
78
+ return 0;
79
+ if (!tableExists(db, "entries") || !tableExists(db, "embeddings"))
80
+ return 0;
81
+ if (!tableHasColumn(db, "entries", "search_text"))
82
+ return 0;
83
+ const page = db.prepare("SELECT e.id AS id, e.search_text AS searchText, em.embedding AS embedding " +
84
+ "FROM entries e JOIN embeddings em ON em.id = e.id WHERE e.id > ? ORDER BY e.id LIMIT ?");
85
+ const insert = db.prepare("INSERT OR REPLACE INTO embedding_salvage (content_hash, fingerprint, embedding, salvaged_at) VALUES (?, ?, ?, ?)");
86
+ const salvagedAt = new Date().toISOString();
87
+ let lastId = 0;
88
+ let total = 0;
89
+ for (;;) {
90
+ const rows = page.all(lastId, SQLITE_CHUNK_SIZE);
91
+ if (rows.length === 0)
92
+ break;
93
+ for (const row of rows) {
94
+ insert.run(hashEmbeddableText(row.searchText), fingerprint, row.embedding, salvagedAt);
95
+ }
96
+ total += rows.length;
97
+ lastId = rows[rows.length - 1]?.id ?? lastId;
98
+ if (rows.length < SQLITE_CHUNK_SIZE)
99
+ break;
100
+ }
101
+ return total;
102
+ }
103
+ /**
104
+ * Remove every salvage row. Called after an embedding pass completes without
105
+ * abort or circuit-break (the salvaged generation has now either been reused
106
+ * or superseded), and by `--reembed` / a canary "rebuild" verdict (the
107
+ * salvaged vectors belong to a different model and are never reusable).
108
+ */
109
+ export function purgeEmbeddingSalvage(db) {
110
+ db.exec("DELETE FROM embedding_salvage");
111
+ }
112
+ /**
113
+ * A canary "keep" verdict means the model did not actually change — only its
114
+ * fingerprint STRING did (e.g. a gateway rename). Salvage rows tagged with
115
+ * the old string are still valid vectors; rewrite them to the new string so
116
+ * they remain reusable instead of silently going stale.
117
+ */
118
+ export function relabelEmbeddingSalvageFingerprint(db, fromFingerprint, toFingerprint) {
119
+ db.prepare("UPDATE embedding_salvage SET fingerprint = ? WHERE fingerprint = ?").run(toFingerprint, fromFingerprint);
120
+ }
121
+ /**
122
+ * Reuse salvaged vectors for `entries` whose `searchText` hash matches a
123
+ * salvage row tagged with the CURRENT `fingerprint` — never across
124
+ * fingerprints, and never when `search_text` differs by even one byte (the
125
+ * hash is exact-match only, by design). Matches are written via
126
+ * `writeReused` in chunks of {@link SQLITE_CHUNK_SIZE}, each its own
127
+ * transaction, mirroring the main pass's per-batch commit (#955) so an
128
+ * interruption partway through the reuse step keeps whatever already wrote.
129
+ *
130
+ * The steady state of every ordinary run is an EMPTY salvage table (nothing
131
+ * was just discarded), so this checks that first with one indexed lookup —
132
+ * `SELECT 1 ... LIMIT 1` — before hashing a single pending entry. Hashing
133
+ * every entry up front to look up a table that is empty 100% of the time
134
+ * outside a rebuild was pure wasted work on the common path (#955,
135
+ * field-report follow-up).
136
+ */
137
+ export function reuseSalvagedEmbeddings(db, entries, fingerprint, writeReused) {
138
+ if (entries.length === 0)
139
+ return { reusedCount: 0, remaining: [] };
140
+ const anySalvageForFingerprint = db
141
+ .prepare("SELECT 1 FROM embedding_salvage WHERE fingerprint = ? LIMIT 1")
142
+ .get(fingerprint);
143
+ if (!anySalvageForFingerprint)
144
+ return { reusedCount: 0, remaining: [...entries] };
145
+ const hashes = entries.map((entry) => hashEmbeddableText(entry.searchText));
146
+ const salvageByHash = new Map();
147
+ const uniqueHashes = [...new Set(hashes)];
148
+ for (let offset = 0; offset < uniqueHashes.length; offset += SQLITE_CHUNK_SIZE) {
149
+ const chunk = uniqueHashes.slice(offset, offset + SQLITE_CHUNK_SIZE);
150
+ const placeholders = chunk.map(() => "?").join(",");
151
+ const rows = db
152
+ .prepare(`SELECT content_hash AS contentHash, embedding FROM embedding_salvage WHERE fingerprint = ? AND content_hash IN (${placeholders})`)
153
+ .all(fingerprint, ...chunk);
154
+ for (const row of rows)
155
+ salvageByHash.set(row.contentHash, row.embedding);
156
+ }
157
+ if (salvageByHash.size === 0)
158
+ return { reusedCount: 0, remaining: [...entries] };
159
+ let reusedCount = 0;
160
+ const remaining = [];
161
+ for (let offset = 0; offset < entries.length; offset += SQLITE_CHUNK_SIZE) {
162
+ const end = Math.min(offset + SQLITE_CHUNK_SIZE, entries.length);
163
+ const chunkMatches = [];
164
+ for (let i = offset; i < end; i++) {
165
+ const entry = entries[i];
166
+ const blob = salvageByHash.get(hashes[i]);
167
+ if (blob)
168
+ chunkMatches.push({ entry, blob });
169
+ else
170
+ remaining.push(entry);
171
+ }
172
+ if (chunkMatches.length === 0)
173
+ continue;
174
+ db.transaction(() => {
175
+ for (const { entry, blob } of chunkMatches) {
176
+ if (writeReused(entry, blobToEmbedding(blob)))
177
+ reusedCount++;
178
+ else
179
+ remaining.push(entry);
180
+ }
181
+ })();
182
+ }
183
+ return { reusedCount, remaining };
184
+ }
@@ -12,6 +12,7 @@
12
12
  */
13
13
  import { ConfigError } from "../../core/errors.js";
14
14
  import { warn } from "../../core/warn.js";
15
+ import { ensureEmbeddingSalvageTable, salvageEmbeddingsBeforeDiscard } from "./embedding-salvage-repository.js";
15
16
  import { CANONICAL_ENTRY_SCHEMA_SQL, CANONICAL_INDEX_DB_VERSION, classifyIndexGeneration, isCanonicalIndexGeneration, } from "./index-entry-schema.js";
16
17
  import { getMeta, setMeta } from "./index-meta-repository.js";
17
18
  import { isVecAvailable, purgeEmbeddings } from "./index-vec-repository.js";
@@ -196,6 +197,13 @@ function rebuildIncompatibleIndexGeneration(db) {
196
197
  vecResetPending = true;
197
198
  }
198
199
  db.transaction(() => {
200
+ // #955: copy embeddings about to be discarded wholesale into
201
+ // `embedding_salvage` (keyed by content hash + the fingerprint they were
202
+ // generated under) BEFORE dropping `embeddings`, in the same transaction
203
+ // as the drop, so the copy and the discard commit or roll back together.
204
+ // The next embedding pass hands salvaged vectors back to unchanged
205
+ // content instead of re-embedding the whole corpus after this bump.
206
+ salvageEmbeddingsBeforeDiscard(db);
199
207
  db.exec("DROP TABLE IF EXISTS graph_file_relations");
200
208
  db.exec("DROP TABLE IF EXISTS graph_file_entities");
201
209
  db.exec("DROP TABLE IF EXISTS graph_files");
@@ -212,6 +220,9 @@ function rebuildIncompatibleIndexGeneration(db) {
212
220
  db.exec("DROP TABLE IF EXISTS index_dir_state");
213
221
  db.exec("DROP TABLE IF EXISTS entries");
214
222
  db.exec("DELETE FROM index_meta");
223
+ // embedding_salvage is deliberately absent from the drop list above —
224
+ // it is the ONE piece of derived state a generation rebuild must not
225
+ // discard.
215
226
  })();
216
227
  if (vecResetPending)
217
228
  setMeta(db, "vecResetPending", "1");
@@ -224,6 +235,11 @@ export function ensureSchema(db, embeddingDim) {
224
235
  value TEXT NOT NULL
225
236
  );
226
237
  `);
238
+ // #955: created before the generation-rebuild check below so a discard
239
+ // has somewhere to copy vectors to. Additive-only — it carries no bearing
240
+ // on the `entries` generation fingerprint (`hasCanonicalEntrySchema`), so
241
+ // adding it does not require a `CANONICAL_INDEX_DB_VERSION` bump.
242
+ ensureEmbeddingSalvageTable(db);
227
243
  rebuildIncompatibleIndexGeneration(db);
228
244
  db.exec(CANONICAL_ENTRY_SCHEMA_SQL);
229
245
  // Workflow source is compiled directly into source IR at each command
@@ -145,6 +145,23 @@ export async function runNativeTask(input) {
145
145
  const logLines = [header];
146
146
  const dbLines = [{ line: header }];
147
147
  let exitCode = null;
148
+ // #956: the task runner's OWN process (`akm task run`, launched by cron /
149
+ // launchd / schtasks, or forwarded a SIGTERM by the published launcher)
150
+ // had no way to end its detached, own-process-group
151
+ // child (spawned by `runManagedSubprocess` for the group-kill guarantee
152
+ // above) — a timeout or SIGTERM to the direct child would reap the whole
153
+ // group, but nothing tied THIS process's own termination to that same
154
+ // ladder, so a signal to the runner left its child running as an orphan.
155
+ // Aborting `runManagedSubprocess` here runs its normal SIGTERM→SIGKILL
156
+ // kill ladder against the child's process group.
157
+ const abortController = new AbortController();
158
+ const forwardTerminationSignal = (signal) => {
159
+ abortController.abort(new Error(`task runner received ${signal}`));
160
+ };
161
+ const onSigterm = () => forwardTerminationSignal("SIGTERM");
162
+ const onSigint = () => forwardTerminationSignal("SIGINT");
163
+ process.once("SIGTERM", onSigterm);
164
+ process.once("SIGINT", onSigint);
148
165
  try {
149
166
  // Re-resolve the authored root/cwd immediately before spawn so a
150
167
  // symlink, ancestor, bundle-root, or directory/file swap cannot redirect
@@ -159,7 +176,9 @@ export async function runNativeTask(input) {
159
176
  }
160
177
  // Managed spawn (src/core/subprocess.ts): process-GROUP kill so a timeout
161
178
  // reaps the whole command tree (no orphans), and a SIGTERM→SIGKILL ladder
162
- // so a child that ignores SIGTERM can't wedge the run forever.
179
+ // so a child that ignores SIGTERM can't wedge the run forever. `signal`
180
+ // ties that same ladder to a termination signal received by this
181
+ // process itself (#956), not only to the timeout.
163
182
  const result = await runManagedSubprocess(cmd, {
164
183
  capture: true,
165
184
  cwd: task.cwd,
@@ -178,6 +197,7 @@ export async function runNativeTask(input) {
178
197
  // layer on top and break any resolved path containing a space.
179
198
  windowsVerbatimArguments: task.kind === "shell" && task.shell === "cmd",
180
199
  timeoutMs,
200
+ signal: abortController.signal,
181
201
  ...(input.spawnFn ? { spawnFn: input.spawnFn } : {}),
182
202
  ...(input.setTimeoutFn ? { setTimeoutFn: input.setTimeoutFn } : {}),
183
203
  ...(input.clearTimeoutFn ? { clearTimeoutFn: input.clearTimeoutFn } : {}),
@@ -221,6 +241,8 @@ export async function runNativeTask(input) {
221
241
  exitCode = 1;
222
242
  }
223
243
  finally {
244
+ process.off("SIGTERM", onSigterm);
245
+ process.off("SIGINT", onSigint);
224
246
  if (materialized)
225
247
  cleanupFrozenScript(materialized);
226
248
  }
@@ -38,10 +38,91 @@ scheduled or opportunistic run step aside instead of contending with a rebuild
38
38
  already in progress; the shipped `index-refresh` scheduled task already passes
39
39
  it.
40
40
 
41
- There is no `embedding.concurrency` config field. Embedding throughput is
42
- tuned by `embedding.batchSize` (documents per request) and
43
- `embedding.maxTokens`/`contextLength` (token budget per request); the number of
44
- requests in flight is fixed (1 for a loopback endpoint, 2 for a remote one).
41
+ `embedding.concurrency` (positive integer, 1-16) overrides the number of
42
+ embedding requests kept in flight at once, which otherwise defaults to 1 for
43
+ a loopback endpoint and 2 for a remote one. Set it only for an endpoint that
44
+ genuinely serves parallel requests — a local model server started with a
45
+ multi-slot flag (llama.cpp's `--parallel N`, vLLM) — since the default
46
+ already protects an ordinary single-slot server from reload-thrash.
47
+ Embedding throughput is still tuned first by `embedding.batchSize`
48
+ (documents per request) and `embedding.maxTokens` (token
49
+ budget per request); the concurrency override is a second lever for a
50
+ server that can actually use it.
51
+
52
+ `embedding.timeoutMs` bounds each embedding request (default 120s, up from a
53
+ prior fixed 30s that cut off a slow local model server mid-response). It is
54
+ the budget for a request at the full token budget — a smaller request gets a
55
+ proportionally smaller timeout, so a dead endpoint is still detected in
56
+ seconds on the common case of small documents. A request TIMEOUT no longer
57
+ drops its batch immediately: akm now backs off (5s, doubling, capped at 60s)
58
+ and retries the same request once, since field evidence showed the endpoint
59
+ keeps computing an abandoned request regardless of the client giving up; a
60
+ second timeout splits the batch in half and retries each half the same way,
61
+ down to individual documents, and a single document that still times out is
62
+ finally skipped. After 3 consecutive failures at single-document size
63
+ (timeout or network error), or 3 consecutive network errors at any size —
64
+ never a batch rejected only for exceeding the endpoint's context window —
65
+ `akm index`'s embedding phase stops dispatching further requests and reports
66
+ failure instead of grinding through every remaining batch against a dead
67
+ endpoint — batches already committed are kept, and a rerun picks up where it
68
+ left off.
69
+
70
+ `akm bundle update` now durably commits its embedding pass instead of
71
+ nesting it inside its own transaction: earlier releases ran the embedding
72
+ phase inside the same transaction as content/lock/index/state, so every
73
+ per-batch commit landed as an unobservable SAVEPOINT and a SIGKILL mid-run
74
+ lost every embedding of the update, not just the one in flight. The
75
+ embedding phase now runs on its own connection after the update's own
76
+ commit; a failing pass (provider down) still leaves the update itself
77
+ successful, with the new `index.semanticStatus` field on `akm bundle
78
+ update`'s response the only sign semantic search fell behind.
79
+
80
+ The published `akm`/`akm-migrate` launchers now forward SIGTERM/SIGINT/
81
+ SIGHUP to their child and exit alongside it, instead of leaving the child
82
+ running as an orphan when only the launcher is signaled. No action needed —
83
+ this is a drop-in fix for anyone running `akm` under a scheduler,
84
+ supervisor, or hook that can time out or kill the launcher process
85
+ directly.
86
+
87
+ `embedding.maxInputTokens` (default 512) now caps how much of a single
88
+ document's text is sent to the embedding provider, truncating to the head
89
+ instead of ever failing a whole batch over one oversized document.
90
+
91
+ - An existing install's already-stored vectors are untouched and stay
92
+ valid — this only changes what happens for entries embedded *after*
93
+ upgrading.
94
+ - New embeddings (any entry indexed for the first time, or re-indexed after
95
+ a content change) go through the new 512-token cap by default. If you
96
+ were relying on documents longer than ~2000 characters being embedded in
97
+ full, set `embedding.maxInputTokens` higher in `config.json`.
98
+ - `akm index --reembed` re-embeds every entry under the new cap — run it if
99
+ you want your entire existing index rebuilt against the new default (or a
100
+ custom `embedding.maxInputTokens` you've set).
101
+ - `embedding.contextLength` is Ollama's `num_ctx` only now; it no longer
102
+ also sets the per-request token budget (`embedding.maxTokens`). If you had
103
+ set `contextLength` specifically to control request batching (not your
104
+ Ollama server's context window), set `embedding.maxTokens` instead.
105
+
106
+ `akm index --full` and an index-generation bump no longer re-embed
107
+ unchanged content: vectors about to be discarded are salvaged and handed
108
+ back to unchanged entries at the start of the next embedding pass instead
109
+ of every upgrade re-embedding the whole corpus once. No action needed —
110
+ this is automatic; `akm index --reembed` still forces a full re-embed when
111
+ you don't trust the salvaged vectors.
112
+
113
+ A field report suspected `akm index` was sending embedding requests with no
114
+ `Authorization` header despite `embedding.apiKey` being set to a
115
+ `secret://` reference. Auditing every path that builds an embedding request
116
+ found all of them already resolve `secret://` through the same store
117
+ lookup, now pinned by integration and contract tests — this was not a bug
118
+ in the code as it stands. `akm index` now prints one line before its first
119
+ provider request naming the endpoint, model, and credential SOURCE (never
120
+ the value), e.g. `[embed] endpoint http://.../v1/embeddings, model
121
+ nomic-embed; credential: secret://lab-api-key (store)`. If you run an
122
+ embedding gateway that enforces auth and still see unauthenticated requests
123
+ after upgrading, compare this line's endpoint and credential source against
124
+ what the gateway's own request log shows for the same request — a
125
+ mismatch there (not in this line) is the next place to look.
45
126
 
46
127
  A config file can now inherit a shared base via `extends: <path|bundle//path>`,
47
128
  deep-merging under the local file so local keys always win. `akm config diff
@@ -9,8 +9,9 @@ live one level up in `docs/migration/`.
9
9
 
10
10
  - [0.9.15](0.9.15.md) — exit-code 75 for lease/state.db contention,
11
11
  `--require-engines` scheduled task templates, `--no-probe` cli-version
12
- skip, thinking-control wire forms, embedding re-embed safety, and
13
- `extends` config inheritance
12
+ skip, thinking-control wire forms, embedding re-embed safety and
13
+ cross-rebuild vector salvage, launcher signal forwarding, a default
14
+ per-document embedding cap, and `extends` config inheritance
14
15
  - [0.9.14](0.9.14.md) — index v22-to-v23 derived-cache rebuild, lexical
15
16
  fragments, and collapse-detector canary re-minting
16
17
  - [0.9.2](0.9.2.md) — task source v4 migration, workflow source IR v1 and
@@ -219,7 +219,7 @@ Build or refresh the search index.
219
219
 
220
220
  ```sh
221
221
  akm index # Incremental (only changed directories)
222
- akm index --full # Full rebuild
222
+ akm index --full # Full rebuild (reuses unchanged embeddings — see below)
223
223
  akm index --verbose # Print phase progress to stderr
224
224
  akm index --clean # Normal index + remove stale entries from the DB
225
225
  akm index --clean --dry-run # Report stale entries without deleting
@@ -234,6 +234,18 @@ semantic-search settings, and phase-by-phase progress to stderr while the
234
234
  index is being built. Malformed workflow assets are skipped with file-path
235
235
  warnings instead of aborting the full run.
236
236
 
237
+ **Progress in non-verbose JSON mode (default output format, #954):** even
238
+ without `--verbose`, phase-start messages and the embedding heartbeat
239
+ (`Still generating embeddings: X/N stored, F failed; waiting on embedding
240
+ provider.`) are now written to stderr, and a failed embedding batch logs at
241
+ the default level instead of `--verbose`-only — a long-running index build
242
+ against a slow or unresponsive provider is no longer silent until the whole
243
+ run finishes. Text-mode output keeps its spinner instead (no stderr line
244
+ growth); JSON stdout output is unaffected either way. The high-frequency
245
+ per-batch `Embedded N/M entries.` line stays out of non-verbose stderr (it
246
+ fires after every committed batch) — pass `--verbose` for that level of
247
+ detail.
248
+
237
249
  **`--clean` flag:** After indexing completes, verifies every indexed entry's source
238
250
  file still exists on disk. Removes any entries whose file is missing (for local
239
251
  bundle sources only; remote entries are skipped). Returns a `clean` block in the
@@ -242,6 +254,18 @@ Use `--clean` to resolve the edge case where a deleted file in an unchanged
242
254
  directory lingers in the index across incremental runs. With `--dry-run`, reports
243
255
  which entries would be removed without modifying the database.
244
256
 
257
+ **`--full` no longer re-embeds unchanged content (#955):** a full rebuild
258
+ (and an index-generation bump on first open under a new binary) used to
259
+ delete every embedding unconditionally, forcing a full re-embed of the
260
+ whole corpus even when nothing changed. Vectors about to be discarded are
261
+ now salvaged (keyed by a hash of their content plus the fingerprint they
262
+ were generated under) and handed straight back to unchanged entries at the
263
+ start of the next embedding pass, with zero provider calls for them — a
264
+ progress line reports the split (`Reused N embeddings from the previous
265
+ generation; embedding M new.`). Content that changed even by one byte, or
266
+ a fingerprint that no longer matches, still goes through the provider
267
+ normally. `--reembed` is the way to force a full re-embed regardless.
268
+
245
269
  **`--reembed` flag:** Forces a full purge and re-embed of every entry,
246
270
  independent of the embedding-model-rename compatibility check described
247
271
  below. Ordinary indexing already tells a config-only rename of
@@ -258,8 +282,9 @@ advisory, never the blocking lock #872 removed (see
258
282
  the lock, it warns and proceeds anyway, contending with the existing run.
259
283
  `--skip-if-locked` changes that only for the invocation that passes it: if
260
284
  the lock is already held by a live process, it skips gracefully (exit 0,
261
- `{ ok: true, skipped: { reason: "lock-held", pid, startedAt } }`) instead of
262
- contending. `akm index` and `akm curate` are both safe to call frequently —
285
+ `{ ok: true, skipped: { reason: "lock-held", pid, launcherPid, startedAt } }`
286
+ — `launcherPid` is the holder's launcher pid when known, `null` otherwise,
287
+ #956) instead of contending. `akm index` and `akm curate` are both safe to call frequently —
263
288
  `curate` never blocks on a rebuild in progress ([read-path indexing stays
264
289
  non-blocking](#curate)) — but a hook, cron job, or scheduled task that
265
290
  invokes `akm index` directly should pass `--skip-if-locked` so it steps
@@ -396,21 +396,62 @@ unless a remote `embedding` config is provided.
396
396
  `akm improve`'s memory-inference/consolidate passes when they call an
397
397
  embedding model: `provider`, `endpoint`, `model`, `apiKey` (symbolic
398
398
  reference, same rules as engine `apiKey`), `dimension`, `localModel`,
399
- `maxTokens`, `batchSize`, `chunkSize`, `contextLength`, and
400
- `ollamaOptions.num_ctx`.
401
-
402
- `akm index` keeps a small, fixed number of `/v1/embeddings` requests in
403
- flight at once (a remote endpoint only; the local transformer path is
404
- unaffected): `1` for a loopback endpoint (`localhost`, `127.0.0.0/8`, etc. —
405
- a local model server serves one inference at a time, and parallel requests
406
- thrash it) and `2` for a remote one. This width is not configurable. The
407
- actual throughput knob is request SIZE, not request count: `embedding.batchSize`
408
- (a document-count cap, default 100) together with `embedding.maxTokens` /
409
- `embedding.contextLength` (an estimated token budget per request, default
410
- 8000) control how many documents land in one request — a batch of 16-32
411
- documents takes about the same wall time as a single one against a healthy
412
- endpoint, so growing the batch is where most of the win is, not adding more
413
- concurrent requests.
399
+ `maxInputTokens`, `maxTokens`, `batchSize`, `chunkSize`, `contextLength`,
400
+ `timeoutMs`, `concurrency`, and `ollamaOptions.num_ctx`.
401
+
402
+ The knobs that bound request/document size and rate, all optional (defaults
403
+ apply when unset), for a remote endpoint (`src/llm/embedders/remote.ts`):
404
+
405
+ | Key | Default | Bounds |
406
+ | --- | --- | --- |
407
+ | `embedding.maxInputTokens` | `512` | Per-DOCUMENT cap, applied before batching (#956). A document's embedded text is truncated to its head (unicode-safe) at this many estimated tokens instead of ever being skipped for size alone — a document is skipped only when its truncated head is empty. |
408
+ | `embedding.maxTokens` | `8000` (`DEFAULT_TOKEN_BUDGET`) | Per-REQUEST token budget: how many (already-capped) documents' estimated tokens fit in one HTTP request. With the 512-token default document cap, a request carries about 16 documents by default. |
409
+ | `embedding.batchSize` | `100` | Per-REQUEST document-COUNT safety cap, independent of the token budget — guards against many tiny documents packing an oversized request. |
410
+ | `embedding.contextLength` | unset | Ollama's `num_ctx` ONLY, forwarded verbatim as `options.num_ctx` on the native `/api/embed` request. Does **not** feed the request token budget above (#956) — the two used to share this one field, so setting it for the server's context window silently changed request batching too. |
411
+ | `embedding.timeoutMs` | `120000` (120s) | Per-request wall timeout — see below. |
412
+ | `embedding.concurrency` | `1` loopback / `2` remote | In-flight request window — see below. |
413
+
414
+ `embedding.timeoutMs` (positive integer, default `120000` — 120s) is the
415
+ budget for a request at the FULL token budget (`embedding.maxTokens`); a
416
+ local model server on a large, token-budget-bounded batch legitimately takes
417
+ longer than the prior fixed 30s cut off. A smaller request gets a
418
+ proportionally smaller timeout —
419
+ `clamp(timeoutMs × requestTokens / tokenBudget, 30000, timeoutMs)` — so a
420
+ dead endpoint is still detected in seconds on the common case of small
421
+ documents. Set `embedding.timeoutMs` lower to fail fast against a
422
+ known-fast endpoint, or higher for a slow local server on large batches.
423
+
424
+ A request TIMEOUT (not a rejection for exceeding the context window) never
425
+ drops its batch immediately: field confirmation showed that once akm
426
+ abandons a timed-out request, the endpoint (e.g. llama-server) keeps
427
+ computing it anyway, so dropping it right away just grows the provider's
428
+ queue while every following batch dies the same way. Instead akm backs off
429
+ (5s, doubling, capped at 60s) and retries the same request once; a second
430
+ timeout splits it in half and retries each half the same way, down to
431
+ individual documents, and a single document that still times out is finally
432
+ skipped (logged at the default `warn` level). After 3 consecutive failures
433
+ at single-document size (timeout or network error), or 3 consecutive
434
+ network errors at any size, the embedding phase stops dispatching further
435
+ requests and reports failure — batches already committed are kept; rerun
436
+ `akm index` once the endpoint is healthy.
437
+
438
+ `akm index` keeps a small number of `/v1/embeddings` requests in flight at
439
+ once (a remote endpoint only; the local transformer path is unaffected):
440
+ `1` for a loopback endpoint (`localhost`, `127.0.0.0/8`, etc. — a local
441
+ model server serves one inference at a time, and parallel requests thrash
442
+ it) and `2` for a remote one, unless `embedding.concurrency` (positive
443
+ integer, 1-16) overrides it. This default holds for the overwhelming
444
+ majority of setups; set the override only for an endpoint that genuinely
445
+ serves parallel requests — a local server started with a multi-slot flag
446
+ (llama.cpp's `--parallel N`, vLLM) — not to "speed up" an ordinary
447
+ single-slot model server, which the default already protects from
448
+ reload-thrash. Request SIZE remains the first throughput lever regardless:
449
+ `embedding.batchSize` (a document-count cap, default 100) together with
450
+ `embedding.maxTokens` (an estimated token budget per request, default 8000
451
+ — NOT `embedding.contextLength`, see the table above) control how many
452
+ documents land in one request — with the default 512-token
453
+ `embedding.maxInputTokens` document cap, that is about 16-32 documents,
454
+ taking about the same wall time as a single one against a healthy endpoint.
414
455
 
415
456
  ## Search tuning
416
457
 
@@ -674,6 +715,17 @@ embedding calls have since 0.9.13 (#917); resolution order for a single
674
715
  `apiKeyFile`, then `secret://<name>` — though in practice a config sets only
675
716
  one of the three per engine.
676
717
 
718
+ `embedding.apiKey` accepts the same three forms and resolves `secret://` the
719
+ same way, on every path that sends an embedding request: `akm index`
720
+ (including its `bundle update` post-commit embedding pass and the targeted
721
+ re-embed a write command like `akm remember` triggers), `akm improve`'s
722
+ consolidate pass (memory dedup and similarity clustering), and the
723
+ fingerprint-rename canary `akm index` runs when the embedding config
724
+ changes. All of them build the
725
+ provider request through the same `RemoteEmbedder`/`resolveSecret` boundary,
726
+ so a `secret://` reference resolves identically regardless of which command
727
+ triggered the request (#953).
728
+
677
729
  Use `AKM_SQLITE_JOURNAL_MODE=DELETE` or `TRUNCATE` when WAL is unavailable,
678
730
  such as on some NFS/SMB mounts. With the default `WAL` setting, AKM detects a
679
731
  network filesystem for the data directory and falls back to `DELETE`.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "akm-cli",
3
- "version": "0.9.15-beta.1",
3
+ "version": "0.9.15-beta.2",
4
4
  "type": "module",
5
5
  "description": "akm (Agent Knowledge Manager) — a portable, local-first capability library for AI agents. Discover, load, share, and improve reusable skills, scripts, workflows, and knowledge across any shell-capable coding agent, including Claude Code, OpenCode, and Cursor.",
6
6
  "keywords": [
@@ -211,6 +211,10 @@
211
211
  "type": "string",
212
212
  "minLength": 1
213
213
  },
214
+ "maxInputTokens": {
215
+ "type": "integer",
216
+ "exclusiveMinimum": 0
217
+ },
214
218
  "maxTokens": {
215
219
  "type": "integer",
216
220
  "exclusiveMinimum": 0
@@ -236,6 +240,15 @@
236
240
  }
237
241
  },
238
242
  "additionalProperties": true
243
+ },
244
+ "timeoutMs": {
245
+ "type": "integer",
246
+ "exclusiveMinimum": 0
247
+ },
248
+ "concurrency": {
249
+ "type": "integer",
250
+ "exclusiveMinimum": 0,
251
+ "maximum": 16
239
252
  }
240
253
  },
241
254
  "additionalProperties": true
@@ -1902,6 +1915,10 @@
1902
1915
  "type": "string",
1903
1916
  "minLength": 1
1904
1917
  },
1918
+ "maxInputTokens": {
1919
+ "type": "integer",
1920
+ "exclusiveMinimum": 0
1921
+ },
1905
1922
  "maxTokens": {
1906
1923
  "type": "integer",
1907
1924
  "exclusiveMinimum": 0
@@ -1927,6 +1944,15 @@
1927
1944
  }
1928
1945
  },
1929
1946
  "additionalProperties": true
1947
+ },
1948
+ "timeoutMs": {
1949
+ "type": "integer",
1950
+ "exclusiveMinimum": 0
1951
+ },
1952
+ "concurrency": {
1953
+ "type": "integer",
1954
+ "exclusiveMinimum": 0,
1955
+ "maximum": 16
1930
1956
  }
1931
1957
  },
1932
1958
  "additionalProperties": true
@@ -3471,6 +3497,10 @@
3471
3497
  "type": "string",
3472
3498
  "minLength": 1
3473
3499
  },
3500
+ "maxInputTokens": {
3501
+ "type": "integer",
3502
+ "exclusiveMinimum": 0
3503
+ },
3474
3504
  "maxTokens": {
3475
3505
  "type": "integer",
3476
3506
  "exclusiveMinimum": 0
@@ -3496,6 +3526,15 @@
3496
3526
  }
3497
3527
  },
3498
3528
  "additionalProperties": true
3529
+ },
3530
+ "timeoutMs": {
3531
+ "type": "integer",
3532
+ "exclusiveMinimum": 0
3533
+ },
3534
+ "concurrency": {
3535
+ "type": "integer",
3536
+ "exclusiveMinimum": 0,
3537
+ "maximum": 16
3499
3538
  }
3500
3539
  },
3501
3540
  "additionalProperties": true