akm-cli 0.9.15-beta.1 → 0.9.15-beta.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,184 @@
1
+ // This Source Code Form is subject to the terms of the Mozilla Public
2
+ // License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ /**
5
+ * `index.db` embedding salvage (#955) — a transient, self-emptying table
6
+ * that lets a full rebuild or an index-generation bump reuse vectors instead
7
+ * of re-embedding a corpus whose content did not change.
8
+ *
9
+ * Zero steady-state cost by design: this is NOT a second embedding cache.
10
+ * Rows are copied aside only at the moment they would otherwise be discarded
11
+ * wholesale — a full-index wipe (`persistDirRecords`) or a generation bump
12
+ * (`rebuildIncompatibleIndexGeneration`) — and are consumed by the very next
13
+ * embedding pass (`generateEmbeddingsForDb`). A pass that completes without
14
+ * abort or circuit-break purges whatever is left; an interrupted pass leaves
15
+ * the table for the next attempt to pick up.
16
+ *
17
+ * Reuse is keyed on `sha256(search_text)` plus the fingerprint the vector was
18
+ * generated under — a fingerprint mismatch or a single-byte content change
19
+ * both correctly fall through to a real provider call. `content_hash` is the
20
+ * PRIMARY KEY (not `(content_hash, fingerprint)`) so relabeling a whole
21
+ * generation's fingerprint after a canary "keep" verdict is one UPDATE, and a
22
+ * hash colliding across two discards simply keeps the most recent copy —
23
+ * salvage is a best-effort optimization, not a durable multi-generation
24
+ * archive.
25
+ */
26
+ import { createHash } from "node:crypto";
27
+ import { blobToEmbedding } from "./embeddings-repository.js";
28
+ import { getMeta } from "./index-meta-repository.js";
29
+ import { SQLITE_CHUNK_SIZE } from "./index-sql.js";
30
+ /**
31
+ * Create the salvage table. Additive-only DDL: it carries no bearing on the
32
+ * `entries` generation fingerprint (`hasCanonicalEntrySchema`), so adding it
33
+ * does not require an index-generation bump.
34
+ */
35
+ export function ensureEmbeddingSalvageTable(db) {
36
+ db.exec(`
37
+ CREATE TABLE IF NOT EXISTS embedding_salvage (
38
+ content_hash TEXT PRIMARY KEY,
39
+ fingerprint TEXT NOT NULL,
40
+ embedding BLOB NOT NULL,
41
+ salvaged_at TEXT NOT NULL
42
+ );
43
+ `);
44
+ }
45
+ /** The one hash function salvage writes and reuse lookups must agree on. */
46
+ export function hashEmbeddableText(searchText) {
47
+ return createHash("sha256").update(searchText, "utf8").digest("hex");
48
+ }
49
+ function tableExists(db, name) {
50
+ return db.prepare("SELECT 1 FROM sqlite_master WHERE type='table' AND name=?").get(name) != null;
51
+ }
52
+ function tableHasColumn(db, table, column) {
53
+ const columns = db.prepare(`PRAGMA table_info(${table})`).all();
54
+ return columns.some((c) => c.name === column);
55
+ }
56
+ /**
57
+ * Copy every (hash of search_text, embedding) pair about to be discarded
58
+ * wholesale into `embedding_salvage`, tagged with the `embeddingFingerprint`
59
+ * the discarded vectors were generated under. The caller MUST run this
60
+ * inside the same transaction as the discard that follows it, so the copy
61
+ * and the delete commit or roll back together.
62
+ *
63
+ * Streams `entries JOIN embeddings` in id-ordered pages of
64
+ * {@link SQLITE_CHUNK_SIZE} instead of loading every row into memory before
65
+ * hashing anything — a full rebuild of a large stash otherwise held the
66
+ * entire corpus's search text and vectors in memory at once just to copy
67
+ * them aside (#955, field-report follow-up).
68
+ *
69
+ * A no-op (returns 0) when there is no stored `embeddingFingerprint` to tag
70
+ * rows with (nothing was ever verified against a provider, so there is
71
+ * nothing worth reusing later) or the generation being discarded predates
72
+ * the `entries.search_text` column or has no `embeddings` table at all — an
73
+ * older generation than that has nothing this can safely read.
74
+ */
75
+ export function salvageEmbeddingsBeforeDiscard(db) {
76
+ const fingerprint = getMeta(db, "embeddingFingerprint");
77
+ if (!fingerprint)
78
+ return 0;
79
+ if (!tableExists(db, "entries") || !tableExists(db, "embeddings"))
80
+ return 0;
81
+ if (!tableHasColumn(db, "entries", "search_text"))
82
+ return 0;
83
+ const page = db.prepare("SELECT e.id AS id, e.search_text AS searchText, em.embedding AS embedding " +
84
+ "FROM entries e JOIN embeddings em ON em.id = e.id WHERE e.id > ? ORDER BY e.id LIMIT ?");
85
+ const insert = db.prepare("INSERT OR REPLACE INTO embedding_salvage (content_hash, fingerprint, embedding, salvaged_at) VALUES (?, ?, ?, ?)");
86
+ const salvagedAt = new Date().toISOString();
87
+ let lastId = 0;
88
+ let total = 0;
89
+ for (;;) {
90
+ const rows = page.all(lastId, SQLITE_CHUNK_SIZE);
91
+ if (rows.length === 0)
92
+ break;
93
+ for (const row of rows) {
94
+ insert.run(hashEmbeddableText(row.searchText), fingerprint, row.embedding, salvagedAt);
95
+ }
96
+ total += rows.length;
97
+ lastId = rows[rows.length - 1]?.id ?? lastId;
98
+ if (rows.length < SQLITE_CHUNK_SIZE)
99
+ break;
100
+ }
101
+ return total;
102
+ }
103
+ /**
104
+ * Remove every salvage row. Called after an embedding pass completes without
105
+ * abort or circuit-break (the salvaged generation has now either been reused
106
+ * or superseded), and by `--reembed` / a canary "rebuild" verdict (the
107
+ * salvaged vectors belong to a different model and are never reusable).
108
+ */
109
+ export function purgeEmbeddingSalvage(db) {
110
+ db.exec("DELETE FROM embedding_salvage");
111
+ }
112
+ /**
113
+ * A canary "keep" verdict means the model did not actually change — only its
114
+ * fingerprint STRING did (e.g. a gateway rename). Salvage rows tagged with
115
+ * the old string are still valid vectors; rewrite them to the new string so
116
+ * they remain reusable instead of silently going stale.
117
+ */
118
+ export function relabelEmbeddingSalvageFingerprint(db, fromFingerprint, toFingerprint) {
119
+ db.prepare("UPDATE embedding_salvage SET fingerprint = ? WHERE fingerprint = ?").run(toFingerprint, fromFingerprint);
120
+ }
121
+ /**
122
+ * Reuse salvaged vectors for `entries` whose `searchText` hash matches a
123
+ * salvage row tagged with the CURRENT `fingerprint` — never across
124
+ * fingerprints, and never when `search_text` differs by even one byte (the
125
+ * hash is exact-match only, by design). Matches are written via
126
+ * `writeReused` in chunks of {@link SQLITE_CHUNK_SIZE}, each its own
127
+ * transaction, mirroring the main pass's per-batch commit (#955) so an
128
+ * interruption partway through the reuse step keeps whatever already wrote.
129
+ *
130
+ * The steady state of every ordinary run is an EMPTY salvage table (nothing
131
+ * was just discarded), so this checks that first with one indexed lookup —
132
+ * `SELECT 1 ... LIMIT 1` — before hashing a single pending entry. Hashing
133
+ * every entry up front to look up a table that is empty 100% of the time
134
+ * outside a rebuild was pure wasted work on the common path (#955,
135
+ * field-report follow-up).
136
+ */
137
+ export function reuseSalvagedEmbeddings(db, entries, fingerprint, writeReused) {
138
+ if (entries.length === 0)
139
+ return { reusedCount: 0, remaining: [] };
140
+ const anySalvageForFingerprint = db
141
+ .prepare("SELECT 1 FROM embedding_salvage WHERE fingerprint = ? LIMIT 1")
142
+ .get(fingerprint);
143
+ if (!anySalvageForFingerprint)
144
+ return { reusedCount: 0, remaining: [...entries] };
145
+ const hashes = entries.map((entry) => hashEmbeddableText(entry.searchText));
146
+ const salvageByHash = new Map();
147
+ const uniqueHashes = [...new Set(hashes)];
148
+ for (let offset = 0; offset < uniqueHashes.length; offset += SQLITE_CHUNK_SIZE) {
149
+ const chunk = uniqueHashes.slice(offset, offset + SQLITE_CHUNK_SIZE);
150
+ const placeholders = chunk.map(() => "?").join(",");
151
+ const rows = db
152
+ .prepare(`SELECT content_hash AS contentHash, embedding FROM embedding_salvage WHERE fingerprint = ? AND content_hash IN (${placeholders})`)
153
+ .all(fingerprint, ...chunk);
154
+ for (const row of rows)
155
+ salvageByHash.set(row.contentHash, row.embedding);
156
+ }
157
+ if (salvageByHash.size === 0)
158
+ return { reusedCount: 0, remaining: [...entries] };
159
+ let reusedCount = 0;
160
+ const remaining = [];
161
+ for (let offset = 0; offset < entries.length; offset += SQLITE_CHUNK_SIZE) {
162
+ const end = Math.min(offset + SQLITE_CHUNK_SIZE, entries.length);
163
+ const chunkMatches = [];
164
+ for (let i = offset; i < end; i++) {
165
+ const entry = entries[i];
166
+ const blob = salvageByHash.get(hashes[i]);
167
+ if (blob)
168
+ chunkMatches.push({ entry, blob });
169
+ else
170
+ remaining.push(entry);
171
+ }
172
+ if (chunkMatches.length === 0)
173
+ continue;
174
+ db.transaction(() => {
175
+ for (const { entry, blob } of chunkMatches) {
176
+ if (writeReused(entry, blobToEmbedding(blob)))
177
+ reusedCount++;
178
+ else
179
+ remaining.push(entry);
180
+ }
181
+ })();
182
+ }
183
+ return { reusedCount, remaining };
184
+ }
@@ -12,6 +12,7 @@
12
12
  */
13
13
  import { ConfigError } from "../../core/errors.js";
14
14
  import { warn } from "../../core/warn.js";
15
+ import { ensureEmbeddingSalvageTable, salvageEmbeddingsBeforeDiscard } from "./embedding-salvage-repository.js";
15
16
  import { CANONICAL_ENTRY_SCHEMA_SQL, CANONICAL_INDEX_DB_VERSION, classifyIndexGeneration, isCanonicalIndexGeneration, } from "./index-entry-schema.js";
16
17
  import { getMeta, setMeta } from "./index-meta-repository.js";
17
18
  import { isVecAvailable, purgeEmbeddings } from "./index-vec-repository.js";
@@ -196,6 +197,13 @@ function rebuildIncompatibleIndexGeneration(db) {
196
197
  vecResetPending = true;
197
198
  }
198
199
  db.transaction(() => {
200
+ // #955: copy embeddings about to be discarded wholesale into
201
+ // `embedding_salvage` (keyed by content hash + the fingerprint they were
202
+ // generated under) BEFORE dropping `embeddings`, in the same transaction
203
+ // as the drop, so the copy and the discard commit or roll back together.
204
+ // The next embedding pass hands salvaged vectors back to unchanged
205
+ // content instead of re-embedding the whole corpus after this bump.
206
+ salvageEmbeddingsBeforeDiscard(db);
199
207
  db.exec("DROP TABLE IF EXISTS graph_file_relations");
200
208
  db.exec("DROP TABLE IF EXISTS graph_file_entities");
201
209
  db.exec("DROP TABLE IF EXISTS graph_files");
@@ -212,6 +220,9 @@ function rebuildIncompatibleIndexGeneration(db) {
212
220
  db.exec("DROP TABLE IF EXISTS index_dir_state");
213
221
  db.exec("DROP TABLE IF EXISTS entries");
214
222
  db.exec("DELETE FROM index_meta");
223
+ // embedding_salvage is deliberately absent from the drop list above —
224
+ // it is the ONE piece of derived state a generation rebuild must not
225
+ // discard.
215
226
  })();
216
227
  if (vecResetPending)
217
228
  setMeta(db, "vecResetPending", "1");
@@ -224,6 +235,11 @@ export function ensureSchema(db, embeddingDim) {
224
235
  value TEXT NOT NULL
225
236
  );
226
237
  `);
238
+ // #955: created before the generation-rebuild check below so a discard
239
+ // has somewhere to copy vectors to. Additive-only — it carries no bearing
240
+ // on the `entries` generation fingerprint (`hasCanonicalEntrySchema`), so
241
+ // adding it does not require a `CANONICAL_INDEX_DB_VERSION` bump.
242
+ ensureEmbeddingSalvageTable(db);
227
243
  rebuildIncompatibleIndexGeneration(db);
228
244
  db.exec(CANONICAL_ENTRY_SCHEMA_SQL);
229
245
  // Workflow source is compiled directly into source IR at each command
@@ -145,6 +145,23 @@ export async function runNativeTask(input) {
145
145
  const logLines = [header];
146
146
  const dbLines = [{ line: header }];
147
147
  let exitCode = null;
148
+ // #956: the task runner's OWN process (`akm task run`, launched by cron /
149
+ // launchd / schtasks, or forwarded a SIGTERM by the published launcher)
150
+ // had no way to end its detached, own-process-group
151
+ // child (spawned by `runManagedSubprocess` for the group-kill guarantee
152
+ // above) — a timeout or SIGTERM to the direct child would reap the whole
153
+ // group, but nothing tied THIS process's own termination to that same
154
+ // ladder, so a signal to the runner left its child running as an orphan.
155
+ // Aborting `runManagedSubprocess` here runs its normal SIGTERM→SIGKILL
156
+ // kill ladder against the child's process group.
157
+ const abortController = new AbortController();
158
+ const forwardTerminationSignal = (signal) => {
159
+ abortController.abort(new Error(`task runner received ${signal}`));
160
+ };
161
+ const onSigterm = () => forwardTerminationSignal("SIGTERM");
162
+ const onSigint = () => forwardTerminationSignal("SIGINT");
163
+ process.once("SIGTERM", onSigterm);
164
+ process.once("SIGINT", onSigint);
148
165
  try {
149
166
  // Re-resolve the authored root/cwd immediately before spawn so a
150
167
  // symlink, ancestor, bundle-root, or directory/file swap cannot redirect
@@ -159,7 +176,9 @@ export async function runNativeTask(input) {
159
176
  }
160
177
  // Managed spawn (src/core/subprocess.ts): process-GROUP kill so a timeout
161
178
  // reaps the whole command tree (no orphans), and a SIGTERM→SIGKILL ladder
162
- // so a child that ignores SIGTERM can't wedge the run forever.
179
+ // so a child that ignores SIGTERM can't wedge the run forever. `signal`
180
+ // ties that same ladder to a termination signal received by this
181
+ // process itself (#956), not only to the timeout.
163
182
  const result = await runManagedSubprocess(cmd, {
164
183
  capture: true,
165
184
  cwd: task.cwd,
@@ -178,6 +197,7 @@ export async function runNativeTask(input) {
178
197
  // layer on top and break any resolved path containing a space.
179
198
  windowsVerbatimArguments: task.kind === "shell" && task.shell === "cmd",
180
199
  timeoutMs,
200
+ signal: abortController.signal,
181
201
  ...(input.spawnFn ? { spawnFn: input.spawnFn } : {}),
182
202
  ...(input.setTimeoutFn ? { setTimeoutFn: input.setTimeoutFn } : {}),
183
203
  ...(input.clearTimeoutFn ? { clearTimeoutFn: input.clearTimeoutFn } : {}),
@@ -221,6 +241,8 @@ export async function runNativeTask(input) {
221
241
  exitCode = 1;
222
242
  }
223
243
  finally {
244
+ process.off("SIGTERM", onSigterm);
245
+ process.off("SIGINT", onSigint);
224
246
  if (materialized)
225
247
  cleanupFrozenScript(materialized);
226
248
  }
@@ -38,10 +38,102 @@ scheduled or opportunistic run step aside instead of contending with a rebuild
38
38
  already in progress; the shipped `index-refresh` scheduled task already passes
39
39
  it.
40
40
 
41
- There is no `embedding.concurrency` config field. Embedding throughput is
42
- tuned by `embedding.batchSize` (documents per request) and
43
- `embedding.maxTokens`/`contextLength` (token budget per request); the number of
44
- requests in flight is fixed (1 for a loopback endpoint, 2 for a remote one).
41
+ `embedding.maxTokens`'s default (the per-request token budget) is now 6000,
42
+ down from 8000: a field report on an 8192-token llama.cpp embedder showed the
43
+ 4-chars-per-token estimator undercounts dense technical text by 7-55%, so the
44
+ old default regularly overshot the endpoint's real context window. If you
45
+ already set `embedding.maxTokens` explicitly, this default change does not
46
+ affect you — your configured value is unchanged. `akm index` also now
47
+ recovers automatically within a run: on the first request rejected for
48
+ exceeding the endpoint's context window, it lowers its effective budget for
49
+ the rest of that run (reported with one line) rather than continuing to hit
50
+ the same wall on every following batch.
51
+
52
+ `embedding.concurrency` (positive integer, 1-16) overrides the number of
53
+ embedding requests kept in flight at once, which otherwise defaults to 1 for
54
+ a loopback endpoint and 2 for a remote one. Set it only for an endpoint that
55
+ genuinely serves parallel requests — a local model server started with a
56
+ multi-slot flag (llama.cpp's `--parallel N`, vLLM) — since the default
57
+ already protects an ordinary single-slot server from reload-thrash.
58
+ Embedding throughput is still tuned first by `embedding.batchSize`
59
+ (documents per request) and `embedding.maxTokens` (token
60
+ budget per request); the concurrency override is a second lever for a
61
+ server that can actually use it.
62
+
63
+ `embedding.timeoutMs` bounds each embedding request (default 120s, up from a
64
+ prior fixed 30s that cut off a slow local model server mid-response). It is
65
+ the budget for a request at the full token budget — a smaller request gets a
66
+ proportionally smaller timeout, so a dead endpoint is still detected in
67
+ seconds on the common case of small documents. A request TIMEOUT no longer
68
+ drops its batch immediately: akm now backs off (5s, doubling, capped at 60s)
69
+ and retries the same request once, since field evidence showed the endpoint
70
+ keeps computing an abandoned request regardless of the client giving up; a
71
+ second timeout splits the batch in half and retries each half the same way,
72
+ down to individual documents, and a single document that still times out is
73
+ finally skipped. After 3 consecutive failures at single-document size
74
+ (timeout or network error), or 3 consecutive network errors at any size —
75
+ never a batch rejected only for exceeding the endpoint's context window —
76
+ `akm index`'s embedding phase stops dispatching further requests and reports
77
+ failure instead of grinding through every remaining batch against a dead
78
+ endpoint — batches already committed are kept, and a rerun picks up where it
79
+ left off.
80
+
81
+ `akm bundle update` now durably commits its embedding pass instead of
82
+ nesting it inside its own transaction: earlier releases ran the embedding
83
+ phase inside the same transaction as content/lock/index/state, so every
84
+ per-batch commit landed as an unobservable SAVEPOINT and a SIGKILL mid-run
85
+ lost every embedding of the update, not just the one in flight. The
86
+ embedding phase now runs on its own connection after the update's own
87
+ commit; a failing pass (provider down) still leaves the update itself
88
+ successful, with the new `index.semanticStatus` field on `akm bundle
89
+ update`'s response the only sign semantic search fell behind.
90
+
91
+ The published `akm`/`akm-migrate` launchers now forward SIGTERM/SIGINT/
92
+ SIGHUP to their child and exit alongside it, instead of leaving the child
93
+ running as an orphan when only the launcher is signaled. No action needed —
94
+ this is a drop-in fix for anyone running `akm` under a scheduler,
95
+ supervisor, or hook that can time out or kill the launcher process
96
+ directly.
97
+
98
+ `embedding.maxInputTokens` (default 512) now caps how much of a single
99
+ document's text is sent to the embedding provider, truncating to the head
100
+ instead of ever failing a whole batch over one oversized document.
101
+
102
+ - An existing install's already-stored vectors are untouched and stay
103
+ valid — this only changes what happens for entries embedded *after*
104
+ upgrading.
105
+ - New embeddings (any entry indexed for the first time, or re-indexed after
106
+ a content change) go through the new 512-token cap by default. If you
107
+ were relying on documents longer than ~2000 characters being embedded in
108
+ full, set `embedding.maxInputTokens` higher in `config.json`.
109
+ - `akm index --reembed` re-embeds every entry under the new cap — run it if
110
+ you want your entire existing index rebuilt against the new default (or a
111
+ custom `embedding.maxInputTokens` you've set).
112
+ - `embedding.contextLength` is Ollama's `num_ctx` only now; it no longer
113
+ also sets the per-request token budget (`embedding.maxTokens`). If you had
114
+ set `contextLength` specifically to control request batching (not your
115
+ Ollama server's context window), set `embedding.maxTokens` instead.
116
+
117
+ `akm index --full` and an index-generation bump no longer re-embed
118
+ unchanged content: vectors about to be discarded are salvaged and handed
119
+ back to unchanged entries at the start of the next embedding pass instead
120
+ of every upgrade re-embedding the whole corpus once. No action needed —
121
+ this is automatic; `akm index --reembed` still forces a full re-embed when
122
+ you don't trust the salvaged vectors.
123
+
124
+ A field report suspected `akm index` was sending embedding requests with no
125
+ `Authorization` header despite `embedding.apiKey` being set to a
126
+ `secret://` reference. Auditing every path that builds an embedding request
127
+ found all of them already resolve `secret://` through the same store
128
+ lookup, now pinned by integration and contract tests — this was not a bug
129
+ in the code as it stands. `akm index` now prints one line before its first
130
+ provider request naming the endpoint, model, and credential SOURCE (never
131
+ the value), e.g. `[embed] endpoint http://.../v1/embeddings, model
132
+ nomic-embed; credential: secret://lab-api-key (store)`. If you run an
133
+ embedding gateway that enforces auth and still see unauthenticated requests
134
+ after upgrading, compare this line's endpoint and credential source against
135
+ what the gateway's own request log shows for the same request — a
136
+ mismatch there (not in this line) is the next place to look.
45
137
 
46
138
  A config file can now inherit a shared base via `extends: <path|bundle//path>`,
47
139
  deep-merging under the local file so local keys always win. `akm config diff
@@ -50,3 +142,10 @@ effective config and another config file or bundle-relative file. `akm config
50
142
  unset` now refuses to unset a key whose value comes only from an
51
143
  `extends`-inherited base, naming the source, since there would be nothing local
52
144
  to remove.
145
+
146
+ The scheduler runs the binary path `akm task sync` recorded at sync time, not
147
+ whichever akm your shell now resolves to. After upgrading akm through a
148
+ different installer than the one active at your last `task sync` (e.g.
149
+ npm-global to a standalone download), run `akm task sync` again so the
150
+ schedule points at the new binary; `akm health --probe` now warns via a new
151
+ `scheduler-binary` advisory when the two diverge.
@@ -9,8 +9,9 @@ live one level up in `docs/migration/`.
9
9
 
10
10
  - [0.9.15](0.9.15.md) — exit-code 75 for lease/state.db contention,
11
11
  `--require-engines` scheduled task templates, `--no-probe` cli-version
12
- skip, thinking-control wire forms, embedding re-embed safety, and
13
- `extends` config inheritance
12
+ skip, thinking-control wire forms, embedding re-embed safety and
13
+ cross-rebuild vector salvage, launcher signal forwarding, a default
14
+ per-document embedding cap, and `extends` config inheritance
14
15
  - [0.9.14](0.9.14.md) — index v22-to-v23 derived-cache rebuild, lexical
15
16
  fragments, and collapse-detector canary re-minting
16
17
  - [0.9.2](0.9.2.md) — task source v4 migration, workflow source IR v1 and
@@ -117,7 +117,7 @@ Every command exits with one of the following codes:
117
117
  | 2 | Usage / bad input | `UsageError` |
118
118
  | 4 | Health warning (`akm health` only) | — |
119
119
  | 70 | Internal / unclassified error | unexpected throw |
120
- | 75 | Transient — retry shortly (sysexits `EX_TEMPFAIL`); another akm process holds a lock or is writing `state.db` right now, not a bad command line | `TransientError` |
120
+ | 75 | Transient — retry shortly (sysexits `EX_TEMPFAIL`); another akm process holds a lock or is writing `state.db` or `index.db` right now, not a bad command line | `TransientError` |
121
121
  | 78 | Configuration error | `ConfigError` |
122
122
 
123
123
  Failures classified by akm emit a JSON error envelope on **stderr** before
@@ -219,7 +219,7 @@ Build or refresh the search index.
219
219
 
220
220
  ```sh
221
221
  akm index # Incremental (only changed directories)
222
- akm index --full # Full rebuild
222
+ akm index --full # Full rebuild (reuses unchanged embeddings — see below)
223
223
  akm index --verbose # Print phase progress to stderr
224
224
  akm index --clean # Normal index + remove stale entries from the DB
225
225
  akm index --clean --dry-run # Report stale entries without deleting
@@ -234,6 +234,18 @@ semantic-search settings, and phase-by-phase progress to stderr while the
234
234
  index is being built. Malformed workflow assets are skipped with file-path
235
235
  warnings instead of aborting the full run.
236
236
 
237
+ **Progress in non-verbose JSON mode (default output format, #954):** even
238
+ without `--verbose`, phase-start messages and the embedding heartbeat
239
+ (`Still generating embeddings: X/N stored, F failed; waiting on embedding
240
+ provider.`) are now written to stderr, and a failed embedding batch logs at
241
+ the default level instead of `--verbose`-only — a long-running index build
242
+ against a slow or unresponsive provider is no longer silent until the whole
243
+ run finishes. Text-mode output keeps its spinner instead (no stderr line
244
+ growth); JSON stdout output is unaffected either way. The high-frequency
245
+ per-batch `Embedded N/M entries.` line stays out of non-verbose stderr (it
246
+ fires after every committed batch) — pass `--verbose` for that level of
247
+ detail.
248
+
237
249
  **`--clean` flag:** After indexing completes, verifies every indexed entry's source
238
250
  file still exists on disk. Removes any entries whose file is missing (for local
239
251
  bundle sources only; remote entries are skipped). Returns a `clean` block in the
@@ -242,6 +254,18 @@ Use `--clean` to resolve the edge case where a deleted file in an unchanged
242
254
  directory lingers in the index across incremental runs. With `--dry-run`, reports
243
255
  which entries would be removed without modifying the database.
244
256
 
257
+ **`--full` no longer re-embeds unchanged content (#955):** a full rebuild
258
+ (and an index-generation bump on first open under a new binary) used to
259
+ delete every embedding unconditionally, forcing a full re-embed of the
260
+ whole corpus even when nothing changed. Vectors about to be discarded are
261
+ now salvaged (keyed by a hash of their content plus the fingerprint they
262
+ were generated under) and handed straight back to unchanged entries at the
263
+ start of the next embedding pass, with zero provider calls for them — a
264
+ progress line reports the split (`Reused N embeddings from the previous
265
+ generation; embedding M new.`). Content that changed even by one byte, or
266
+ a fingerprint that no longer matches, still goes through the provider
267
+ normally. `--reembed` is the way to force a full re-embed regardless.
268
+
245
269
  **`--reembed` flag:** Forces a full purge and re-embed of every entry,
246
270
  independent of the embedding-model-rename compatibility check described
247
271
  below. Ordinary indexing already tells a config-only rename of
@@ -255,11 +279,17 @@ opt-in, PID-liveness-only rebuild lock and releases it on exit — this is
255
279
  advisory, never the blocking lock #872 removed (see
256
280
  [Locks](https://github.com/itlackey/akm/blob/main/docs/architecture/internals/indexing.md#locks)). A human-typed
257
281
  `akm index` with no flag is never gated by it: if another run already holds
258
- the lock, it warns and proceeds anyway, contending with the existing run.
282
+ the lock, it warns and proceeds anyway, contending with the existing run. If
283
+ that contention makes index.db genuinely busy (SQLite `database is locked`)
284
+ long enough to exhaust the driver's retry window, the run now fails with
285
+ exit 75 (`TransientError`, code `INDEX_DB_CONTENDED`) instead of the raw
286
+ driver error at exit 70 — the same retry-shortly contract as
287
+ `STATE_DB_CONTENDED`, so a scheduler can branch on it instead of alerting.
259
288
  `--skip-if-locked` changes that only for the invocation that passes it: if
260
289
  the lock is already held by a live process, it skips gracefully (exit 0,
261
- `{ ok: true, skipped: { reason: "lock-held", pid, startedAt } }`) instead of
262
- contending. `akm index` and `akm curate` are both safe to call frequently —
290
+ `{ ok: true, skipped: { reason: "lock-held", pid, launcherPid, startedAt } }`
291
+ — `launcherPid` is the holder's launcher pid when known, `null` otherwise,
292
+ #956) instead of contending. `akm index` and `akm curate` are both safe to call frequently —
263
293
  `curate` never blocks on a rebuild in progress ([read-path indexing stays
264
294
  non-blocking](#curate)) — but a hook, cron job, or scheduled task that
265
295
  invokes `akm index` directly should pass `--skip-if-locked` so it steps
@@ -333,15 +363,17 @@ akm health --report --window-compare 7d --format html
333
363
  | `--window-compare` | Compare the current window against the prior window of the same duration (e.g. `24h`, `7d`). With `--report`, overrides the default trend window. |
334
364
  | `--group-by` | Group rows by `run` (one row per `improve_runs` entry). Omit for the default summary. |
335
365
  | `--windows` | Explicit comparison window(s) as `name=...,since=ISO,until=ISO` (repeatable, up to 4). Mutually exclusive with `--window-compare`. |
336
- | `--no-probe` | Skip the `default-llm-engine` / `configured-engines` reachability probes and the `cli-version` update check (for an offline or air-gapped host). |
366
+ | `--no-probe` | Skip the `default-llm-engine` / `configured-engines` reachability probes, the `cli-version` update check, and the `scheduler-binary` version check (for an offline or air-gapped host). |
337
367
 
338
368
  The command reads `state.db`, verifies that the required tables exist, performs a
339
369
  write-read probe against the events stream, inspects `task_history`, checks the
340
370
  default agent engine, and summarizes recent `improve_*` events. Unless
341
371
  `--no-probe` is given, it also sends a bounded (3s timeout) reachability probe
342
372
  to the `default-llm-engine` and every `configured-engines` LLM connection (and
343
- an SDK engine's LLM fallback), one probe per distinct endpoint, and checks the
344
- installed akm-cli version against the latest GitHub release (`cli-version`).
373
+ an SDK engine's LLM fallback), one probe per distinct endpoint, checks the
374
+ installed akm-cli version against the latest GitHub release (`cli-version`),
375
+ and runs the scheduler's recorded akm binary with `--version` to check it
376
+ against the running CLI (`scheduler-binary`).
345
377
 
346
378
  Primary result fields:
347
379
 
@@ -1214,6 +1246,14 @@ Shipping akm inside your own product (a Docker image, a plugin's own
1214
1246
  `node_modules`)? See [Bundling akm](../integration/bundling-akm.md) for the
1215
1247
  full boot contract, JSON shapes, and exit codes.
1216
1248
 
1249
+ `akm upgrade` replaces the binary in place for its own install method, but a
1250
+ scheduler binding recorded by an earlier `akm task sync` under a *different*
1251
+ install method is not repointed automatically — the scheduler runs the
1252
+ binary path recorded at sync time, not whichever akm `upgrade` just
1253
+ installed. Run `akm task sync` after switching installers so scheduled runs
1254
+ pick up the new binary; see [`task sync`](#task) and `akm health`'s
1255
+ `scheduler-binary` advisory.
1256
+
1217
1257
  ### clone
1218
1258
 
1219
1259
  Copy an asset from any source into a managed writable bundle or an unmanaged
@@ -2847,6 +2887,13 @@ task template (both the core set and the improve-schedule set) and asks once
2847
2887
  before changing task files or scheduler state; non-interactive setup changes
2848
2888
  neither.
2849
2889
 
2890
+ Because the scheduler runs the exact binary path recorded at the last `task
2891
+ sync`, upgrading akm through a different installer than the one active at
2892
+ that sync (npm-global to a standalone download, or vice versa) leaves
2893
+ scheduled runs invoking the old, now-stale binary — `task sync` re-resolves
2894
+ the current path and repoints them. `akm health --probe`'s `scheduler-binary`
2895
+ advisory warns when the two diverge, naming both versions.
2896
+
2850
2897
  Setup reconfiguration preserves existing scheduler runtime bindings. Changing
2851
2898
  the AKM storage path or installed runtime path therefore requires an explicit
2852
2899
  `akm task sync --rebind`; setup does not silently migrate those entries.