akm-cli 0.9.14 → 0.9.15-beta.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (110) hide show
  1. package/CHANGELOG.md +397 -0
  2. package/STABILITY.md +6 -3
  3. package/dist/assets/prompts/reflect-feedback-framing.md +1 -0
  4. package/dist/assets/prompts/reflect-llm-framed-contract.md +2 -0
  5. package/dist/assets/prompts/reflect-llm-schema-contract.md +2 -0
  6. package/dist/assets/tasks/core/improve.yml +1 -1
  7. package/dist/assets/tasks/core/index-refresh.yml +1 -1
  8. package/dist/assets/tasks/improve/akm-graph-refresh-weekly.yml +1 -1
  9. package/dist/assets/tasks/improve/akm-improve-catchup.yml +1 -1
  10. package/dist/assets/tasks/improve/akm-improve-consolidate.yml +1 -1
  11. package/dist/assets/tasks/improve/akm-improve-frequent.yml +1 -1
  12. package/dist/assets/tasks/improve/akm-improve-nightly.yml +1 -1
  13. package/dist/cli/retired-commands.js +0 -1
  14. package/dist/cli/shared.js +9 -0
  15. package/dist/cli/unknown-flags.js +1 -0
  16. package/dist/cli.js +3 -2
  17. package/dist/commands/config-cli.js +85 -3
  18. package/dist/commands/env/env-cli.js +1 -42
  19. package/dist/commands/env/env.js +1 -1
  20. package/dist/commands/env/secret-cli.js +1 -2
  21. package/dist/commands/health/checks.js +357 -63
  22. package/dist/commands/health/engine-usage.js +45 -0
  23. package/dist/commands/health/improve-metrics.js +18 -0
  24. package/dist/commands/health/llm-usage.js +41 -1
  25. package/dist/commands/health/plugin-staleness.js +7 -3
  26. package/dist/commands/health/version-drift.js +93 -0
  27. package/dist/commands/health/windows.js +3 -1
  28. package/dist/commands/health.js +44 -9
  29. package/dist/commands/improve/consolidate/chunking.js +4 -2
  30. package/dist/commands/improve/improve-cli.js +99 -5
  31. package/dist/commands/improve/improve-report.js +154 -0
  32. package/dist/commands/improve/improve-result-file.js +45 -33
  33. package/dist/commands/improve/improve-strategies.js +133 -3
  34. package/dist/commands/improve/improve-usage-report.js +182 -0
  35. package/dist/commands/improve/improve.js +40 -3
  36. package/dist/commands/improve/locks.js +27 -78
  37. package/dist/commands/improve/planner.js +1 -0
  38. package/dist/commands/improve/preparation.js +9 -1
  39. package/dist/commands/improve/reflect.js +44 -4
  40. package/dist/commands/models-cli.js +50 -1
  41. package/dist/commands/proposal/repository.js +8 -3
  42. package/dist/commands/proposal/validators/proposal-quality-validators.js +41 -6
  43. package/dist/commands/proposal/validators/proposal-validators.js +24 -0
  44. package/dist/commands/read/search-cli.js +38 -2
  45. package/dist/commands/read/show.js +103 -4
  46. package/dist/commands/sources/info.js +5 -1
  47. package/dist/commands/sources/self-update.js +2 -2
  48. package/dist/commands/sources/stash-cli.js +31 -0
  49. package/dist/commands/tasks/tasks-cli.js +49 -2
  50. package/dist/commands/workflow-cli.js +86 -12
  51. package/dist/core/asset/markdown-fragments.js +35 -0
  52. package/dist/core/config/config-schema.js +14 -0
  53. package/dist/core/config/config.js +302 -24
  54. package/dist/core/env-secret-ref.js +58 -5
  55. package/dist/core/errors.js +30 -0
  56. package/dist/core/improve-result.js +51 -0
  57. package/dist/core/loopback.js +17 -0
  58. package/dist/core/paths.js +11 -0
  59. package/dist/core/run-lock.js +96 -0
  60. package/dist/core/sensitive-marker-path.js +19 -0
  61. package/dist/core/state-db.js +74 -14
  62. package/dist/indexer/index-rebuild-lock.js +73 -0
  63. package/dist/indexer/index-writer-lock.js +40 -1
  64. package/dist/indexer/index-written-assets.js +21 -1
  65. package/dist/indexer/indexer.js +18 -17
  66. package/dist/indexer/materialize-embeddings.js +282 -32
  67. package/dist/indexer/search/db-search.js +49 -2
  68. package/dist/integrations/agent/engine-resolution.js +96 -6
  69. package/dist/integrations/agent/execution-definitions.js +6 -15
  70. package/dist/integrations/agent/execution-lowering.js +6 -1
  71. package/dist/integrations/agent/execution-preparation.js +1 -1
  72. package/dist/integrations/agent/model-map.js +123 -20
  73. package/dist/integrations/agent/prompts.js +40 -8
  74. package/dist/integrations/agent/runner-dispatch.js +9 -3
  75. package/dist/integrations/agent/runner.js +2 -0
  76. package/dist/llm/client.js +8 -3
  77. package/dist/llm/embedder.js +20 -8
  78. package/dist/llm/embedders/local.js +10 -2
  79. package/dist/llm/embedders/remote.js +188 -21
  80. package/dist/output/shapes/helpers.js +38 -2
  81. package/dist/output/shapes/models-list.js +16 -0
  82. package/dist/output/shapes/passthrough.js +2 -0
  83. package/dist/output/shapes.js +4 -0
  84. package/dist/output/text/command-format.js +29 -0
  85. package/dist/output/text/helpers.js +1 -1
  86. package/dist/output/text/improve-report.js +27 -0
  87. package/dist/{commands/env/marker-path.js → output/text/models.js} +4 -3
  88. package/dist/output/text/show-format.js +4 -0
  89. package/dist/output/text.js +4 -0
  90. package/dist/scripts/akm-migrate-node.js +24798 -21732
  91. package/dist/scripts/akm-migrate.js +23408 -20343
  92. package/dist/storage/repositories/improve-runs-repository.js +34 -0
  93. package/dist/storage/repositories/index-fts-repository.js +49 -6
  94. package/dist/storage/repositories/index-vec-repository.js +30 -0
  95. package/dist/storage/repositories/workflow-runs-repository.js +55 -18
  96. package/dist/tasks/backends/cron.js +14 -7
  97. package/dist/tasks/run/run-workflow-task.js +16 -0
  98. package/dist/workflows/exec/child-workflow.js +2 -2
  99. package/dist/workflows/exec/dispatch-redaction.js +21 -9
  100. package/dist/workflows/exec/run-workflow.js +6 -5
  101. package/dist/workflows/runtime/runs.js +33 -5
  102. package/docs/migration/release-notes/0.9.15.md +52 -0
  103. package/docs/migration/release-notes/README.md +4 -0
  104. package/docs/reference/cli.md +245 -29
  105. package/docs/reference/configuration.md +180 -19
  106. package/docs/reference/data-and-telemetry.md +8 -0
  107. package/docs/reference/tasks.md +16 -1
  108. package/docs/reference/workflow-schema.md +5 -1
  109. package/package.json +1 -1
  110. package/schemas/akm-config.json +8 -0
@@ -102,6 +102,40 @@ export function queryImproveRuns(db, since, until) {
102
102
  : "SELECT id, started_at, completed_at, ok, scope_mode, scope_value, strategy, result_json FROM improve_runs WHERE started_at >= ? AND dry_run = 0 ORDER BY started_at DESC";
103
103
  return (until ? db.prepare(sql).all(since, until) : db.prepare(sql).all(since));
104
104
  }
105
+ const IMPROVE_RUN_ROW_COLUMNS = "id, started_at, completed_at, stash_dir, dry_run, strategy, scope_mode, scope_value, guidance, ok, result_json, metrics_json, metadata_json";
106
+ /**
107
+ * Look up a single `improve_runs` row by its id (#944 — `akm improve report --run <id>`).
108
+ * `id` is the PRIMARY KEY, so at most one row matches. Returns `undefined`
109
+ * for an unknown id rather than throwing — the caller decides how to report
110
+ * "no such run".
111
+ */
112
+ export function getImproveRunById(db, id) {
113
+ return db.prepare(`SELECT ${IMPROVE_RUN_ROW_COLUMNS} FROM improve_runs WHERE id = ?`).get(id);
114
+ }
115
+ /**
116
+ * Look up the most recent real (non-dry-run) `improve_runs` row (#944 —
117
+ * `akm improve report`'s default target, and `--last`). Same `dry_run = 0`
118
+ * filter as {@link queryImproveRuns} so a dry-run preview never becomes the
119
+ * implicit "last run" a report reads.
120
+ */
121
+ export function getLatestImproveRun(db) {
122
+ return db
123
+ .prepare(`SELECT ${IMPROVE_RUN_ROW_COLUMNS} FROM improve_runs WHERE dry_run = 0 ORDER BY started_at DESC LIMIT 1`)
124
+ .get();
125
+ }
126
+ /**
127
+ * #950: count real (non-dry-run) improve_runs rows whose `started_at` falls
128
+ * in `[since, now)` — same filter as {@link queryImproveRuns}, but a `SELECT
129
+ * COUNT(*)` for a caller (the `engine-last-used` gate) that only needs to
130
+ * know whether any run happened, not the rows themselves (which would pull
131
+ * every `result_json` blob in the window just to test for zero).
132
+ */
133
+ export function countImproveRunsSince(db, since) {
134
+ const row = db
135
+ .prepare("SELECT COUNT(*) AS cnt FROM improve_runs WHERE started_at >= ? AND dry_run = 0")
136
+ .get(since);
137
+ return row.cnt;
138
+ }
105
139
  /**
106
140
  * Delete improve_runs rows older than `retentionDays` (default: 90). Mirrors
107
141
  * {@link purgeOldEvents} — same default, same return shape (number of rows
@@ -7,7 +7,7 @@
7
7
  * Owns the `entries_fts` full-text query path, per-entry projections, and the
8
8
  * explicit full recovery rebuild.
9
9
  */
10
- import { fragmentForSelector, splitMarkdownFragments } from "../../core/asset/markdown-fragments.js";
10
+ import { splitMarkdownFragments } from "../../core/asset/markdown-fragments.js";
11
11
  import { stableFtsScore } from "../../core/lexical-score.js";
12
12
  import { warn } from "../../core/warn.js";
13
13
  import { buildLexicalQueryPlan } from "../../indexer/search/fts-query.js";
@@ -87,11 +87,54 @@ export function searchFts(db, query, limit, entryType, excludeTypes) {
87
87
  * file edit; the next index refresh atomically publishes the new revision.
88
88
  */
89
89
  export function getIndexedMarkdownFragment(db, itemRef, fragmentId) {
90
- const row = db
91
- .prepare("SELECT s.safe_markdown FROM entry_fragments_fts f JOIN entry_fragments s ON s.entry_id = f.entry_id JOIN entries e ON e.id = f.entry_id WHERE e.item_ref = ? AND f.fragment_id = ?")
92
- .get(itemRef, fragmentId);
93
- const fragment = row ? fragmentForSelector(row.safe_markdown, fragmentId) : undefined;
94
- return fragment ? { content: fragment.text } : undefined;
90
+ return getIndexedMarkdownFragments(db, [{ itemRef, fragmentId }])[0];
91
+ }
92
+ /**
93
+ * Batch the selected-hit projection read. Search commonly enriches several
94
+ * fragment hits at once; reading all indexed-safe parents in chunks avoids an
95
+ * N-query loop, while grouping selectors by parent ensures each safe revision
96
+ * is split at most once.
97
+ */
98
+ export function getIndexedMarkdownFragments(db, selections) {
99
+ if (selections.length === 0)
100
+ return [];
101
+ const itemRefs = [...new Set(selections.map((selection) => selection.itemRef))];
102
+ const sourceByRef = new Map();
103
+ for (let offset = 0; offset < itemRefs.length; offset += SQLITE_CHUNK_SIZE) {
104
+ const chunk = itemRefs.slice(offset, offset + SQLITE_CHUNK_SIZE);
105
+ const placeholders = chunk.map(() => "?").join(",");
106
+ const rows = db
107
+ .prepare(`SELECT e.item_ref, s.safe_markdown FROM entry_fragments s JOIN entries e ON e.id = s.entry_id WHERE e.item_ref IN (${placeholders})`)
108
+ .all(...chunk);
109
+ for (const row of rows)
110
+ sourceByRef.set(row.item_ref, row.safe_markdown);
111
+ }
112
+ const fragmentsByRef = new Map();
113
+ for (const [itemRef, safeMarkdown] of sourceByRef) {
114
+ fragmentsByRef.set(itemRef, splitMarkdownFragments(safeMarkdown));
115
+ }
116
+ return selections.map((selection) => {
117
+ const safeMarkdown = sourceByRef.get(selection.itemRef);
118
+ const fragments = fragmentsByRef.get(selection.itemRef);
119
+ if (safeMarkdown === undefined || !fragments)
120
+ return undefined;
121
+ const fragment = fragments.find((candidate) => candidate.fragmentId === selection.fragmentId || candidate.headingSlug === selection.fragmentId);
122
+ return fragment ? materializeIndexedMarkdownFragment(fragment, fragments, safeMarkdown.length) : undefined;
123
+ });
124
+ }
125
+ function materializeIndexedMarkdownFragment(fragment, fragments, parentChars) {
126
+ return {
127
+ content: fragment.text,
128
+ ordinal: fragment.ordinal,
129
+ count: fragments.length,
130
+ startLine: fragment.startLine,
131
+ endLine: fragment.endLine,
132
+ previousFragmentId: fragments[fragment.ordinal - 1]?.fragmentId,
133
+ nextFragmentId: fragments[fragment.ordinal + 1]?.fragmentId,
134
+ fragmentChars: fragment.text.length,
135
+ parentChars,
136
+ fragments,
137
+ };
95
138
  }
96
139
  function runFtsQuery(db, ftsQuery, lexicalMatch, limit, entryType, excludeTypes) {
97
140
  // Preserve the repository's ordinary limit contract for direct callers.
@@ -333,3 +333,33 @@ export function getEmbeddingCount(db) {
333
333
  const row = db.prepare("SELECT COUNT(*) AS cnt FROM embeddings").get();
334
334
  return row.cnt;
335
335
  }
336
+ /**
337
+ * Sample up to `limit` already-embedded entries (id, search text, and the
338
+ * stored vector) for the embedding-fingerprint canary check: re-embedding
339
+ * these texts with the CURRENT config and comparing against `vector` is how
340
+ * a model-string rename is told apart from a genuine model/dimension change
341
+ * (#955), without trusting the config string alone.
342
+ *
343
+ * Ordered by `id` for a deterministic, cheap sample (no `ORDER BY RANDOM()`)
344
+ * — the canary only needs "some" already-verified vectors, not a
345
+ * statistically representative one. A corrupt stored BLOB (see
346
+ * `bufferToFloat32`) is skipped rather than failing the whole sample.
347
+ */
348
+ export function sampleEmbeddedEntriesForCanary(db, limit) {
349
+ const rows = db
350
+ .prepare(`
351
+ SELECT e.id, e.search_text AS searchText, em.embedding AS embedding
352
+ FROM entries e
353
+ JOIN embeddings em ON em.id = e.id
354
+ ORDER BY e.id
355
+ LIMIT ?
356
+ `)
357
+ .all(limit);
358
+ const samples = [];
359
+ for (const row of rows) {
360
+ const vector = bufferToFloat32(row.embedding, Math.floor(row.embedding.byteLength / 4));
361
+ if (vector)
362
+ samples.push({ id: row.id, searchText: row.searchText, vector });
363
+ }
364
+ return samples;
365
+ }
@@ -2,8 +2,8 @@
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  import { randomUUID } from "node:crypto";
5
- import { NotFoundError, UsageError } from "../../core/errors.js";
6
- import { openStateDatabase, withImmediateTransaction } from "../../core/state-db.js";
5
+ import { NotFoundError, TransientError, UsageError } from "../../core/errors.js";
6
+ import { isSqliteContentionError, openStateDatabase, withImmediateTransaction } from "../../core/state-db.js";
7
7
  import { borrowScopedStateDb, withStateDbScope } from "../../core/state-db-scope.js";
8
8
  import { sleepSync } from "../../runtime.js";
9
9
  import { escapeLikePattern } from "../like-pattern.js";
@@ -25,22 +25,21 @@ function assertAttemptReservationLease(input, run) {
25
25
  /**
26
26
  * Whether `error` is one of the specific SQLite conditions a run-lease
27
27
  * statement can throw under real cross-process contention on the same row:
28
- * `SQLITE_BUSY`/`SQLITE_LOCKED` (both drivers), or the message text a
29
- * transient contention blip has been observed producing, "database is
30
- * locked", "disk I/O error", or "database disk image is malformed".
31
- * Matching on this set alone is never sufficient to call something lease
32
- * contention — see {@link WorkflowRunsRepository.acquireEngineLease}, which
33
- * additionally requires a fresh read confirming a live lease before
34
- * substituting the lease-held message for the original error.
28
+ * the shared {@link isSqliteContentionError} classifier (SQLITE_BUSY/LOCKED,
29
+ * "database is locked", "database table is locked", the phantom-BEGIN
30
+ * marker) plus two corruption-shaped message texts a transient contention
31
+ * blip has been observed producing on this specific race, "disk I/O error"
32
+ * and "database disk image is malformed". Matching on this set alone is
33
+ * never sufficient to call something lease contention — see
34
+ * {@link WorkflowRunsRepository.acquireEngineLease}, which additionally
35
+ * requires a fresh read confirming a live lease before substituting the
36
+ * lease-held message for the original error.
35
37
  */
36
38
  function isLeaseContentionSqliteError(error) {
37
- const code = error?.code;
38
- if (code === "SQLITE_BUSY" || code === "SQLITE_LOCKED")
39
+ if (isSqliteContentionError(error))
39
40
  return true;
40
41
  const message = error instanceof Error ? error.message : String(error);
41
- return (message.includes("database is locked") ||
42
- message.includes("disk I/O error") ||
43
- message.includes("database disk image is malformed"));
42
+ return message.includes("disk I/O error") || message.includes("database disk image is malformed");
44
43
  }
45
44
  const LEASE_RETRY_ATTEMPTS = 4;
46
45
  const LEASE_RETRY_BASE_DELAY_MS = 15;
@@ -110,6 +109,9 @@ export class WorkflowRunsRepository {
110
109
  * operator to `akm workflow abandon` a child a parent is actively driving.
111
110
  * For any database with no child rows the result is byte-identical, same
112
111
  * as the other three B-N10 sites below.
112
+ *
113
+ * Always scope-local: `scopeKey` is a real scope, never "every scope" — see
114
+ * {@link findActiveRunOutsideScope} for the cross-scope warning's query.
113
115
  */
114
116
  findActiveRunForScope(workflowRefs, scopeKey) {
115
117
  const refs = typeof workflowRefs === "string" ? [workflowRefs] : [...workflowRefs];
@@ -119,6 +121,26 @@ export class WorkflowRunsRepository {
119
121
  .prepare(`SELECT id, current_step_id FROM workflow_runs WHERE workflow_ref IN (${refs.map(() => "?").join(", ")}) AND scope_key = ? AND status = 'active' AND parent_run_id IS NULL ORDER BY updated_at DESC, created_at DESC LIMIT 1`)
120
122
  .get(...refs, scopeKey);
121
123
  }
124
+ /**
125
+ * The cross-scope start warning's query (#942): the most recently active
126
+ * run of these refs OUTSIDE `scopeKey` — i.e. `scope_key IS NULL OR
127
+ * scope_key != scopeKey`, not merely "the most recent active run of these
128
+ * refs anywhere". A same-scope active row must never win the `LIMIT 1` and
129
+ * mask a DIFFERENT scope's run: with `--new`/`--force`, the caller's own
130
+ * active run could otherwise sort first (most recently updated) and hide a
131
+ * third scope's run entirely, so `startWorkflowRun` silently warned about
132
+ * nothing while a genuinely stale run in another scope went unreported.
133
+ * `findActiveRunForScope` deliberately stays scope-local and is never used
134
+ * for this purpose.
135
+ */
136
+ findActiveRunOutsideScope(workflowRefs, scopeKey) {
137
+ const refs = typeof workflowRefs === "string" ? [workflowRefs] : [...workflowRefs];
138
+ if (refs.length === 0)
139
+ return undefined;
140
+ return (this.db
141
+ .prepare(`SELECT * FROM workflow_runs WHERE workflow_ref IN (${refs.map(() => "?").join(", ")}) AND (scope_key IS NULL OR scope_key != ?) AND status = 'active' AND parent_run_id IS NULL ORDER BY updated_at DESC, created_at DESC LIMIT 1`)
142
+ .get(...refs, scopeKey) ?? undefined);
143
+ }
122
144
  getRunById(runId) {
123
145
  return (this.db.prepare("SELECT * FROM workflow_runs WHERE id = ?").get(runId) ??
124
146
  undefined);
@@ -130,6 +152,9 @@ export class WorkflowRunsRepository {
130
152
  * to directly through this path — a parent-driven child would then have
131
153
  * TWO drivers. For any database with no child rows the result is
132
154
  * byte-identical.
155
+ *
156
+ * Always scope-local: `scopeKey` is a real scope, never "every scope" — see
157
+ * {@link findActiveRunOutsideScope} for the cross-scope warning's query.
133
158
  */
134
159
  getActiveRunRowForScope(workflowRefs, scopeKey) {
135
160
  const refs = typeof workflowRefs === "string" ? [workflowRefs] : [...workflowRefs];
@@ -159,8 +184,13 @@ export class WorkflowRunsRepository {
159
184
  listRuns(filter) {
160
185
  const filters = [];
161
186
  const params = [];
162
- filters.push("scope_key = ?");
163
- params.push(filter.scopeKey);
187
+ // `scopeKey: null` (#942) is "every scope" — the predicate is omitted
188
+ // rather than bound as SQL NULL (which would match nothing, since a real
189
+ // `scope_key` column value is never NULL for a fresh run).
190
+ if (filter.scopeKey !== null) {
191
+ filters.push("scope_key = ?");
192
+ params.push(filter.scopeKey);
193
+ }
164
194
  if (filter.workflowRef) {
165
195
  filters.push("workflow_ref = ?");
166
196
  params.push(filter.workflowRef);
@@ -282,10 +312,17 @@ export class WorkflowRunsRepository {
282
312
  this.immediateTransaction((db) => {
283
313
  input.revalidateSources();
284
314
  if (!input.force) {
315
+ // The uniqueness guard must never silently skip its scope predicate
316
+ // (#942) — every real caller (`startWorkflowRun`) stamps a concrete
317
+ // scope key, so a null one here means the caller is misusing this
318
+ // top-level guard, not that scoping should be waived.
319
+ if (input.run.scopeKey === null) {
320
+ throw new Error("publishWorkflowRunV4: run.scopeKey must not be null for the scope uniqueness guard.");
321
+ }
285
322
  const existing = this.findActiveRunForScope(input.workflowRefs, input.run.scopeKey);
286
323
  if (existing) {
287
324
  throw new UsageError(`Workflow ${input.run.workflowRef} already has an active run in this scope ` +
288
- `(id=${existing.id}, step=${existing.current_step_id ?? "—"}). ` +
325
+ `(id=${existing.id}, scope=${input.run.scopeKey}, step=${existing.current_step_id ?? "—"}). ` +
289
326
  `Use 'akm workflow run ${input.run.workflowRef}' to resume it or ` +
290
327
  `'akm workflow abandon ${existing.id}' to give up on it.`, "RESOURCE_ALREADY_EXISTS");
291
328
  }
@@ -443,7 +480,7 @@ export class WorkflowRunsRepository {
443
480
  throw error;
444
481
  const row = this.tryReadLeaseColumns(runId);
445
482
  if (row?.engine_lease_holder && row.engine_lease_until && row.engine_lease_until >= now) {
446
- throw new UsageError(`Workflow run ${runId} is already being driven by engine ${row.engine_lease_holder} ` +
483
+ throw new TransientError(`Workflow run ${runId} is already being driven by engine ${row.engine_lease_holder} ` +
447
484
  `(run lease expires ${row.engine_lease_until}). A second \`akm workflow run\` would race it — ` +
448
485
  `wait for that invocation to finish or for the lease to expire.`, "RUN_LEASE_HELD");
449
486
  }
@@ -7,7 +7,7 @@
7
7
  // its other lines untouched:
8
8
  //
9
9
  // # akm:task <id> BEGIN
10
- // [SCHED] /abs/akm task run <id> >> /home/.../tasks/logs/<id>.log 2>&1
10
+ // [SCHED] /abs/akm task run <id> > /home/.../tasks/logs/<id>.log 2>&1
11
11
  // # akm:task <id> END
12
12
  //
13
13
  // The backend reads/writes the user's crontab via `crontab -l` and
@@ -58,9 +58,9 @@ export function CRON_BACKEND(options = {}) {
58
58
  install(task, opts, expected) {
59
59
  if (expected)
60
60
  assertSchedulerExpectationIdentity(expected, task);
61
- // Create the log directory before writing the crontab line — cron
62
- // appends with `>>` and the surrounding shell will fail the entire
63
- // entry if the parent directory doesn't exist.
61
+ // Create the log directory before writing the crontab line — the
62
+ // redirect target's parent directory must exist or the surrounding
63
+ // shell will fail the entire entry.
64
64
  const cronLineParts = buildCronLineParts(task, [...(opts?.binding ?? akmArgv)], logDir, opts?.contextPath ?? defaultContextPath, opts?.target);
65
65
  const cronLine = cronLineParts.line;
66
66
  assertPortableCronLine(cronLine);
@@ -260,14 +260,17 @@ function buildCronLineParts(task, akmArgv, logDir, contextPath, _target) {
260
260
  const logPath = path.join(logDir, `${nativeId}.log`);
261
261
  const invocation = buildScheduledBindingInvocation(akmArgv, contextPath, task.invocation);
262
262
  const cmd = invocation.argv.map((part) => quoteForCron(part)).join(" ");
263
- const directLine = `${cronExpr} ${cmd} >> ${quoteForCron(logPath)} 2>&1`;
263
+ // #951: truncate (not append) so this bootstrap safety-net file always
264
+ // holds exactly the latest run's raw output rather than growing forever.
265
+ // akm's own per-run log (src/tasks/run/task-log.ts) already keeps history.
266
+ const directLine = `${cronExpr} ${cmd} > ${quoteForCron(logPath)} 2>&1`;
264
267
  if (Buffer.byteLength(directLine, "utf8") <= PORTABLE_CRON_LINE_LIMIT) {
265
268
  return { line: directLine };
266
269
  }
267
270
  const content = cronWrapperScriptContent(invocation.argv);
268
271
  const contentHash = createHash("sha256").update(content).digest("hex").slice(0, 16);
269
272
  const wrapperPath = path.join(logDir, `${CRON_WRAPPER_PREFIX}${nativeId}-${contentHash}.sh`);
270
- const line = `${cronExpr} sh ${quoteForCron(wrapperPath)} >> ${quoteForCron(logPath)} 2>&1`;
273
+ const line = `${cronExpr} sh ${quoteForCron(wrapperPath)} > ${quoteForCron(logPath)} 2>&1`;
271
274
  return { line, wrapper: { path: wrapperPath, content } };
272
275
  }
273
276
  const CRON_WRAPPER_PREFIX = ".akm-cron-wrapper-";
@@ -360,7 +363,11 @@ export function extractCronInvocation(body) {
360
363
  let commandStart = 5;
361
364
  while (commandStart < fields.length && /^[A-Za-z_][A-Za-z0-9_]*=/.test(fields[commandStart]))
362
365
  commandStart += 1;
363
- const redirectIndex = fields.indexOf(">>", commandStart);
366
+ // #951: newly-installed rows redirect with `>` (truncate); tolerate `>>`
367
+ // (append) too, since that's what every row written before this change —
368
+ // including ones a still-running older akm binary installs during a
369
+ // rolling upgrade — looks like on disk.
370
+ const redirectIndex = fields.findIndex((field, index) => index >= commandStart && (field === ">" || field === ">>"));
364
371
  if (redirectIndex === -1)
365
372
  return undefined;
366
373
  const tail = fields.slice(commandStart, redirectIndex);
@@ -187,6 +187,22 @@ export async function runWorkflowTask(input) {
187
187
  * here is a *compile* error rather than silently collapsing to "completed".
188
188
  * The previous silent `default: "completed"` is preserved only for the
189
189
  * `undefined` (no-detail) case, which is handled up front.
190
+ *
191
+ * #943: is `undefined` reachable from a timeout (i.e. can a wedged/killed run
192
+ * produce `status: "completed"`)? No. This function is only ever called as
193
+ * `mapWorkflowStatus(detail?.status)` behind `failure ? "failed" : ...` in
194
+ * {@link runWorkflowTask} — `failure` is `error ?? gateError ?? timeoutError`,
195
+ * and `detail` is left `undefined` only in the `catch` branch that also sets
196
+ * `error` (making `failure` truthy). So whenever this function runs, either
197
+ * `failure` already short-circuited the caller to `"failed"` without
198
+ * consulting it, or the run awaited successfully and `detail` (thus
199
+ * `detail.status`) is guaranteed set by `RunWorkflowResult["run"]`, a
200
+ * required field. `status === undefined` is therefore unreachable today; it
201
+ * stays mapped to `"completed"` defensively rather than thrown on, since a
202
+ * literal `undefined` can only mean "the runtime told us nothing went
203
+ * wrong" — never observed from `runWorkflowStepsImpl`, but if some future
204
+ * caller ever passes it, failing loudly on "no evidence of failure" would be
205
+ * a worse trade than the status quo.
190
206
  */
191
207
  function mapWorkflowStatus(status) {
192
208
  // No run detail → treat as completed (unchanged from the prior silent default).
@@ -53,7 +53,7 @@
53
53
  * `driveRun` itself is never exported (B-N5's "no second executor" holds).
54
54
  */
55
55
  import { randomUUID } from "node:crypto";
56
- import { UsageError } from "../../core/errors.js";
56
+ import { TransientError } from "../../core/errors.js";
57
57
  import { withWorkflowRunsRepo } from "../../storage/repositories/workflow-runs-repository.js";
58
58
  import { validateWorkflowParams } from "../ir/params.js";
59
59
  import { canonicalPlanJson, computePlanHash } from "../ir/plan-hash.js";
@@ -76,7 +76,7 @@ function errorMessage(err) {
76
76
  }
77
77
  /** `acquireRunLease`'s exact refusal shape (run-workflow.ts) — matched by text, since this module cannot import that private helper. */
78
78
  function isLeaseBusyError(err) {
79
- return err instanceof UsageError && err.message.includes("is already being driven by engine");
79
+ return err instanceof TransientError && err.message.includes("is already being driven by engine");
80
80
  }
81
81
  /** §3.4's exact `child_workflow_failed` message. */
82
82
  function childWorkflowFailedMessage(input) {
@@ -15,23 +15,24 @@
15
15
  * @module workflows/exec/dispatch-redaction
16
16
  */
17
17
  import { collectSensitiveValues, isEnvPassthroughValueSafeToExpose, redactSensitiveText, redactSensitiveValue, } from "../../core/redaction.js";
18
- import { lookupApiKeyFileValue } from "../../integrations/agent/engine-resolution.js";
18
+ import { lookupApiKeyFileValue, lookupApiKeySecretRefValue } from "../../integrations/agent/engine-resolution.js";
19
19
  /**
20
20
  * Every exact value that must never survive into the journal from ONE frozen
21
21
  * dispatch: the resolved `env` bindings injected into the child, the selected
22
- * engine's (and its SDK fallback's) credential — an env value, or a
23
- * file-backed one (#905) — and any `envPassthrough` value the redaction
24
- * policy does not consider safe to expose.
22
+ * engine's (and its SDK fallback's) credential — an env value, a file-backed
23
+ * one (#905), or a secret-store-backed one (#953) — and any `envPassthrough`
24
+ * value the redaction policy does not consider safe to expose.
25
25
  *
26
26
  * Shared by the unit path and the gate-judge path. There is deliberately ONE
27
27
  * collector: a second, parallel implementation is exactly how a dispatch path
28
28
  * silently loses the scrub.
29
29
  *
30
- * The credential values are read from `process.env` (or disk, for a file-
31
- * backed one) AT CALL TIME, so a caller must collect no earlier than the
32
- * dispatch whose outcome it scrubs. A snapshot taken when the dispatch was
33
- * merely *planned* can predate a credential the dispatch then resolves live,
34
- * leaving the exact value it must remove out of the set.
30
+ * The credential values are read from `process.env` (or disk / the secret
31
+ * store, for a file- or store-backed one) AT CALL TIME, so a caller must
32
+ * collect no earlier than the dispatch whose outcome it scrubs. A snapshot
33
+ * taken when the dispatch was merely *planned* can predate a credential the
34
+ * dispatch then resolves live, leaving the exact value it must remove out of
35
+ * the set.
35
36
  */
36
37
  export function collectWorkflowDispatchSensitiveValues(dispatch, env) {
37
38
  const values = new Set([...Object.values(env ?? {}), ...(dispatch.sensitiveValues ?? [])]);
@@ -51,6 +52,12 @@ export function collectWorkflowDispatchSensitiveValues(dispatch, env) {
51
52
  if (value)
52
53
  values.add(value);
53
54
  }
55
+ // #953: secret-store-backed credential — same best-effort rationale.
56
+ if (runner.apiKeySecretRef) {
57
+ const value = lookupApiKeySecretRefValue(runner.apiKeySecretRef);
58
+ if (value)
59
+ values.add(value);
60
+ }
54
61
  return;
55
62
  }
56
63
  for (const name of runner.profile.envPassthrough ?? []) {
@@ -69,6 +76,11 @@ export function collectWorkflowDispatchSensitiveValues(dispatch, env) {
69
76
  if (value)
70
77
  values.add(value);
71
78
  }
79
+ if (runner.fallbackApiKeySecretRef) {
80
+ const value = lookupApiKeySecretRefValue(runner.fallbackApiKeySecretRef);
81
+ if (value)
82
+ values.add(value);
83
+ }
72
84
  }
73
85
  };
74
86
  addCredential(dispatch.runner);
@@ -15,7 +15,7 @@
15
15
  * full design history behind each of these invariants.
16
16
  */
17
17
  import { randomUUID } from "node:crypto";
18
- import { UsageError } from "../../core/errors.js";
18
+ import { TransientError, UsageError } from "../../core/errors.js";
19
19
  import { withMaintenanceStartBarrierAsync } from "../../core/maintenance-barrier.js";
20
20
  import { disposeDispatchResources } from "../../integrations/agent/runner-dispatch.js";
21
21
  import { withWorkflowRunsConnection, withWorkflowRunsRepo } from "../../storage/repositories/workflow-runs-repository.js";
@@ -183,16 +183,17 @@ function leaseExpiry() {
183
183
  return new Date(Date.now() + RUN_LEASE_TTL_MS).toISOString();
184
184
  }
185
185
  /**
186
- * Atomically claim the run lease or refuse with a UsageError naming the
187
- * current holder + expiry. The single-UPDATE claim in the repository is the
188
- * arbiter — two racing invocations cannot both win.
186
+ * Atomically claim the run lease or refuse with a TransientError naming the
187
+ * current holder + expiry (#948 addendum — moved off UsageError, exit 75).
188
+ * The single-UPDATE claim in the repository is the arbiter — two racing
189
+ * invocations cannot both win.
189
190
  */
190
191
  async function acquireRunLease(runId, holder) {
191
192
  await withMaintenanceStartBarrierAsync(() => withWorkflowRunsRepo((repo) => {
192
193
  if (repo.acquireEngineLease(runId, holder, leaseExpiry(), new Date().toISOString()))
193
194
  return;
194
195
  const row = repo.getRunById(runId);
195
- throw new UsageError(`Workflow run ${runId} is already being driven by engine ${row?.engine_lease_holder ?? "(unknown)"} ` +
196
+ throw new TransientError(`Workflow run ${runId} is already being driven by engine ${row?.engine_lease_holder ?? "(unknown)"} ` +
196
197
  `(run lease expires ${row?.engine_lease_until ?? "(unknown)"}). A second \`akm workflow run\` would race it — ` +
197
198
  `wait for that invocation to finish or for the lease to expire.`, "RUN_LEASE_HELD");
198
199
  }));
@@ -4,7 +4,7 @@
4
4
  import { randomUUID } from "node:crypto";
5
5
  import { parseBundleRef } from "../../core/asset/asset-ref.js";
6
6
  import { loadConfig } from "../../core/config/config.js";
7
- import { ConfigError, NotFoundError, UsageError } from "../../core/errors.js";
7
+ import { ConfigError, NotFoundError, TransientError, UsageError } from "../../core/errors.js";
8
8
  import { appendEvent } from "../../core/events.js";
9
9
  import { warn } from "../../core/warn.js";
10
10
  import { insertEventOnce } from "../../storage/repositories/events-repository.js";
@@ -127,6 +127,25 @@ export async function startWorkflowRun(ref, params = {}, options) {
127
127
  // per the workflow-agent check-in ADR) so a stalled run can be
128
128
  // re-targeted with a `continue` directive. The agent harness + session id
129
129
  // are already resolved above (agentHarness/agentSessionId, from #501).
130
+ // #942: an active run of this ref may already exist in a DIFFERENT
131
+ // scope — the incident this issue reports (a scheduled task's cwd and a
132
+ // human's shell hash to different scope keys, so each believed it held
133
+ // no active run and each started one). The scope-local uniqueness guard
134
+ // stays scope-local (a documented, deliberate per-project partition —
135
+ // see storage-locations.md); this only warns, once, so the operator can
136
+ // resume or abandon the other run instead of silently accumulating a
137
+ // second one. `findActiveRunOutsideScope` excludes the caller's own
138
+ // scope IN SQL (never merely post-filtered) so the caller's own active
139
+ // run can never sort first under `LIMIT 1` and mask a genuinely different
140
+ // scope's run — the failure mode a same-scope-inclusive query plus a
141
+ // post-filter has with `--new`/`--force`.
142
+ const crossScopeActive = repo.findActiveRunOutsideScope(workflowRefs, scopeKey);
143
+ const crossScopeWarning = crossScopeActive
144
+ ? `Workflow ${asset.ref} already has an active run in another scope ` +
145
+ `(id ${crossScopeActive.id}, started ${crossScopeActive.created_at}, scope ${crossScopeActive.scope_key ?? "unknown"}); ` +
146
+ `starting a separate run here. Resume it from anywhere with "akm workflow run ${crossScopeActive.id}" ` +
147
+ `or free it with "akm workflow abandon ${crossScopeActive.id}".`
148
+ : undefined;
130
149
  repo.publishWorkflowRunV4({
131
150
  workflowRefs,
132
151
  ...(options?.force ? { force: true } : {}),
@@ -157,6 +176,8 @@ export async function startWorkflowRun(ref, params = {}, options) {
157
176
  revalidateSources: () => frozen.sourceCollector.revalidate(),
158
177
  });
159
178
  const result = await getWorkflowStatus(runId);
179
+ if (crossScopeWarning)
180
+ result.warnings = [...(result.warnings ?? []), crossScopeWarning];
160
181
  // #13: params are declared non-secret (they are copied verbatim into every
161
182
  // unit prompt and hashed into the unit identity, so they cannot be redacted
162
183
  // without breaking replay determinism). Surface a loud, best-effort warning
@@ -193,7 +214,7 @@ export async function getWorkflowStatus(runId, opts) {
193
214
  });
194
215
  }
195
216
  export async function listWorkflowRuns(input) {
196
- const scopeKey = getCurrentWorkflowScopeKey();
217
+ const scopeKey = input?.allScopes === true ? null : getCurrentWorkflowScopeKey();
197
218
  const activeOnly = input?.activeOnly === true;
198
219
  const includeChildren = input?.includeChildren === true;
199
220
  if (input?.workflowRef === undefined) {
@@ -205,6 +226,7 @@ export async function listWorkflowRuns(input) {
205
226
  ...(includeChildren ? { includeChildren: true } : {}),
206
227
  })
207
228
  .map(toWorkflowRunSummary),
229
+ scopeKey,
208
230
  }));
209
231
  }
210
232
  const exactRef = input.workflowRef.trim();
@@ -227,6 +249,7 @@ export async function listWorkflowRuns(input) {
227
249
  throw error;
228
250
  return {
229
251
  runs: exactRows.filter((row) => !activeOnly || row.status === "active").map(toWorkflowRunSummary),
252
+ scopeKey,
230
253
  };
231
254
  }
232
255
  if (!(error instanceof NotFoundError))
@@ -241,6 +264,7 @@ export async function listWorkflowRuns(input) {
241
264
  ...(includeChildren ? { includeChildren: true } : {}),
242
265
  })
243
266
  .map(toWorkflowRunSummary),
267
+ scopeKey,
244
268
  }));
245
269
  }
246
270
  export async function getNextWorkflowStep(specifier, params, options) {
@@ -558,7 +582,8 @@ async function resolveRunSpecifier(repo, specifier, params, parameterFlags, forc
558
582
  catch (error) {
559
583
  if (detached) {
560
584
  if (hasParameters) {
561
- throw new UsageError(`Workflow parameter flags can only be set on a new run; ${specifier} is already active.`, "INVALID_FLAG_VALUE", `Pass --new to start a separate run, or run "akm workflow abandon ${detached.id}" to free up ${specifier} first.`);
585
+ throw new UsageError(`Workflow parameter flags can only be set on a new run; ${specifier} is already active ` +
586
+ `(id ${detached.id}, scope ${scopeKey}).`, "INVALID_FLAG_VALUE", `Pass --new to start a separate run, or run "akm workflow abandon ${detached.id}" to free up ${specifier} first.`);
562
587
  }
563
588
  return { run: detached, autoStarted: false, resumed: true };
564
589
  }
@@ -570,7 +595,8 @@ async function resolveRunSpecifier(repo, specifier, params, parameterFlags, forc
570
595
  const active = forceNew ? undefined : repo.getActiveRunRowForScope(await workflowRunRefSet(ref, exactRef), scopeKey);
571
596
  if (active) {
572
597
  if (hasParameters) {
573
- throw new UsageError(`Workflow parameter flags can only be set on a new run; ${ref} is already active.`, "INVALID_FLAG_VALUE", `Pass --new to start a separate run, or run "akm workflow abandon ${active.id}" to free up ${ref} first.`);
598
+ throw new UsageError(`Workflow parameter flags can only be set on a new run; ${ref} is already active ` +
599
+ `(id ${active.id}, scope ${scopeKey}).`, "INVALID_FLAG_VALUE", `Pass --new to start a separate run, or run "akm workflow abandon ${active.id}" to free up ${ref} first.`);
574
600
  }
575
601
  return { run: active, autoStarted: false, resumed: true };
576
602
  }
@@ -808,7 +834,9 @@ function assertLeaseAllowsSpineAdvance(run, leaseHolder) {
808
834
  return;
809
835
  if (run.engine_lease_until < new Date().toISOString())
810
836
  return; // expired ⇒ claimable, not live
811
- throw new UsageError(`Workflow run ${run.id} is being driven by engine ${run.engine_lease_holder} ` +
837
+ // #948 addendum: moved off UsageError (exit 75, not exit 2) — a held lease
838
+ // is ordinary contention, not a bad command line.
839
+ throw new TransientError(`Workflow run ${run.id} is being driven by engine ${run.engine_lease_holder} ` +
812
840
  `(run lease expires ${run.engine_lease_until}). The engine owns the step spine while it runs — ` +
813
841
  `wait for it to finish or for the lease to expire before advancing steps manually.`, "RUN_LEASE_HELD");
814
842
  }
@@ -0,0 +1,52 @@
1
+ Migration notes for akm v0.9.15
2
+
3
+ `RUN_LEASE_HELD` (a workflow run-lease refusal) and the new
4
+ `STATE_DB_CONTENDED` (state.db write contention) now exit 75 (`EX_TEMPFAIL`)
5
+ instead of exit 2 and exit 70 respectively. Both are unrelated to a bad command
6
+ line — they mean "another akm process is using this database or lease right
7
+ now, retry shortly." If a script or scheduler wrapper special-cases exit 2 to
8
+ detect a held lease, switch it to exit 75, or read the JSON envelope's `code`
9
+ field instead.
10
+
11
+ The six shipped scheduled `improve` task templates now run with
12
+ `--require-engines`, which aborts (exit 78) before any index work when a
13
+ process's engine or credential cannot be resolved in the task's own
14
+ environment, instead of silently skipping that process. This only affects new
15
+ installs and new `akm setup` task seeding — an existing install's
16
+ already-materialized task files are not rewritten. To get the same protection
17
+ on an existing scheduled task, add `--require-engines` to its `run:` command
18
+ yourself, then run `akm task sync`.
19
+
20
+ `akm health --no-probe` now also skips the `cli-version` update check (a GitHub
21
+ release lookup), alongside the engine-reachability checks it already skipped.
22
+ An air-gapped or offline host's existing `--no-probe` habit now suppresses both
23
+ network calls with no config change needed.
24
+
25
+ Thinking-control wire forms (`chat_template_kwargs.enable_thinking` and
26
+ `enable_thinking`) are now sent for every provider whenever `enableThinking`
27
+ resolves to a value, not only when `provider: "vllm"` is set. If your engine
28
+ sits behind Bifrost and the gateway does not honor either form, also set
29
+ `reasoningEffort: "none"` on that engine. If your engine talks to a strict
30
+ hosted API that rejects unknown request keys, leave `enableThinking` unset on
31
+ that engine so neither wire form is sent.
32
+
33
+ An `embedding.model` rename no longer forces a full re-embed by itself: `akm
34
+ index` re-embeds a small sample first and keeps the existing vectors when they
35
+ still verify against the endpoint. `akm index --reembed` forces a full re-embed
36
+ when you don't trust that verdict. `akm index --skip-if-locked` lets a
37
+ scheduled or opportunistic run step aside instead of contending with a rebuild
38
+ already in progress; the shipped `index-refresh` scheduled task already passes
39
+ it.
40
+
41
+ There is no `embedding.concurrency` config field. Embedding throughput is
42
+ tuned by `embedding.batchSize` (documents per request) and
43
+ `embedding.maxTokens`/`contextLength` (token budget per request); the number of
44
+ requests in flight is fixed (1 for a loopback endpoint, 2 for a remote one).
45
+
46
+ A config file can now inherit a shared base via `extends: <path|bundle//path>`,
47
+ deep-merging under the local file so local keys always win. `akm config diff
48
+ <path|bundle//path>` prints every leaf that differs between this instance's
49
+ effective config and another config file or bundle-relative file. `akm config
50
+ unset` now refuses to unset a key whose value comes only from an
51
+ `extends`-inherited base, naming the source, since there would be nothing local
52
+ to remove.
@@ -7,6 +7,10 @@ live one level up in `docs/migration/`.
7
7
 
8
8
  ## Available notes
9
9
 
10
+ - [0.9.15](0.9.15.md) — exit-code 75 for lease/state.db contention,
11
+ `--require-engines` scheduled task templates, `--no-probe` cli-version
12
+ skip, thinking-control wire forms, embedding re-embed safety, and
13
+ `extends` config inheritance
10
14
  - [0.9.14](0.9.14.md) — index v22-to-v23 derived-cache rebuild, lexical
11
15
  fragments, and collapse-detector canary re-minting
12
16
  - [0.9.2](0.9.2.md) — task source v4 migration, workflow source IR v1 and