akm-cli 0.9.24 → 0.9.25-alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/CHANGELOG.md +173 -0
  2. package/dist/cli.js +1 -1
  3. package/dist/commands/health/checks.js +10 -11
  4. package/dist/commands/improve/consolidate/pair-pass.js +4 -2
  5. package/dist/commands/improve/consolidate.js +10 -4
  6. package/dist/commands/improve/execution.js +4 -11
  7. package/dist/commands/improve/extract-prompt.js +4 -4
  8. package/dist/commands/improve/extract.js +11 -13
  9. package/dist/commands/improve/improve-cli.js +65 -34
  10. package/dist/commands/improve/improve-strategies.js +49 -43
  11. package/dist/commands/improve/improve-usage-report.js +8 -17
  12. package/dist/commands/improve/loop-stages.js +3 -0
  13. package/dist/commands/improve/preparation.js +3 -1
  14. package/dist/commands/improve/reflect-noise.js +125 -0
  15. package/dist/commands/improve/reflect.js +105 -172
  16. package/dist/commands/improve/retrieval-gate.js +7 -2
  17. package/dist/commands/improve/stage.js +67 -24
  18. package/dist/commands/proposal/drain.js +11 -2
  19. package/dist/commands/proposal/proposal-cli.js +1 -5
  20. package/dist/commands/proposal/propose-cli.js +2 -2
  21. package/dist/commands/proposal/propose.js +72 -84
  22. package/dist/commands/proposal/validators/proposal-quality-validators.js +4 -2
  23. package/dist/commands/read/search-cli.js +0 -38
  24. package/dist/commands/remember.js +3 -3
  25. package/dist/commands/sources/schema-repair.js +1 -1
  26. package/dist/core/config/config-schema.js +37 -60
  27. package/dist/core/config/engine-semantics.js +15 -11
  28. package/dist/core/config/schema/improve-processes.js +18 -2
  29. package/dist/core/improve-result.js +3 -3
  30. package/dist/core/redaction.js +4 -0
  31. package/dist/core/spawn-env.js +25 -0
  32. package/dist/core/structured.js +11 -1
  33. package/dist/execution/source.js +10 -0
  34. package/dist/indexer/passes/memory-inference.js +2 -1
  35. package/dist/integrations/agent/builder-shared.js +15 -0
  36. package/dist/integrations/agent/config.js +1 -1
  37. package/dist/integrations/agent/engine-resolution.js +13 -31
  38. package/dist/integrations/agent/execution.js +48 -22
  39. package/dist/integrations/agent/index.js +1 -1
  40. package/dist/integrations/agent/profiles.js +2 -2
  41. package/dist/integrations/agent/prompts.js +55 -114
  42. package/dist/integrations/agent/request-lowering.js +21 -8
  43. package/dist/integrations/agent/runner-dispatch.js +96 -3
  44. package/dist/integrations/agent/runner.js +8 -2
  45. package/dist/integrations/harnesses/aider/agent-builder.js +10 -23
  46. package/dist/integrations/harnesses/aider/index.js +0 -5
  47. package/dist/integrations/harnesses/amazonq/agent-builder.js +9 -21
  48. package/dist/integrations/harnesses/amazonq/index.js +0 -5
  49. package/dist/integrations/harnesses/claude/agent-builder.js +47 -36
  50. package/dist/integrations/harnesses/claude/index.js +0 -14
  51. package/dist/integrations/harnesses/codex/agent-builder.js +3 -4
  52. package/dist/integrations/harnesses/codex/index.js +0 -4
  53. package/dist/integrations/harnesses/copilot/agent-builder.js +4 -13
  54. package/dist/integrations/harnesses/copilot/index.js +2 -7
  55. package/dist/integrations/harnesses/gemini/agent-builder.js +4 -13
  56. package/dist/integrations/harnesses/gemini/index.js +0 -5
  57. package/dist/integrations/harnesses/ids.js +12 -10
  58. package/dist/integrations/harnesses/opencode/agent-builder.js +41 -9
  59. package/dist/integrations/harnesses/opencode/index.js +0 -8
  60. package/dist/integrations/harnesses/opencode/model-config.js +33 -0
  61. package/dist/integrations/harnesses/opencode/model-work-agent.js +115 -0
  62. package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -7
  63. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +114 -28
  64. package/dist/integrations/harnesses/openhands/agent-builder.js +7 -21
  65. package/dist/integrations/harnesses/openhands/index.js +0 -5
  66. package/dist/integrations/harnesses/pi/agent-builder.js +7 -21
  67. package/dist/integrations/harnesses/pi/index.js +0 -5
  68. package/dist/llm/client.js +5 -0
  69. package/dist/llm/feature-gate.js +12 -9
  70. package/dist/llm/index-passes.js +2 -5
  71. package/dist/llm/memory-infer.js +6 -5
  72. package/dist/llm/structured-call.js +33 -11
  73. package/dist/output/shapes/passthrough.js +1 -0
  74. package/dist/scripts/akm-migrate-node.js +298 -228
  75. package/dist/scripts/akm-migrate.js +298 -228
  76. package/dist/workflows/exec/step-work.js +6 -5
  77. package/dist/workflows/exec/unit-dispatch.js +4 -13
  78. package/dist/workflows/freeze/step-values.js +1 -1
  79. package/docs/reference/cli.md +41 -18
  80. package/docs/reference/configuration.md +165 -12
  81. package/docs/reference/data-and-telemetry.md +2 -3
  82. package/docs/reference/workflow-schema.md +10 -9
  83. package/package.json +1 -1
  84. package/schemas/akm-config.json +108 -0
@@ -13,6 +13,7 @@ import unitPreambleTemplate from "../../assets/prompts/workflow-unit-preamble.md
13
13
  import { UsageError } from "../../core/errors.js";
14
14
  import { validateJsonSchemaSubset } from "../../core/json-schema.js";
15
15
  import { parseEmbeddedJsonResponse } from "../../core/parse.js";
16
+ import { withSchemaInstruction } from "../../core/structured.js";
16
17
  import { canonicalInputJson, validateInputs } from "../../execution/input-contract.js";
17
18
  import { withWorkflowRunsRepo, } from "../../storage/repositories/workflow-runs-repository.js";
18
19
  import { canonicalJson } from "../ir/plan-hash.js";
@@ -248,7 +249,9 @@ function buildStepWorkUnit(ctx, unitId, item, index) {
248
249
  // `inputBindings`, when non-empty — see StepWorkUnitContext.taskInputs.
249
250
  ...(taskInputs && Object.keys(taskInputs).length > 0 ? { taskInputs } : {}),
250
251
  ...(input.gateFeedback ? { gateFeedback: input.gateFeedback } : {}),
251
- ...(template.schema ? { schema: template.schema } : {}),
252
+ // An agent or SDK transport's request lowering appends the schema
253
+ // instruction itself; a direct-LLM unit carries it in its prompt.
254
+ ...(template.schema && ctx.runner === "llm" ? { schema: template.schema } : {}),
252
255
  instructions: template.instructions,
253
256
  });
254
257
  const inputHash = computeUnitInputHash(ctx, item);
@@ -373,10 +376,8 @@ export function buildUnitPrompt(input) {
373
376
  ? `\nUnmet criteria:\n${gateFeedback.missing.map((m) => `- ${m}`).join("\n")}`
374
377
  : "")
375
378
  : "";
376
- const schemaDirective = schema
377
- ? `\n\nRespond with ONLY a JSON value matching this JSON Schema (no prose, no code fences):\n${safeJson(schema)}`
378
- : "";
379
- return `${preamble}\n${instructions}${itemBlock}${inputsBlock}${taskInputsBlock}${gateBlock}${schemaDirective}`;
379
+ const prompt = `${preamble}\n${instructions}${itemBlock}${inputsBlock}${taskInputsBlock}${gateBlock}`;
380
+ return schema ? withSchemaInstruction(prompt, schema) : prompt;
380
381
  }
381
382
  /**
382
383
  * Content-derived unit identity (module doc): `<node_id>:<sha256>` for a
@@ -5,8 +5,7 @@ import { ConfigError } from "../../core/errors.js";
5
5
  import { assertFrozenDirectoryContained } from "../../execution/directory-identity.js";
6
6
  import { canonicalResolvedExecutionRequest } from "../../execution/resolved-request.js";
7
7
  import { buildExecutionFromWire } from "../../integrations/agent/execution.js";
8
- import { runExecution } from "../../integrations/agent/runner-dispatch.js";
9
- import { getHarness } from "../../integrations/harnesses/index.js";
8
+ import { runExecution, unwrapHarnessReply } from "../../integrations/agent/runner-dispatch.js";
10
9
  /** Lower a frozen common request through its frozen runner only. */
11
10
  export function prepareWorkflowExecution(request, prompt = request.prompt) {
12
11
  const target = request.frozenTarget;
@@ -107,17 +106,9 @@ export async function dispatchWorkflowExecution(request, feedback) {
107
106
  ...(notices ? { notices } : {}),
108
107
  };
109
108
  }
110
- let text = result.stdout;
111
- let sessionId = result.sessionId;
112
- if (lowered.runner.kind === "agent" && result.ok) {
113
- const harness = getHarness(lowered.runner.profile.platform ?? lowered.runner.profile.name);
114
- if (harness?.resultExtractor) {
115
- const extraction = harness.resultExtractor(result);
116
- text = extraction.text;
117
- if (extraction.sessionId !== undefined)
118
- sessionId = extraction.sessionId;
119
- }
120
- }
109
+ const extraction = result.ok ? unwrapHarnessReply(lowered.runner, result) : undefined;
110
+ const text = extraction?.text ?? result.stdout;
111
+ const sessionId = extraction?.sessionId ?? result.sessionId;
121
112
  return {
122
113
  ok: result.ok,
123
114
  text,
@@ -33,7 +33,7 @@ export function targetConcurrency(runner, config) {
33
33
  if (runner.kind !== "sdk" || !runner.fallbackConnection)
34
34
  return undefined;
35
35
  const selected = typeof runner.engine === "string" ? config.engines?.[runner.engine] : undefined;
36
- const fallbackName = selected?.kind === "agent" ? (selected.llmEngine ?? config.defaults?.llmEngine) : undefined;
36
+ const fallbackName = selected?.kind === "agent" ? selected.llmEngine : undefined;
37
37
  const fallback = fallbackName ? config.engines?.[fallbackName] : undefined;
38
38
  return defaultLlmEngineConcurrency(runner.fallbackConnection.endpoint, fallback?.kind === "llm" ? fallback.concurrency : undefined);
39
39
  }
@@ -517,7 +517,6 @@ kept.
517
517
  | `--filter` | `<key>=<value>` | _(none)_ | Scope filter — repeatable. Valid keys: `user`, `agent`, `run`, `channel`. Example: `--filter user=alice --filter channel=ops`. Narrows the result set; ranking is unchanged. |
518
518
  | `--include-proposed` | flag | `false` | Include entries with `quality: "proposed"` in the result set. Default search excludes them; `generated` and `curated` quality entries are always included. Unknown quality values warn once and remain searchable. |
519
519
  | `--belief` | `all`, `current`, `historical` | `all` | Memory belief filter. `current` keeps active memory beliefs; `historical` keeps contradicted/superseded/archived ones. |
520
- | `--track-usage`, `--no-track-usage` | flag | `true` | Record or suppress local usage events for this successful read |
521
520
  | `--include-sessions` | flag | `false` | Include session assets, which are excluded from default results via `config.search.defaultExcludeTypes` |
522
521
  | `--format` | `json`, `jsonl`, `yaml`, `text`, `md`, `html` | `json` | Output format |
523
522
  | `--detail` | `brief`, `normal`, `full` | `brief` | Output verbosity level |
@@ -593,7 +592,6 @@ akm curate "learn the release workflow" --from all --format text
593
592
  | `--type` | `skill`, `command`, `agent`, `knowledge`, `instruction`, `workflow`, `script`, `memory`, `env`, `secret`, `lesson`, `task`, `session`, `fact`, `any` | `any` | Filter curated results by asset type |
594
593
  | `--limit` | number | `4` | Maximum curated results |
595
594
  | `--from` | `local`, `registry`, `all` | `local` | Where to search before curating |
596
- | `--track-usage`, `--no-track-usage` | flag | `true` | Record or suppress local usage events for this successful read |
597
595
 
598
596
  `akm curate` takes the top `--limit` hits of one search, in search order, and
599
597
  enriches each with a preview, run details and up to two support refs: the
@@ -623,7 +621,6 @@ curated like any other.
623
621
  every prompt: it only ever reads the index as it currently stands (the same
624
622
  non-blocking `ensureIndex()` path `search` uses) and never waits on or
625
623
  contends with a full `akm index` rebuild in progress.
626
- Use `--no-track-usage` when this inspection must not record usage events.
627
624
 
628
625
  ### show
629
626
 
@@ -631,9 +628,6 @@ Display an asset by ref. On a markdown document `#fragment` selects one
631
628
  section by heading slug (falling back to case-insensitive heading text); an
632
629
  unmatched fragment lists the available slugs.
633
630
 
634
- Successful reads record local usage events by default; pass
635
- `--no-track-usage` to suppress them.
636
-
637
631
  ```sh
638
632
  akm show scripts/deploy.sh
639
633
  akm show skills/code-review
@@ -662,7 +656,6 @@ akm show memories/retro --filter user=alice --filter agent=claude
662
656
  | `--max-chars` | positive integer | `3200` for `lead` | Hard contextual content budget in characters; requires `--context lead` and is mutually exclusive with `--max-tokens`. |
663
657
  | `--max-tokens` | positive integer | _(none)_ | Approximate contextual budget using four characters per token; requires `--context lead` and is mutually exclusive with `--max-chars`. |
664
658
  | `--filter` | `<key>=<value>` | _(none)_ | Repeatable scope filter (`user`, `agent`, `run`, `channel`). |
665
- | `--track-usage`, `--no-track-usage` | flag | `true` | Record or suppress local usage events for this successful read. |
666
659
 
667
660
  `meta` is not an asset type — `[<origin>//]meta[:<name>]` direct-reads a
668
661
  human-authored orientation doc from a bundle's optional `.meta/` directory
@@ -2326,7 +2319,8 @@ or inference payload was selected.
2326
2319
 
2327
2320
  **Platform-specific dispatch:** akm uses a platform builder to construct the
2328
2321
  CLI argv for each engine's harness platform. `platform: "opencode"` engines emit:
2329
- `opencode run [--system-prompt "..."] [--model opencode/claude-opus-4-7] "<prompt>"`.
2322
+ `opencode run [--model opencode/claude-opus-4-7] "<prompt>"`. `opencode run` has
2323
+ no system-prompt option, so akm composes a persona into the prompt.
2330
2324
  `platform: "claude"` engines emit:
2331
2325
  `claude [--system-prompt "..."] [--model claude-opus-4-7] --print -- "<prompt>"`.
2332
2326
  Agent engines may set `bin`, `args`, `workspace`, `model`, and `timeoutMs` in
@@ -2430,6 +2424,7 @@ akm improve lessons/my-lesson --show-prompt --format text # print the composed r
2430
2424
  akm improve report # LLM usage/routing report for the most recent real run
2431
2425
  akm improve report --run <id> # ...for one specific improve_runs id
2432
2426
  akm improve report --since 7d # ...aggregated over every real run started in the last 7 days
2427
+ akm improve judge < revision.json # reflect's quality judge on one revision; writes nothing
2433
2428
  ```
2434
2429
 
2435
2430
  | Flag | Description |
@@ -2446,7 +2441,7 @@ akm improve report --since 7d # ...aggregated over every real run start
2446
2441
  | `--strategy <name>` | Override the active improve strategy (a built-in or entry under `improve.strategies`) |
2447
2442
  | `--json-to-stdout` | Also emit the full persisted JSON result on stdout for a live run. Without this flag, stdout stays empty. Dry-runs always emit their result and are never persisted. |
2448
2443
  | `--skip-if-locked` | If another improve run already holds the lock, skip gracefully (exit 0) instead of failing with "already running" (exit 75, `TransientError`, code `IMPROVE_LOCK_HELD` — field follow-up to #948: two legitimate `improve` invocations colliding on this lock is ordinary, retryable contention, not a broken config file). Use for high-frequency scheduled runs so they don't pile up failures while a longer run is in progress. |
2449
- | `--require-engines` | Abort (exit 78, before any indexing, lock, or log side effect) if the active strategy would enable a process whose engine or credential cannot be resolved in this process's environment, OR whose endpoint fails a bounded reachability probe — the same probe `akm health`'s `default-llm-engine`/`configured-engines` checks run, once per distinct endpoint. Without this flag, improve degrades gracefully: it skips the affected processes and reports them in the result's `skippedProcesses`. Recommended alongside `--skip-if-locked` for scheduled runs, since the operator's own shell can pass config validation while a scheduler's stripped-down environment (see #953) cannot. |
2444
+ | `--require-engines` | Abort (exit 78, before any indexing, lock, or log side effect) if the active strategy would enable a process whose engine or credential cannot be resolved in this process's environment, OR whose endpoint fails a bounded reachability probe — the same probe `akm health`'s `default-llm-engine`/`configured-engines` checks run, once per distinct endpoint. An agent engine's check is that its binary is on PATH, and an `opencode-sdk` engine's is both its binary and, when it sets `llmEngine`, that LLM fallback's endpoint. Without this flag, improve degrades gracefully: it skips the affected processes and reports them in the result's `skippedProcesses`. Recommended alongside `--skip-if-locked` for scheduled runs, since the operator's own shell can pass config validation while a scheduler's stripped-down environment (see #953) cannot. |
2450
2445
  | `--show-prompt` | Print the composed reflect prompt (#952) for one asset and exit — before any lock, index write, or engine dispatch. Requires a fully-qualified asset ref as the scope (`akm improve lessons/my-lesson --show-prompt`); rejected with a type or whole-bundle scope. The default output format is JSON, which carries the prompt as a `prompt` field (escaped into one line) alongside the resolved `engine`/`engineKind`; pass `--format text` to print the prompt itself, unwrapped and readable by eye. |
2451
2446
  | `--sync` / `--no-sync` | Commit (and optionally push) the git-backed primary bundle when the run finishes. Default: on for git-backed bundles (per profile config). |
2452
2447
  | `--push` / `--no-push` | Push after the end-of-run sync commit when writable with a remote configured. `--no-push` commits only, skipping the push. Default: per profile config (`true`). `sync.push` stays outside the autonomy gate — this is a per-run opt-out, not a default change. |
@@ -2582,7 +2577,7 @@ table: one row per improve process (`reflect`, `distill`, `consolidate`,
2582
2577
  `memoryInference`, `extract`, `validation`, `triage`,
2583
2578
  `proactiveMaintenance`), plus a `triage.judgment` row when the strategy
2584
2579
  configures a judgment engine. Each row carries `enabled`, the resolved
2585
- `engine`/`model` (llm-backed processes only) and `engineKind`, this process's
2580
+ `engine`, its `model` (when the engine has an LLM connection) and `engineKind`, this process's
2586
2581
  own lowering `notices`, and — for reflect/distill/consolidate only —
2587
2582
  `eligibleRefs`, the count of this run's `effectiveRefs` the process would act
2588
2583
  on (`shouldSkipRef`'s allowedTypes/excludeRefPrefixes (reflect only)/
@@ -2607,7 +2602,9 @@ default probe-on behavior) to check whether a named engine actually answers.
2607
2602
  builds the exact prompt reflect would send for one asset — the same source
2608
2603
  resolution, runner selection, feedback/schema-hint/related-lesson/rejected-
2609
2604
  proposal gathering `akm improve`'s live reflect step uses — and prints it
2610
- without reading a credential, so it never calls an engine. Add
2605
+ without reading a credential, so it never calls an engine. An LLM engine
2606
+ receives the reply's JSON Schema as `response_format`, and an agent engine as
2607
+ an instruction that dispatch appends to this prompt. Add
2611
2608
  `--format text` (the default JSON/yaml envelope escapes the prompt into one
2612
2609
  line, which defeats a by-eye read) to confirm by eye that recent feedback is
2613
2610
  framed as an unverified report to investigate (never a fact to insert
@@ -2619,7 +2616,7 @@ destination than `memory`.
2619
2616
 
2620
2617
  #### improve report
2621
2618
 
2622
- `akm improve report` (#944) answers "which engine did each LLM-backed process
2619
+ `akm improve report` (#944) answers "which engine did each model-calling process
2623
2620
  use this run, how much did it cost, and which enabled processes made zero
2624
2621
  calls (and why)" without hand-written SQLite against `state.db`. It is a
2625
2622
  `scope` value, not a subcommand — `report` is not, and will never be, a real
@@ -2632,7 +2629,7 @@ field on the result (`result_json` in `improve_runs`, and in the
2632
2629
  `byProcessEngineModel` is a cross-tab of this run's own `llm_usage` events
2633
2630
  (#576) — one row per distinct `(process, engine, model)` triple, each with
2634
2631
  `calls`, `failures`, `promptTokens`, `completionTokens`, `totalTokens`,
2635
- `reasoningTokens`, and `totalDurationMs`. `noCalls` lists every LLM-backed
2632
+ `reasoningTokens`, and `totalDurationMs`. `noCalls` lists every model-calling
2636
2633
  process (`reflect`, `distill`, `consolidate`, `memoryInference`,
2637
2634
  `extract`, `validation` — not `triage`/`proactiveMaintenance`,
2638
2635
  which never make an attributable LLM call themselves) the active strategy
@@ -2657,6 +2654,17 @@ that run's own `llm_usage` events instead of erroring, sets `noCalls` to `[]`
2657
2654
  `notes` entry saying so rather than fabricating precision the old row can't
2658
2655
  support.
2659
2656
 
2657
+ #### improve judge
2658
+
2659
+ `akm improve judge` runs reflect's quality judge on one revision and prints its
2660
+ verdict, for testing a judge engine on revisions whose right answer you know. It
2661
+ reads `{"source": "...", "candidate": "...", "feedback": "...", "ref": "..."}` JSON
2662
+ from stdin (`feedback` and `ref` are optional; `ref` names the revised asset,
2663
+ which a judge on an agent engine may read), judges with the engine the strategy's
2664
+ `processes.reflect.qualityGate.engine` names (`--strategy` picks the strategy),
2665
+ and prints `{ engine, pass, score, reason, criteria }` with the gate's prompt and
2666
+ pass rule. A `score` of `-1` means the judge gave no verdict. It writes nothing.
2667
+
2660
2668
  ### proposal
2661
2669
 
2662
2670
  Manage the proposal queue. The canonical grammar is `akm proposal <verb>`:
@@ -2706,7 +2714,7 @@ akm proposal extract --type claude --location /custom/path --session-id <id>
2706
2714
  | `--dry-run` | Show candidates without queuing proposals. |
2707
2715
  | `--force` | Re-process sessions even if they were already extracted and have no new events. Default: skip already-seen sessions. |
2708
2716
  | `--timeout-ms <ms>` | Per-session LLM timeout in ms (default `600000`). |
2709
- | `--engine <name>` | Named LLM engine for this invocation. Mutually exclusive with `--strategy`. |
2717
+ | `--engine <name>` | Named engine for this invocation: an LLM engine, or a `claude`, `opencode` or `opencode-sdk` agent engine. Mutually exclusive with `--strategy`. |
2710
2718
  | `--strategy <name>` | Improve strategy supplying extract behavior and engine. Mutually exclusive with `--engine`. |
2711
2719
 
2712
2720
  `--type` and `--auto` are mutually exclusive; one of them is required.
@@ -2721,8 +2729,10 @@ dropped — a foreground polling daemon in a one-shot CLI); the shipped
2721
2729
  `core/extract.yml` cron template (`akm proposal extract --auto` on a
2722
2730
  schedule) is the answer.
2723
2731
 
2724
- Requires an LLM engine: pass `--engine`, select a `--strategy` whose
2725
- `processes.extract.engine` is set, or configure `defaults.llmEngine`.
2732
+ Requires an engine that can run unattended model work (an LLM engine, or a
2733
+ `claude`, `opencode` or `opencode-sdk` agent): pass `--engine`, select a
2734
+ `--strategy` whose `processes.extract.engine` is set, or configure
2735
+ `defaults.llmEngine`.
2726
2736
 
2727
2737
  **Output.** `ok` means the command ran to completion — it is `true` even when
2728
2738
  every session was skipped (an unreachable LLM engine included); it does not
@@ -2732,7 +2742,7 @@ harvest" branch on `skipReasons`, `warnings`, or `sessionsProcessed` /
2732
2742
 
2733
2743
  | Field | Description |
2734
2744
  | --- | --- |
2735
- | `engine` | Resolved LLM engine name for this run. Absent only when extract is disabled by the selected improve strategy (the run returns before an engine is resolved). |
2745
+ | `engine` | Resolved engine name for this run. Absent only when extract is disabled by the selected improve strategy (the run returns before an engine is resolved). |
2736
2746
  | `engineKind` | `"llm"`, `"sdk"`, or `"agent"` — the kind of runner `engine` resolved to. Same absence condition as `engine`. |
2737
2747
  | `skipReasons` | Per-`skipReason` count across `sessions[]` (e.g. `{ "llm_unavailable": 25 }`). Present only when `sessionsSkipped > 0`. |
2738
2748
  | `warnings` | Includes one aggregate line per infrastructure skip reason that fired (`llm_unavailable`, `read_failed`, `exception`, `locked_concurrent`) — e.g. `25 of 25 sessions skipped: llm_unavailable (engine "default")` — so an engine outage is visible without inspecting `sessions[]`. Session-content skips (`already_extracted`, `too_short`, `triaged_out`) are counted in `skipReasons` but never produce a warning line. |
@@ -2755,11 +2765,24 @@ akm proposal new skill code-review --path team --task "PR-style review skill" #
2755
2765
  | `--path` | Relative subdirectory under the type dir to place the proposed asset in (e.g. `release`). The filename comes from `<name>`. |
2756
2766
  | `--task` | Inline task text |
2757
2767
  | `--file` | Read task text from a UTF-8 file |
2758
- | `--engine` | Override the default execution engine |
2768
+ | `--engine` | Override the default execution engine. Any kind works: an LLM engine, an agent CLI or `opencode-sdk` |
2759
2769
  | `--timeout-ms` | Override the selected engine timeout for this call |
2760
2770
 
2761
2771
  Exactly one of `--task` or `--file` is required. Emits `propose_invoked`.
2762
2772
 
2773
+ Every engine kind returns the proposal the same way: as one JSON object on
2774
+ stdout with the asset's `ref`, its full `content` and a self-rated
2775
+ `confidence`. akm sends the object's JSON Schema with the request, as
2776
+ `response_format` to an LLM engine and as an instruction at the end of the
2777
+ prompt to an agent engine (codex also gets it as `--output-schema`). akm
2778
+ captures the reply; an agent CLI runs headless, with no live terminal session.
2779
+ A harness's own JSON envelope, such as claude's
2780
+ `--output-format json` result, is unwrapped first.
2781
+ A reply that is not a valid proposal gets one corrective retry that says what
2782
+ was wrong. If that reply is not valid either, the command exits 1 with
2783
+ `reason: "parse_error"` and an error that names the engine, for example
2784
+ `Engine "local" reply was not valid proposal JSON after 2 attempts: …`.
2785
+
2763
2786
  **Prompt-task `timeoutMs`:** a version-2 prompt task may set `timeoutMs` to
2764
2787
  override its selected engine timeout. Set it to `null` to disable the timer, or
2765
2788
  to a positive integer (milliseconds) to apply a task-specific limit.
@@ -116,9 +116,31 @@ settable via `extraParams`. A response with reasoning tokens despite
116
116
  `enableThinking: false` triggers a runtime warning and the `akm health`
117
117
  `thinking-control` advisory.
118
118
 
119
- An agent engine may set `bin`, `args`, `workspace`, `model`, and `timeoutMs`.
120
- Only `platform: "opencode-sdk"` may set `llmEngine`; it names
121
- the LLM engine used as that SDK engine's fallback connection.
119
+ When a call asks for JSON that matches a schema, an LLM engine sends the schema
120
+ as `response_format` (`json_schema`, strict) unless the engine sets
121
+ `supportsJsonSchema: false`. If the endpoint rejects it with a 4xx other than
122
+ 429, AKM retries once without `response_format` and stops sending it to that
123
+ endpoint and model for the rest of the process. An agent engine receives the
124
+ schema as one instruction at the end of its prompt (`Respond with ONLY a JSON
125
+ value matching this JSON Schema (no prose, no code fences):` followed by the
126
+ schema), plus the harness's own schema channel where it has one (codex
127
+ `--output-schema`).
128
+
129
+ An agent engine may set `bin`, `args`, `workspace`, `model`, and `timeoutMs`;
130
+ it takes no inference of its own (see
131
+ [Inference on an agent engine](#inference-on-an-agent-engine)). Only `platform: "opencode-sdk"` may set `llmEngine`; it names
132
+ the LLM engine used as that SDK engine's fallback connection. With no
133
+ `llmEngine`, an SDK engine has no fallback connection and opencode resolves
134
+ provider, model and auth from its own configuration. `defaults.llmEngine` is
135
+ not a substitute.
136
+
137
+ On an engine without `timeoutMs`, model work (an improve process, a quality or
138
+ triage judge, or an index pass) stops after 600 seconds, whatever the engine's
139
+ kind; other work on an agent engine runs until it finishes. An improve stage's
140
+ reply that the stage cannot read gets one corrective retry. Reflect holds its
141
+ reply to its own contract the same way on every engine kind: a reply that is not
142
+ the JSON object of reflect's schema gets one repair turn, and one still invalid
143
+ fails with `parse_error`.
122
144
 
123
145
  Executable assets may request tools, but the request is not authority. Configure
124
146
  the host-local `execution.allowedTools` list to define the ceiling; `"*"` is an
@@ -132,6 +154,85 @@ client with no dependencies — it spawns `opencode serve` and talks to it — s
132
154
  the npm dependency alone does not make the platform usable. Install the binary
133
155
  with `npm i -g opencode-ai` or opencode's own installer.
134
156
 
157
+ ### Inference on an agent engine
158
+
159
+ An agent engine sets no inference of its own. For `opencode`, set it on the
160
+ model in your own opencode config, which akm leaves as it is. akm carries
161
+ inference (`temperature`, `reasoningEffort`, `enableThinking`, `maxTokens`,
162
+ `contextLength`) into opencode only where it writes opencode's config itself:
163
+
164
+ - **Model work on `opencode` and `opencode-sdk`:** an improve process's `llm`
165
+ overlay becomes the options of the `akm-model-work` agent that runs the
166
+ dispatch, so opencode's own title call on the same model keeps its defaults.
167
+ - **An `opencode-sdk` engine's `llmEngine` fallback:** the model akm declares for
168
+ it under `akm-custom` carries the fallback's inference and the request's.
169
+ `maxTokens` and `contextLength` become `limit.output` and `limit.context`, and
170
+ only together: opencode refuses half a `limit`.
171
+
172
+ Inference from an asset, a workflow's `llm:` or a `models.json` alias reaches an
173
+ LLM engine. On an agent engine it is reported as an `untranslated-field`
174
+ notice and dispatch continues; model work's agent carries `temperature`,
175
+ `reasoningEffort` and `enableThinking` without one.
176
+
177
+ Reasoning effort has one word in a request, `reasoningEffort`. `effort`, as a
178
+ `models.json` alias or an asset's `effort:` frontmatter spells it, is read as
179
+ `reasoningEffort` wherever layers are merged, so an LLM engine sends it as
180
+ `reasoning_effort`.
181
+
182
+ ### Engines for unattended model work
183
+
184
+ Unattended model work is the work akm hands a model with no one watching:
185
+ - the improve processes (reflect, distill, consolidate, memory inference,
186
+ extract, and validation's repair);
187
+ - the quality, triage and retrieval-gate judges;
188
+ - index passes;
189
+ - `akm remember --enrich`.
190
+
191
+ It runs on any engine kind under one tool policy. The model may read, edit
192
+ files only inside a scratch working directory that akm creates for the
193
+ dispatch and removes after it, and run `akm search` and `akm show`. The
194
+ stash stays read-only to it: what model work changes reaches the stash only
195
+ as a proposal, through the review queue.
196
+
197
+ Each engine enforces as much of the policy as it can, and grants nothing it
198
+ cannot enforce:
199
+
200
+ | Engine | What the model gets |
201
+ |---|---|
202
+ | LLM | No tools. |
203
+ | `claude` | Read and Edit inside the working directory, and Bash for `akm search` and `akm show` only. akm runs it with `--restricted`, so your user, project and local settings cannot widen that. |
204
+ | `opencode`, `opencode-sdk` | Read, grep and glob in the working directory and the stash, edit in the working directory only, and `akm_search` and `akm_show`, through an injected `akm-model-work` agent. No bash, because opencode cannot stop a redirect such as `akm show x > file` from writing elsewhere. |
205
+ | `codex`, `copilot`, `pi`, `gemini`, `aider`, `amazonq`, `openhands` | Cannot run model work: akm refuses the request before it starts. |
206
+
207
+ **The one rule.** Every key model work reads its engine from must name an
208
+ LLM engine or a `claude`, `opencode` or `opencode-sdk` agent engine. Those
209
+ keys are:
210
+ - `defaults.llmEngine`;
211
+ - `index.defaults.engine` and `index.<pass>.engine`;
212
+ - `improve.strategies.<name>.engine`;
213
+ - `improve.strategies.<name>.processes.<process>.engine`;
214
+ - an enabled `processes.triage.judgment.engine`;
215
+ - `processes.<process>.qualityGate.engine`.
216
+
217
+ A config that breaks the rule fails to load, with an error that names the
218
+ key, the engine and its platform. Other engine keys, `defaults.engine` and
219
+ `workflow.judgeEngine`, may name any configured engine.
220
+
221
+ **Further details:**
222
+ - A model-work dispatch builds its own command, so the engine's `args` do not
223
+ apply to it, except a `--model` they name, and neither does its `workspace`.
224
+ - An improve process's `llm` overlay reaches the `akm-model-work` agent on
225
+ `opencode` and `opencode-sdk` (see
226
+ [Inference on an agent engine](#inference-on-an-agent-engine)); on any other
227
+ agent engine it is reported as `untranslated-field` notices, not errors.
228
+ - opencode model work may read the stash and write only its working directory.
229
+ `akm_search` and `akm_show` come from the akm-opencode plugin (0.9.21 or
230
+ later), which akm does not load: put it in your opencode config
231
+ (`"plugin": ["akm-opencode"]`). akm turns off the plugin's curation, learning
232
+ and write gate for these dispatches and keeps its state in akm's state
233
+ directory. The stash is protected from edits only while the temporary
234
+ directory is outside a git repository.
235
+
135
236
  ### Model-map files
136
237
 
137
238
  AKM ships an immutable `models.json` package asset with three intent aliases:
@@ -176,7 +277,8 @@ profile may omit `model` when the installed layer already supplies it, as the
176
277
  partial Claude override above does. After overlay, every alias/engine entry
177
278
  must have a usable model. Unknown profile fields are rejected; JSON-safe
178
279
  fields inside `inference` are preserved for engine adapters to lower
179
- optimistically.
280
+ optimistically. An `inference.effort` is read as `reasoningEffort`
281
+ (see [Inference on an agent engine](#inference-on-an-agent-engine)).
180
282
 
181
283
  A profile's `engine` field (0.9.15, #946) borrows a column's `model` (and, for
182
284
  an `llm`-kind engine, its inference defaults) from a configured
@@ -265,16 +367,24 @@ the embedded copy, and release tests pin copied bytes to `src/assets/models.json
265
367
  The health check passes when the optional user file is absent and warns with
266
368
  its path and JSON location when the user file is unreadable or invalid.
267
369
 
268
- `defaults.engine` names an LLM or agent engine. `defaults.llmEngine` must name
269
- an LLM engine. There is no first-engine fallback: an unset `defaults.engine`
370
+ `defaults.engine` names an LLM or agent engine. `defaults.llmEngine` names the
371
+ default engine for unattended model work, so it follows
372
+ [the one rule](#engines-for-unattended-model-work). There is no first-engine
373
+ fallback: an unset `defaults.engine`
270
374
  never resolves to some arbitrary entry in `engines`. It resolves instead to a
271
375
  synthesized, config-free `opencode-sdk` engine when the `opencode` binary is on
272
376
  PATH — announced once per run, and preempted by any `opencode-sdk` engine you
273
377
  configure yourself. Naming an engine that is not configured is always an error
274
378
  and is never rescued by that fallback.
275
379
 
380
+ `defaults.llmEngine` is not an `opencode-sdk` engine's fallback connection. An
381
+ SDK engine gets an LLM fallback only from its own `llmEngine`, so the
382
+ synthesized engine, which sets none, runs on opencode's own provider, model and
383
+ auth.
384
+
276
385
  Index passes select engines through `index.defaults.engine` or
277
- `index.<pass>.engine`. Per-pass `model`, `timeoutMs`, and `llm` fields are
386
+ `index.<pass>.engine`, which follow
387
+ [the one rule](#engines-for-unattended-model-work). Per-pass `model`, `timeoutMs`, and `llm` fields are
278
388
  invocation overrides; `enabled: false` disables that pass. Connection fields
279
389
  such as `endpoint`, `provider`, `apiKey`, and `apiKeyFile` belong only on
280
390
  named engines.
@@ -323,7 +433,8 @@ can select `engine`, `model`, `timeoutMs`, and LLM request overrides:
323
433
  }
324
434
  ```
325
435
 
326
- LLM-only improve processes require an LLM engine; an explicit invalid or
436
+ An improve process's engine follows
437
+ [the one rule](#engines-for-unattended-model-work); an explicit invalid or
327
438
  incompatible engine never falls back to another engine. Built-in strategies
328
439
  are complete presets. User-defined strategies inherit omitted fields from the
329
440
  built-in `default` strategy before applying their own overrides.
@@ -359,11 +470,13 @@ each process's LLM-as-judge quality gate. Each is on unless it sets
359
470
  `enabled: false`, and each follows only its own switch. A reflect revision
360
471
  that changes the body is never auto-accepted; when the judge passes it, it
361
472
  waits for review. With the gate off, it waits for review too. The judge is the
362
- process's own LLM engine, or `defaults.llmEngine` when an agent generates.
363
- `engine`, `model`, `timeoutMs` and `llm` give the gate a judge of its own,
473
+ process's own engine when that is an LLM engine, or the `defaults.llmEngine`
474
+ engine when an agent generates. `engine`, `model`, `timeoutMs` and `llm` give the gate a judge of its own,
364
475
  resolved over the process's settings the way `triage.judgment` resolves over
365
- triage's. A gate whose settings resolve to no LLM engine fails before anything
366
- is generated; it never falls back to another engine. The judge runs at
476
+ triage's. The judge may be any engine that follows
477
+ [the one rule](#engines-for-unattended-model-work). A gate whose settings
478
+ resolve to no engine fails before anything is generated; it never falls back
479
+ to another engine. The judge runs at
367
480
  temperature 0 with thinking off unless its engine sets `enableThinking: true`.
368
481
  Thinking is slow: on a 27B llama.cpp server, a thinking judgment took a median
369
482
  of 30–67 s and up to about 3 minutes, against about 5 s without.
@@ -392,6 +505,46 @@ judge's engine at the server directly.
392
505
  }
393
506
  ```
394
507
 
508
+ `processes.reflect.defectFilter` sets the wording of the checks reflect runs
509
+ before the judge. With the quality gate on or off, reflect refuses a revision
510
+ that adds placeholder text, talks about its own edit, or copies frontmatter into
511
+ its body: no proposal, and no judge call. Each rule counts only what the
512
+ revision adds to its source, so wording the asset already had and kept is not
513
+ held against it. Each of the three lists is optional. A list you set replaces
514
+ that rule's default list, and `[]` turns the rule off.
515
+
516
+ | List | Rule | Default |
517
+ |---|---|---|
518
+ | `placeholders` | `placeholder_added` | `please confirm`, `please verify`, `to be confirmed`, `to be determined`, `to be verified` |
519
+ | `metaCommentary` | `meta_commentary_added` | `feedback signal`, `feedback signals`, `feedback indicate`, `feedback indicates`, `feedback suggest`, `feedback suggests`, `feedback ask`, `feedback asks`, `feedback says`, `feedback report`, `feedback reports`, `feedback request`, `feedback requests`, `this revision`, `the source asset`, `the source note`, `the source memory`, `the original asset`, `the original note`, `the original memory`, `the original version of this`, `quality gate rejected`, `proposal rejected` |
520
+ | `frontmatterKeys` | `frontmatter_copied_into_body` | `sources`, `updated`, `inferenceProcessed`, `captureMode`, `beliefState`, `xrefs`, `contradictedBy`, `outcomeData`, `orderedActions`, `generated`, `verified`, `description`, `when_to_use`, `tags`, `searchHints`, `quality`, `salience`, `salienceInputs`, `lint_skip`, `type` |
521
+
522
+ `placeholders` and `metaCommentary` entries are plain phrases, not patterns:
523
+ whole words, in any case, with any run of whitespace between words.
524
+ `frontmatterKeys` entries are exact key names: a line outside a code fence that
525
+ starts with `key:` counts. The rule also refuses a `sources`, `xrefs` or
526
+ `contradictedBy` value copied into the body; `frontmatterKeys: []` turns that
527
+ off too. Every entry must be a non-empty string, or the config does not load.
528
+
529
+ ```jsonc
530
+ {
531
+ "improve": {
532
+ "strategies": {
533
+ "nightly": {
534
+ "processes": {
535
+ "reflect": {
536
+ "defectFilter": {
537
+ "placeholders": ["please confirm", "to be confirmed", "[draft]"],
538
+ "frontmatterKeys": []
539
+ }
540
+ }
541
+ }
542
+ }
543
+ }
544
+ }
545
+ }
546
+ ```
547
+
395
548
  No shipped strategy turns improve-stage session extraction on.
396
549
  `proactiveMaintenance` is on only in the `proactive-maintenance` preset; run
397
550
  `akm improve --strategy proactive-maintenance` to use that opt-in preset.
@@ -245,9 +245,8 @@ engagement, feedback signals, stable refs, and timestamps. It never leaves the
245
245
  machine unless you explicitly copy the database or send derived content to a
246
246
  configured endpoint.
247
247
 
248
- Successful `search`, `curate`, and `show` commands record usage by default.
249
- Pass `--no-track-usage` to any of those commands to leave local usage events
250
- unchanged.
248
+ Successful `search`, `curate`, and `show` commands always record usage. Machine
249
+ reads are stamped by source (below), so they never skew ranking or eval.
251
250
 
252
251
  Every runtime writer stamps provenance as `user`, `improve`, `task`, `audit`, or
253
252
  `unknown`. Direct interactive CLI traffic defaults to `user`; internal improve,
@@ -297,13 +297,14 @@ families) plus the orchestration keys:
297
297
  - `params` — name → `{ type, description }` (JSON-Schema-typed, unlike a bare
298
298
  description string).
299
299
  - `defaults` — run-level dispatch defaults (`engine`, `model`, `llm`,
300
- `timeout`, `on_error`), overridable per unit. `defaults.llm` is the
301
- exception: `llm:` tuning applies only to engines of kind `llm`, and a
302
- document-level `llm:` reaches EVERY step, so a document that also has a step
303
- on an agent engine fails to freeze — naming the step and the engine — rather
304
- than dropping the settings for that step. There is no per-step opt-out (`llm:
305
- {}` is a no-op and `llm: null` is a parse error), so in a mixed document put
306
- `llm:` on the `unit:` of each LLM step instead of in `defaults:`.
300
+ `timeout`, `on_error`), overridable per unit. `llm:` tuning reaches an engine
301
+ of any kind, and a document-level `llm:` reaches EVERY step. An LLM engine
302
+ sends it; an agent engine reports it as an `untranslated-field` notice on that
303
+ step (see
304
+ [Inference on an agent engine](configuration.md#inference-on-an-agent-engine)).
305
+ There is no per-step opt-out (`llm: {}` is a no-op and `llm: null` is a parse
306
+ error), so in a mixed document put `llm:` on the `unit:` of each step it is
307
+ meant for instead of in `defaults:`.
307
308
  - `outputs` — name → `{ from, schema? }`, a run-level export projected from
308
309
  a step's own artifact (Markdown-only; see [Workflow
309
310
  outputs](#workflow-outputs) below).
@@ -1179,8 +1180,8 @@ and return HTTP 500 — a hard failure, so loopback endpoints stay at 1 and a
1179
1180
  `engines.<name>.concurrency` yourself. Remote providers fail softly (a
1180
1181
  retryable 429), and four concurrent completions is well inside any hosted
1181
1182
  provider's entry tier. Agent engines carry no concurrency limit of their own —
1182
- except an `opencode-sdk` engine with an `llmEngine` fallback, which inherits
1183
- that fallback engine's limit.
1183
+ except an `opencode-sdk` engine that sets `llmEngine`, which inherits that
1184
+ fallback engine's limit.
1184
1185
 
1185
1186
  #### What counts as a loopback endpoint
1186
1187
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "akm-cli",
3
- "version": "0.9.24",
3
+ "version": "0.9.25-alpha.2",
4
4
  "type": "module",
5
5
  "description": "akm (Agent Knowledge Manager) — a portable, local-first capability library for AI agents. Discover, load, share, and improve reusable skills, scripts, workflows, and knowledge across any shell-capable coding agent, including Claude Code, OpenCode, and Cursor.",
6
6
  "keywords": [