akm-cli 0.9.24 → 0.9.25-alpha.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/CHANGELOG.md +294 -0
  2. package/dist/commands/health/checks.js +10 -11
  3. package/dist/commands/improve/consolidate/pair-pass.js +3 -2
  4. package/dist/commands/improve/consolidate.js +3 -2
  5. package/dist/commands/improve/execution.js +4 -10
  6. package/dist/commands/improve/extract-prompt.js +4 -4
  7. package/dist/commands/improve/extract.js +10 -13
  8. package/dist/commands/improve/improve-cli.js +32 -33
  9. package/dist/commands/improve/improve-strategies.js +49 -43
  10. package/dist/commands/improve/improve-usage-report.js +8 -17
  11. package/dist/commands/improve/preparation.js +3 -1
  12. package/dist/commands/improve/reflect.js +108 -172
  13. package/dist/commands/improve/stage.js +69 -18
  14. package/dist/commands/proposal/drain.js +14 -2
  15. package/dist/commands/proposal/proposal-cli.js +1 -5
  16. package/dist/commands/proposal/propose-cli.js +2 -2
  17. package/dist/commands/proposal/propose.js +80 -83
  18. package/dist/commands/remember.js +3 -3
  19. package/dist/commands/sources/schema-repair.js +1 -1
  20. package/dist/core/config/config-schema.js +37 -60
  21. package/dist/core/config/engine-semantics.js +15 -11
  22. package/dist/core/config/schema/engines.js +33 -15
  23. package/dist/core/config/schema/improve-processes.js +2 -2
  24. package/dist/core/improve-result.js +3 -3
  25. package/dist/core/structured.js +10 -0
  26. package/dist/execution/source.js +14 -0
  27. package/dist/indexer/passes/memory-inference.js +2 -1
  28. package/dist/integrations/agent/builder-shared.js +15 -0
  29. package/dist/integrations/agent/config.js +2 -0
  30. package/dist/integrations/agent/engine-resolution.js +16 -31
  31. package/dist/integrations/agent/execution.js +46 -21
  32. package/dist/integrations/agent/index.js +1 -1
  33. package/dist/integrations/agent/model-map.js +16 -15
  34. package/dist/integrations/agent/prompts.js +53 -76
  35. package/dist/integrations/agent/request-lowering.js +20 -9
  36. package/dist/integrations/agent/runner-dispatch.js +103 -4
  37. package/dist/integrations/agent/runner.js +8 -2
  38. package/dist/integrations/harnesses/aider/agent-builder.js +11 -23
  39. package/dist/integrations/harnesses/aider/index.js +0 -5
  40. package/dist/integrations/harnesses/amazonq/agent-builder.js +10 -21
  41. package/dist/integrations/harnesses/amazonq/index.js +0 -5
  42. package/dist/integrations/harnesses/claude/agent-builder.js +61 -38
  43. package/dist/integrations/harnesses/claude/index.js +0 -14
  44. package/dist/integrations/harnesses/codex/agent-builder.js +4 -4
  45. package/dist/integrations/harnesses/codex/index.js +0 -4
  46. package/dist/integrations/harnesses/copilot/agent-builder.js +4 -13
  47. package/dist/integrations/harnesses/copilot/index.js +2 -7
  48. package/dist/integrations/harnesses/gemini/agent-builder.js +4 -13
  49. package/dist/integrations/harnesses/gemini/index.js +0 -5
  50. package/dist/integrations/harnesses/ids.js +18 -10
  51. package/dist/integrations/harnesses/opencode/agent-builder.js +52 -10
  52. package/dist/integrations/harnesses/opencode/index.js +0 -8
  53. package/dist/integrations/harnesses/opencode/model-config.js +80 -0
  54. package/dist/integrations/harnesses/opencode/model-work-agent.js +80 -0
  55. package/dist/integrations/harnesses/opencode-sdk/harness.js +6 -7
  56. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +168 -24
  57. package/dist/integrations/harnesses/openhands/agent-builder.js +8 -21
  58. package/dist/integrations/harnesses/openhands/index.js +0 -5
  59. package/dist/integrations/harnesses/pi/agent-builder.js +8 -21
  60. package/dist/integrations/harnesses/pi/index.js +0 -5
  61. package/dist/llm/client.js +5 -0
  62. package/dist/llm/feature-gate.js +10 -4
  63. package/dist/llm/index-passes.js +3 -6
  64. package/dist/llm/memory-infer.js +6 -5
  65. package/dist/llm/structured-call.js +33 -11
  66. package/dist/scripts/akm-migrate-node.js +298 -204
  67. package/dist/scripts/akm-migrate.js +298 -204
  68. package/dist/workflows/exec/step-work.js +6 -5
  69. package/dist/workflows/freeze/step-values.js +1 -1
  70. package/docs/reference/cli.md +29 -11
  71. package/docs/reference/configuration.md +168 -11
  72. package/docs/reference/workflow-schema.md +13 -9
  73. package/package.json +1 -1
  74. package/schemas/akm-config.json +36 -0
@@ -13,6 +13,7 @@ import unitPreambleTemplate from "../../assets/prompts/workflow-unit-preamble.md
13
13
  import { UsageError } from "../../core/errors.js";
14
14
  import { validateJsonSchemaSubset } from "../../core/json-schema.js";
15
15
  import { parseEmbeddedJsonResponse } from "../../core/parse.js";
16
+ import { withSchemaInstruction } from "../../core/structured.js";
16
17
  import { canonicalInputJson, validateInputs } from "../../execution/input-contract.js";
17
18
  import { withWorkflowRunsRepo, } from "../../storage/repositories/workflow-runs-repository.js";
18
19
  import { canonicalJson } from "../ir/plan-hash.js";
@@ -248,7 +249,9 @@ function buildStepWorkUnit(ctx, unitId, item, index) {
248
249
  // `inputBindings`, when non-empty — see StepWorkUnitContext.taskInputs.
249
250
  ...(taskInputs && Object.keys(taskInputs).length > 0 ? { taskInputs } : {}),
250
251
  ...(input.gateFeedback ? { gateFeedback: input.gateFeedback } : {}),
251
- ...(template.schema ? { schema: template.schema } : {}),
252
+ // An agent or SDK transport's request lowering appends the schema
253
+ // instruction itself; a direct-LLM unit carries it in its prompt.
254
+ ...(template.schema && ctx.runner === "llm" ? { schema: template.schema } : {}),
252
255
  instructions: template.instructions,
253
256
  });
254
257
  const inputHash = computeUnitInputHash(ctx, item);
@@ -373,10 +376,8 @@ export function buildUnitPrompt(input) {
373
376
  ? `\nUnmet criteria:\n${gateFeedback.missing.map((m) => `- ${m}`).join("\n")}`
374
377
  : "")
375
378
  : "";
376
- const schemaDirective = schema
377
- ? `\n\nRespond with ONLY a JSON value matching this JSON Schema (no prose, no code fences):\n${safeJson(schema)}`
378
- : "";
379
- return `${preamble}\n${instructions}${itemBlock}${inputsBlock}${taskInputsBlock}${gateBlock}${schemaDirective}`;
379
+ const prompt = `${preamble}\n${instructions}${itemBlock}${inputsBlock}${taskInputsBlock}${gateBlock}`;
380
+ return schema ? withSchemaInstruction(prompt, schema) : prompt;
380
381
  }
381
382
  /**
382
383
  * Content-derived unit identity (module doc): `<node_id>:<sha256>` for a
@@ -33,7 +33,7 @@ export function targetConcurrency(runner, config) {
33
33
  if (runner.kind !== "sdk" || !runner.fallbackConnection)
34
34
  return undefined;
35
35
  const selected = typeof runner.engine === "string" ? config.engines?.[runner.engine] : undefined;
36
- const fallbackName = selected?.kind === "agent" ? (selected.llmEngine ?? config.defaults?.llmEngine) : undefined;
36
+ const fallbackName = selected?.kind === "agent" ? selected.llmEngine : undefined;
37
37
  const fallback = fallbackName ? config.engines?.[fallbackName] : undefined;
38
38
  return defaultLlmEngineConcurrency(runner.fallbackConnection.endpoint, fallback?.kind === "llm" ? fallback.concurrency : undefined);
39
39
  }
@@ -2326,7 +2326,8 @@ or inference payload was selected.
2326
2326
 
2327
2327
  **Platform-specific dispatch:** akm uses a platform builder to construct the
2328
2328
  CLI argv for each engine's harness platform. `platform: "opencode"` engines emit:
2329
- `opencode run [--system-prompt "..."] [--model opencode/claude-opus-4-7] "<prompt>"`.
2329
+ `opencode run [--model opencode/claude-opus-4-7] "<prompt>"`. `opencode run` has
2330
+ no system-prompt option, so akm composes a persona into the prompt.
2330
2331
  `platform: "claude"` engines emit:
2331
2332
  `claude [--system-prompt "..."] [--model claude-opus-4-7] --print -- "<prompt>"`.
2332
2333
  Agent engines may set `bin`, `args`, `workspace`, `model`, and `timeoutMs` in
@@ -2446,7 +2447,7 @@ akm improve report --since 7d # ...aggregated over every real run start
2446
2447
  | `--strategy <name>` | Override the active improve strategy (a built-in or entry under `improve.strategies`) |
2447
2448
  | `--json-to-stdout` | Also emit the full persisted JSON result on stdout for a live run. Without this flag, stdout stays empty. Dry-runs always emit their result and are never persisted. |
2448
2449
  | `--skip-if-locked` | If another improve run already holds the lock, skip gracefully (exit 0) instead of failing with "already running" (exit 75, `TransientError`, code `IMPROVE_LOCK_HELD` — field follow-up to #948: two legitimate `improve` invocations colliding on this lock is ordinary, retryable contention, not a broken config file). Use for high-frequency scheduled runs so they don't pile up failures while a longer run is in progress. |
2449
- | `--require-engines` | Abort (exit 78, before any indexing, lock, or log side effect) if the active strategy would enable a process whose engine or credential cannot be resolved in this process's environment, OR whose endpoint fails a bounded reachability probe — the same probe `akm health`'s `default-llm-engine`/`configured-engines` checks run, once per distinct endpoint. Without this flag, improve degrades gracefully: it skips the affected processes and reports them in the result's `skippedProcesses`. Recommended alongside `--skip-if-locked` for scheduled runs, since the operator's own shell can pass config validation while a scheduler's stripped-down environment (see #953) cannot. |
2450
+ | `--require-engines` | Abort (exit 78, before any indexing, lock, or log side effect) if the active strategy would enable a process whose engine or credential cannot be resolved in this process's environment, OR whose endpoint fails a bounded reachability probe — the same probe `akm health`'s `default-llm-engine`/`configured-engines` checks run, once per distinct endpoint. An agent engine's check is that its binary is on PATH, and an `opencode-sdk` engine's is both its binary and, when it sets `llmEngine`, that LLM fallback's endpoint. Without this flag, improve degrades gracefully: it skips the affected processes and reports them in the result's `skippedProcesses`. Recommended alongside `--skip-if-locked` for scheduled runs, since the operator's own shell can pass config validation while a scheduler's stripped-down environment (see #953) cannot. |
2450
2451
  | `--show-prompt` | Print the composed reflect prompt (#952) for one asset and exit — before any lock, index write, or engine dispatch. Requires a fully-qualified asset ref as the scope (`akm improve lessons/my-lesson --show-prompt`); rejected with a type or whole-bundle scope. The default output format is JSON, which carries the prompt as a `prompt` field (escaped into one line) alongside the resolved `engine`/`engineKind`; pass `--format text` to print the prompt itself, unwrapped and readable by eye. |
2451
2452
  | `--sync` / `--no-sync` | Commit (and optionally push) the git-backed primary bundle when the run finishes. Default: on for git-backed bundles (per profile config). |
2452
2453
  | `--push` / `--no-push` | Push after the end-of-run sync commit when writable with a remote configured. `--no-push` commits only, skipping the push. Default: per profile config (`true`). `sync.push` stays outside the autonomy gate — this is a per-run opt-out, not a default change. |
@@ -2582,7 +2583,7 @@ table: one row per improve process (`reflect`, `distill`, `consolidate`,
2582
2583
  `memoryInference`, `extract`, `validation`, `triage`,
2583
2584
  `proactiveMaintenance`), plus a `triage.judgment` row when the strategy
2584
2585
  configures a judgment engine. Each row carries `enabled`, the resolved
2585
- `engine`/`model` (llm-backed processes only) and `engineKind`, this process's
2586
+ `engine`, its `model` (when the engine has an LLM connection) and `engineKind`, this process's
2586
2587
  own lowering `notices`, and — for reflect/distill/consolidate only —
2587
2588
  `eligibleRefs`, the count of this run's `effectiveRefs` the process would act
2588
2589
  on (`shouldSkipRef`'s allowedTypes/excludeRefPrefixes (reflect only)/
@@ -2607,7 +2608,9 @@ default probe-on behavior) to check whether a named engine actually answers.
2607
2608
  builds the exact prompt reflect would send for one asset — the same source
2608
2609
  resolution, runner selection, feedback/schema-hint/related-lesson/rejected-
2609
2610
  proposal gathering `akm improve`'s live reflect step uses — and prints it
2610
- without reading a credential, so it never calls an engine. Add
2611
+ without reading a credential, so it never calls an engine. An LLM engine
2612
+ receives the reply's JSON Schema as `response_format`, and an agent engine as
2613
+ an instruction that dispatch appends to this prompt. Add
2611
2614
  `--format text` (the default JSON/yaml envelope escapes the prompt into one
2612
2615
  line, which defeats a by-eye read) to confirm by eye that recent feedback is
2613
2616
  framed as an unverified report to investigate (never a fact to insert
@@ -2619,7 +2622,7 @@ destination than `memory`.
2619
2622
 
2620
2623
  #### improve report
2621
2624
 
2622
- `akm improve report` (#944) answers "which engine did each LLM-backed process
2625
+ `akm improve report` (#944) answers "which engine did each model-calling process
2623
2626
  use this run, how much did it cost, and which enabled processes made zero
2624
2627
  calls (and why)" without hand-written SQLite against `state.db`. It is a
2625
2628
  `scope` value, not a subcommand — `report` is not, and will never be, a real
@@ -2632,7 +2635,7 @@ field on the result (`result_json` in `improve_runs`, and in the
2632
2635
  `byProcessEngineModel` is a cross-tab of this run's own `llm_usage` events
2633
2636
  (#576) — one row per distinct `(process, engine, model)` triple, each with
2634
2637
  `calls`, `failures`, `promptTokens`, `completionTokens`, `totalTokens`,
2635
- `reasoningTokens`, and `totalDurationMs`. `noCalls` lists every LLM-backed
2638
+ `reasoningTokens`, and `totalDurationMs`. `noCalls` lists every model-calling
2636
2639
  process (`reflect`, `distill`, `consolidate`, `memoryInference`,
2637
2640
  `extract`, `validation` — not `triage`/`proactiveMaintenance`,
2638
2641
  which never make an attributable LLM call themselves) the active strategy
@@ -2706,7 +2709,7 @@ akm proposal extract --type claude --location /custom/path --session-id <id>
2706
2709
  | `--dry-run` | Show candidates without queuing proposals. |
2707
2710
  | `--force` | Re-process sessions even if they were already extracted and have no new events. Default: skip already-seen sessions. |
2708
2711
  | `--timeout-ms <ms>` | Per-session LLM timeout in ms (default `600000`). |
2709
- | `--engine <name>` | Named LLM engine for this invocation. Mutually exclusive with `--strategy`. |
2712
+ | `--engine <name>` | Named engine for this invocation: an LLM engine, or a `claude`, `opencode` or `opencode-sdk` agent engine. Mutually exclusive with `--strategy`. |
2710
2713
  | `--strategy <name>` | Improve strategy supplying extract behavior and engine. Mutually exclusive with `--engine`. |
2711
2714
 
2712
2715
  `--type` and `--auto` are mutually exclusive; one of them is required.
@@ -2721,8 +2724,10 @@ dropped — a foreground polling daemon in a one-shot CLI); the shipped
2721
2724
  `core/extract.yml` cron template (`akm proposal extract --auto` on a
2722
2725
  schedule) is the answer.
2723
2726
 
2724
- Requires an LLM engine: pass `--engine`, select a `--strategy` whose
2725
- `processes.extract.engine` is set, or configure `defaults.llmEngine`.
2727
+ Requires an engine that can run unattended model work (an LLM engine, or a
2728
+ `claude`, `opencode` or `opencode-sdk` agent): pass `--engine`, select a
2729
+ `--strategy` whose `processes.extract.engine` is set, or configure
2730
+ `defaults.llmEngine`.
2726
2731
 
2727
2732
  **Output.** `ok` means the command ran to completion — it is `true` even when
2728
2733
  every session was skipped (an unreachable LLM engine included); it does not
@@ -2732,7 +2737,7 @@ harvest" branch on `skipReasons`, `warnings`, or `sessionsProcessed` /
2732
2737
 
2733
2738
  | Field | Description |
2734
2739
  | --- | --- |
2735
- | `engine` | Resolved LLM engine name for this run. Absent only when extract is disabled by the selected improve strategy (the run returns before an engine is resolved). |
2740
+ | `engine` | Resolved engine name for this run. Absent only when extract is disabled by the selected improve strategy (the run returns before an engine is resolved). |
2736
2741
  | `engineKind` | `"llm"`, `"sdk"`, or `"agent"` — the kind of runner `engine` resolved to. Same absence condition as `engine`. |
2737
2742
  | `skipReasons` | Per-`skipReason` count across `sessions[]` (e.g. `{ "llm_unavailable": 25 }`). Present only when `sessionsSkipped > 0`. |
2738
2743
  | `warnings` | Includes one aggregate line per infrastructure skip reason that fired (`llm_unavailable`, `read_failed`, `exception`, `locked_concurrent`) — e.g. `25 of 25 sessions skipped: llm_unavailable (engine "default")` — so an engine outage is visible without inspecting `sessions[]`. Session-content skips (`already_extracted`, `too_short`, `triaged_out`) are counted in `skipReasons` but never produce a warning line. |
@@ -2755,11 +2760,24 @@ akm proposal new skill code-review --path team --task "PR-style review skill" #
2755
2760
  | `--path` | Relative subdirectory under the type dir to place the proposed asset in (e.g. `release`). The filename comes from `<name>`. |
2756
2761
  | `--task` | Inline task text |
2757
2762
  | `--file` | Read task text from a UTF-8 file |
2758
- | `--engine` | Override the default execution engine |
2763
+ | `--engine` | Override the default execution engine. Any kind works: an LLM engine, an agent CLI or `opencode-sdk` |
2759
2764
  | `--timeout-ms` | Override the selected engine timeout for this call |
2760
2765
 
2761
2766
  Exactly one of `--task` or `--file` is required. Emits `propose_invoked`.
2762
2767
 
2768
+ Every engine kind returns the proposal the same way: as one JSON object on
2769
+ stdout with the asset's `ref`, its full `content` and a self-rated
2770
+ `confidence`. akm sends the object's JSON Schema with the request, as
2771
+ `response_format` to an LLM engine and as an instruction at the end of the
2772
+ prompt to an agent engine (codex also gets it as `--output-schema`). akm
2773
+ captures the reply; an agent CLI runs headless, with no live terminal session.
2774
+ A harness's own JSON envelope, such as claude's
2775
+ `--output-format json` result, is unwrapped first.
2776
+ A reply that is not a valid proposal gets one corrective retry that says what
2777
+ was wrong. If that reply is not valid either, the command exits 1 with
2778
+ `reason: "parse_error"` and an error that names the engine, for example
2779
+ `Engine "local" reply was not valid proposal JSON after 2 attempts: …`.
2780
+
2763
2781
  **Prompt-task `timeoutMs`:** a version-2 prompt task may set `timeoutMs` to
2764
2782
  override its selected engine timeout. Set it to `null` to disable the timer, or
2765
2783
  to a positive integer (milliseconds) to apply a task-specific limit.
@@ -116,9 +116,32 @@ settable via `extraParams`. A response with reasoning tokens despite
116
116
  `enableThinking: false` triggers a runtime warning and the `akm health`
117
117
  `thinking-control` advisory.
118
118
 
119
- An agent engine may set `bin`, `args`, `workspace`, `model`, and `timeoutMs`.
119
+ When a call asks for JSON that matches a schema, an LLM engine sends the schema
120
+ as `response_format` (`json_schema`, strict) unless the engine sets
121
+ `supportsJsonSchema: false`. If the endpoint rejects it with a 4xx other than
122
+ 429, AKM retries once without `response_format` and stops sending it to that
123
+ endpoint and model for the rest of the process. An agent engine receives the
124
+ schema as one instruction at the end of its prompt (`Respond with ONLY a JSON
125
+ value matching this JSON Schema (no prose, no code fences):` followed by the
126
+ schema), plus the harness's own schema channel where it has one (codex
127
+ `--output-schema`).
128
+
129
+ An agent engine may set `bin`, `args`, `workspace`, `model`, and `timeoutMs`,
130
+ and the inference fields its platform translates (see
131
+ [Inference on an agent engine](#inference-on-an-agent-engine)).
120
132
  Only `platform: "opencode-sdk"` may set `llmEngine`; it names
121
- the LLM engine used as that SDK engine's fallback connection.
133
+ the LLM engine used as that SDK engine's fallback connection. With no
134
+ `llmEngine`, an SDK engine has no fallback connection and opencode resolves
135
+ provider, model and auth from its own configuration. `defaults.llmEngine` is
136
+ not a substitute.
137
+
138
+ On an engine without `timeoutMs`, model work (an improve process, a quality or
139
+ triage judge, or an index pass) stops after 600 seconds, whatever the engine's
140
+ kind; other work on an agent engine runs until it finishes. An improve stage's
141
+ reply that does not match the stage's JSON Schema gets one corrective retry
142
+ before the stage reads it. Reflect holds its reply to its own contract the same
143
+ way on every engine kind: a reply that is not the JSON object of reflect's
144
+ schema gets one repair turn, and one still invalid fails with `parse_error`.
122
145
 
123
146
  Executable assets may request tools, but the request is not authority. Configure
124
147
  the host-local `execution.allowedTools` list to define the ceiling; `"*"` is an
@@ -132,6 +155,126 @@ client with no dependencies — it spawns `opencode serve` and talks to it — s
132
155
  the npm dependency alone does not make the platform usable. Install the binary
133
156
  with `npm i -g opencode-ai` or opencode's own installer.
134
157
 
158
+ ### Inference on an agent engine
159
+
160
+ A request's inference (`temperature`, `maxTokens`, `contextLength`,
161
+ `enableThinking`, `reasoningEffort`) reaches an agent engine's harness from
162
+ every place it can come from: the engine's own settings, an improve process's
163
+ `llm` overlay, a task, command or agent asset's `inference`, a workflow's
164
+ `llm:`, and a `models.json` alias. The nearest layer wins, field by field, as
165
+ for an LLM engine. With no setting anywhere akm sends nothing of its own, so
166
+ the model's own default applies: a `reasoningEffort: "none"` that your opencode
167
+ config sets on a model stays in force until a layer overrides it.
168
+
169
+ Each platform translates what it can carry:
170
+
171
+ | Platform | Translates | How |
172
+ |---|---|---|
173
+ | `claude` | `reasoningEffort` | `--effort <level>`, in the harness's own levels (`low`, `medium`, `high`, `xhigh`, `max` in Claude Code 2.1.283). The value is passed as given, so one Claude Code rejects fails the dispatch. |
174
+ | `opencode`, `opencode-sdk` | `temperature`, `reasoningEffort`, `enableThinking`, and `maxTokens` with `contextLength` | Injected opencode config, below. |
175
+ | every other platform | nothing | |
176
+
177
+ An agent engine that sets a field its platform does not translate fails to
178
+ load, with an error that names the platform and the fields it does translate.
179
+ Inference that reaches the engine from an asset or a caller, and that the
180
+ platform does not translate, is reported as an `untranslated-field` notice and
181
+ dispatch continues, because the same asset may run on any engine.
182
+
183
+ On `opencode` and `opencode-sdk`, inference goes into opencode's config for the
184
+ model the dispatch names: the request's model, or the `--model` an `opencode`
185
+ engine's `args` name. That must be a `provider/model`, except for an
186
+ `opencode-sdk` engine with an `llmEngine` fallback, which routes the
187
+ fallback's model through its own provider. Without a model to attach to the
188
+ fields are reported as untranslated, except that model work's agent (below)
189
+ carries its options whichever model opencode picks, so an `opencode-sdk` engine
190
+ with no `model` and no `llmEngine` still gets them. The injected entry merges
191
+ over your own opencode config for the same provider and model, so what the
192
+ request does not set stays as you wrote it. A dispatch with no inference injects
193
+ nothing. akm
194
+ checked these shapes against opencode 1.18.25 and an OpenAI-compatible provider
195
+ (`@ai-sdk/openai-compatible`); another provider package is given the same
196
+ options and may ignore one.
197
+
198
+ - `temperature` becomes `options.temperature`, and `reasoningEffort` becomes
199
+ `options.reasoningEffort`. opencode reads the camelCase name: its
200
+ `reasoning_effort` spelling is dropped, so a model option written that way in
201
+ `opencode.jsonc` does nothing.
202
+ - `enableThinking` becomes `options.chat_template_kwargs.enable_thinking` and
203
+ `options.enable_thinking`, the two forms an LLM engine sends.
204
+ - `maxTokens` becomes `limit.output` and `contextLength` becomes
205
+ `limit.context`. Set them together. Without `limit.output` opencode asks for
206
+ `max_tokens: 32000`, which a small-context server rejects, and a wrong
207
+ `limit.context` lets opencode build requests longer than the server's window.
208
+ opencode refuses a `limit` with only one of them, and a half would overwrite
209
+ the other half of a limit you declared for the model, so akm declares `limit`
210
+ only when it has both and reports a lone `maxTokens` or `contextLength` as
211
+ untranslated.
212
+ - For model work (below) the options go on the `akm-model-work` agent that runs
213
+ the dispatch, so opencode's own call on the same model, a title for the
214
+ session, keeps the model's defaults. Any other dispatch runs your own agent,
215
+ whose name akm cannot rely on, so the options go on the model and the title
216
+ call sees them too. `opencode-sdk` names its session, so it makes no title
217
+ call.
218
+ - `opencode-sdk` declares the model of its `llmEngine` fallback with the
219
+ fallback's own inference, under the request's, field by field. Each distinct
220
+ set of inference starts its own `opencode serve`, as a different model does.
221
+
222
+ Reasoning effort has one word in a request, `reasoningEffort`. `effort`, as a
223
+ `models.json` alias or an asset's `effort:` frontmatter spells it, is the same
224
+ setting and is read as `reasoningEffort` wherever layers are merged, so the
225
+ nearest layer wins whichever word it used. On an LLM engine an alias's or
226
+ asset's `effort` is therefore sent as `reasoning_effort`; before, it was
227
+ reported as untranslated.
228
+
229
+ ### Engines for unattended model work
230
+
231
+ Unattended model work is the work akm hands a model with no one watching:
232
+ - the improve processes (reflect, distill, consolidate, memory inference,
233
+ extract, and validation's repair);
234
+ - the quality, triage and retrieval-gate judges;
235
+ - index passes;
236
+ - `akm remember --enrich`.
237
+
238
+ It runs on any engine kind under one tool policy. The model may read, edit
239
+ files only inside a scratch working directory that akm creates for the
240
+ dispatch and removes after it, and run `akm search` and `akm show`. The
241
+ stash stays read-only to it: what model work changes reaches the stash only
242
+ as a proposal, through the review queue.
243
+
244
+ Each engine enforces as much of the policy as it can, and grants nothing it
245
+ cannot enforce:
246
+
247
+ | Engine | What the model gets |
248
+ |---|---|
249
+ | LLM | No tools. |
250
+ | `claude` | Read and Edit inside the working directory, and Bash for `akm search` and `akm show` only. akm runs it with `--restricted`, so your user, project and local settings cannot widen that. |
251
+ | `opencode`, `opencode-sdk` | Read and edit inside the working directory, through an injected `akm-model-work` agent. No bash, so no `akm search` or `akm show`, because opencode cannot stop a redirect such as `akm show x > file` from writing elsewhere. |
252
+ | `codex`, `copilot`, `pi`, `gemini`, `aider`, `amazonq`, `openhands` | Cannot run model work: akm refuses the request before it starts. |
253
+
254
+ **The one rule.** Every key model work reads its engine from must name an
255
+ LLM engine or a `claude`, `opencode` or `opencode-sdk` agent engine. Those
256
+ keys are:
257
+ - `defaults.llmEngine`;
258
+ - `index.defaults.engine` and `index.<pass>.engine`;
259
+ - `improve.strategies.<name>.engine`;
260
+ - `improve.strategies.<name>.processes.<process>.engine`;
261
+ - an enabled `processes.triage.judgment.engine`;
262
+ - `processes.<process>.qualityGate.engine`.
263
+
264
+ A config that breaks the rule fails to load, with an error that names the
265
+ key, the engine and its platform. Other engine keys, `defaults.engine` and
266
+ `workflow.judgeEngine`, may name any configured engine.
267
+
268
+ **Further details:**
269
+ - A model-work dispatch builds its own command, so the engine's `args` do not
270
+ apply to it, except a `--model` they name, and neither does its `workspace`.
271
+ - The temporary directory must not be inside a git repository, because
272
+ opencode would treat the whole repository as its working directory. Point
273
+ `TMPDIR` elsewhere if it is.
274
+ - `llm` overrides that reach an agent engine are translated when its platform
275
+ translates them (see [Inference on an agent engine](#inference-on-an-agent-engine))
276
+ and reported as `untranslated-field` notices, not errors, when it does not.
277
+
135
278
  ### Model-map files
136
279
 
137
280
  AKM ships an immutable `models.json` package asset with three intent aliases:
@@ -176,7 +319,10 @@ profile may omit `model` when the installed layer already supplies it, as the
176
319
  partial Claude override above does. After overlay, every alias/engine entry
177
320
  must have a usable model. Unknown profile fields are rejected; JSON-safe
178
321
  fields inside `inference` are preserved for engine adapters to lower
179
- optimistically.
322
+ optimistically. An `inference.effort` is read as `reasoningEffort`
323
+ (see [Inference on an agent engine](#inference-on-an-agent-engine)), so the
324
+ starter's `reasoning` alias sets `--effort high` on `claude` and
325
+ `options.reasoningEffort: "high"` on `opencode` and `opencode-sdk`.
180
326
 
181
327
  A profile's `engine` field (0.9.15, #946) borrows a column's `model` (and, for
182
328
  an `llm`-kind engine, its inference defaults) from a configured
@@ -265,16 +411,24 @@ the embedded copy, and release tests pin copied bytes to `src/assets/models.json
265
411
  The health check passes when the optional user file is absent and warns with
266
412
  its path and JSON location when the user file is unreadable or invalid.
267
413
 
268
- `defaults.engine` names an LLM or agent engine. `defaults.llmEngine` must name
269
- an LLM engine. There is no first-engine fallback: an unset `defaults.engine`
414
+ `defaults.engine` names an LLM or agent engine. `defaults.llmEngine` names the
415
+ default engine for unattended model work, so it follows
416
+ [the one rule](#engines-for-unattended-model-work). There is no first-engine
417
+ fallback: an unset `defaults.engine`
270
418
  never resolves to some arbitrary entry in `engines`. It resolves instead to a
271
419
  synthesized, config-free `opencode-sdk` engine when the `opencode` binary is on
272
420
  PATH — announced once per run, and preempted by any `opencode-sdk` engine you
273
421
  configure yourself. Naming an engine that is not configured is always an error
274
422
  and is never rescued by that fallback.
275
423
 
424
+ `defaults.llmEngine` is not an `opencode-sdk` engine's fallback connection. An
425
+ SDK engine gets an LLM fallback only from its own `llmEngine`, so the
426
+ synthesized engine, which sets none, runs on opencode's own provider, model and
427
+ auth.
428
+
276
429
  Index passes select engines through `index.defaults.engine` or
277
- `index.<pass>.engine`. Per-pass `model`, `timeoutMs`, and `llm` fields are
430
+ `index.<pass>.engine`, which follow
431
+ [the one rule](#engines-for-unattended-model-work). Per-pass `model`, `timeoutMs`, and `llm` fields are
278
432
  invocation overrides; `enabled: false` disables that pass. Connection fields
279
433
  such as `endpoint`, `provider`, `apiKey`, and `apiKeyFile` belong only on
280
434
  named engines.
@@ -323,7 +477,8 @@ can select `engine`, `model`, `timeoutMs`, and LLM request overrides:
323
477
  }
324
478
  ```
325
479
 
326
- LLM-only improve processes require an LLM engine; an explicit invalid or
480
+ An improve process's engine follows
481
+ [the one rule](#engines-for-unattended-model-work); an explicit invalid or
327
482
  incompatible engine never falls back to another engine. Built-in strategies
328
483
  are complete presets. User-defined strategies inherit omitted fields from the
329
484
  built-in `default` strategy before applying their own overrides.
@@ -359,11 +514,13 @@ each process's LLM-as-judge quality gate. Each is on unless it sets
359
514
  `enabled: false`, and each follows only its own switch. A reflect revision
360
515
  that changes the body is never auto-accepted; when the judge passes it, it
361
516
  waits for review. With the gate off, it waits for review too. The judge is the
362
- process's own LLM engine, or `defaults.llmEngine` when an agent generates.
363
- `engine`, `model`, `timeoutMs` and `llm` give the gate a judge of its own,
517
+ process's own engine when that is an LLM engine, or the `defaults.llmEngine`
518
+ engine when an agent generates. `engine`, `model`, `timeoutMs` and `llm` give the gate a judge of its own,
364
519
  resolved over the process's settings the way `triage.judgment` resolves over
365
- triage's. A gate whose settings resolve to no LLM engine fails before anything
366
- is generated; it never falls back to another engine. The judge runs at
520
+ triage's. The judge may be any engine that follows
521
+ [the one rule](#engines-for-unattended-model-work). A gate whose settings
522
+ resolve to no engine fails before anything is generated; it never falls back
523
+ to another engine. The judge runs at
367
524
  temperature 0 with thinking off unless its engine sets `enableThinking: true`.
368
525
  Thinking is slow: on a 27B llama.cpp server, a thinking judgment took a median
369
526
  of 30–67 s and up to about 3 minutes, against about 5 s without.
@@ -297,13 +297,17 @@ families) plus the orchestration keys:
297
297
  - `params` — name → `{ type, description }` (JSON-Schema-typed, unlike a bare
298
298
  description string).
299
299
  - `defaults` — run-level dispatch defaults (`engine`, `model`, `llm`,
300
- `timeout`, `on_error`), overridable per unit. `defaults.llm` is the
301
- exception: `llm:` tuning applies only to engines of kind `llm`, and a
302
- document-level `llm:` reaches EVERY step, so a document that also has a step
303
- on an agent engine fails to freeze — naming the step and the engine — rather
304
- than dropping the settings for that step. There is no per-step opt-out (`llm:
305
- {}` is a no-op and `llm: null` is a parse error), so in a mixed document put
306
- `llm:` on the `unit:` of each LLM step instead of in `defaults:`.
300
+ `timeout`, `on_error`), overridable per unit. `llm:` tuning reaches an engine
301
+ of any kind, and a document-level `llm:` reaches EVERY step. An LLM engine
302
+ sends it; an agent engine translates what its platform can carry
303
+ (`temperature`, `reasoning_effort`, `enable_thinking`, and `max_tokens` with
304
+ `context_length` on `opencode` and `opencode-sdk`; `reasoning_effort` on
305
+ `claude`; see
306
+ [Inference on an agent engine](configuration.md#inference-on-an-agent-engine))
307
+ and reports the rest as an `untranslated-field` notice on that step. There is
308
+ no per-step opt-out (`llm: {}` is a no-op and `llm: null` is a parse error),
309
+ so in a mixed document put `llm:` on the `unit:` of each step it is meant for
310
+ instead of in `defaults:`.
307
311
  - `outputs` — name → `{ from, schema? }`, a run-level export projected from
308
312
  a step's own artifact (Markdown-only; see [Workflow
309
313
  outputs](#workflow-outputs) below).
@@ -1179,8 +1183,8 @@ and return HTTP 500 — a hard failure, so loopback endpoints stay at 1 and a
1179
1183
  `engines.<name>.concurrency` yourself. Remote providers fail softly (a
1180
1184
  retryable 429), and four concurrent completions is well inside any hosted
1181
1185
  provider's entry tier. Agent engines carry no concurrency limit of their own —
1182
- except an `opencode-sdk` engine with an `llmEngine` fallback, which inherits
1183
- that fallback engine's limit.
1186
+ except an `opencode-sdk` engine that sets `llmEngine`, which inherits that
1187
+ fallback engine's limit.
1184
1188
 
1185
1189
  #### What counts as a loopback endpoint
1186
1190
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "akm-cli",
3
- "version": "0.9.24",
3
+ "version": "0.9.25-alpha.1",
4
4
  "type": "module",
5
5
  "description": "akm (Agent Knowledge Manager) — a portable, local-first capability library for AI agents. Discover, load, share, and improve reusable skills, scripts, workflows, and knowledge across any shell-capable coding agent, including Claude Code, OpenCode, and Cursor.",
6
6
  "keywords": [
@@ -146,6 +146,24 @@
146
146
  "type": "string",
147
147
  "maxLength": 63,
148
148
  "pattern": "^(?!akm-)[a-z][a-z0-9]*(?:-[a-z0-9]+)*$"
149
+ },
150
+ "temperature": {
151
+ "type": "number"
152
+ },
153
+ "maxTokens": {
154
+ "type": "integer",
155
+ "exclusiveMinimum": 0
156
+ },
157
+ "contextLength": {
158
+ "type": "integer",
159
+ "exclusiveMinimum": 0
160
+ },
161
+ "enableThinking": {
162
+ "type": "boolean"
163
+ },
164
+ "reasoningEffort": {
165
+ "type": "string",
166
+ "minLength": 1
149
167
  }
150
168
  },
151
169
  "required": [
@@ -1798,6 +1816,24 @@
1798
1816
  "type": "string",
1799
1817
  "maxLength": 63,
1800
1818
  "pattern": "^(?!akm-)[a-z][a-z0-9]*(?:-[a-z0-9]+)*$"
1819
+ },
1820
+ "temperature": {
1821
+ "type": "number"
1822
+ },
1823
+ "maxTokens": {
1824
+ "type": "integer",
1825
+ "exclusiveMinimum": 0
1826
+ },
1827
+ "contextLength": {
1828
+ "type": "integer",
1829
+ "exclusiveMinimum": 0
1830
+ },
1831
+ "enableThinking": {
1832
+ "type": "boolean"
1833
+ },
1834
+ "reasoningEffort": {
1835
+ "type": "string",
1836
+ "minLength": 1
1801
1837
  }
1802
1838
  },
1803
1839
  "required": [