akm-cli 0.9.23 → 0.9.25-alpha.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +339 -0
- package/dist/commands/health/checks.js +10 -11
- package/dist/commands/improve/consolidate/pair-pass.js +3 -2
- package/dist/commands/improve/consolidate.js +3 -2
- package/dist/commands/improve/execution.js +4 -10
- package/dist/commands/improve/extract-prompt.js +4 -4
- package/dist/commands/improve/extract.js +10 -13
- package/dist/commands/improve/improve-cli.js +32 -33
- package/dist/commands/improve/improve-strategies.js +49 -43
- package/dist/commands/improve/improve-usage-report.js +8 -17
- package/dist/commands/improve/preparation.js +3 -1
- package/dist/commands/improve/reflect.js +132 -177
- package/dist/commands/improve/stage.js +69 -18
- package/dist/commands/proposal/drain.js +19 -7
- package/dist/commands/proposal/proposal-cli.js +1 -5
- package/dist/commands/proposal/propose-cli.js +2 -2
- package/dist/commands/proposal/propose.js +80 -83
- package/dist/commands/remember.js +3 -3
- package/dist/commands/sources/schema-repair.js +1 -1
- package/dist/core/config/config-schema.js +37 -60
- package/dist/core/config/engine-semantics.js +15 -11
- package/dist/core/config/schema/engines.js +33 -15
- package/dist/core/config/schema/improve-processes.js +2 -2
- package/dist/core/improve-result.js +3 -3
- package/dist/core/structured.js +10 -0
- package/dist/execution/source.js +14 -0
- package/dist/indexer/passes/memory-inference.js +2 -1
- package/dist/integrations/agent/builder-shared.js +15 -0
- package/dist/integrations/agent/config.js +2 -0
- package/dist/integrations/agent/engine-resolution.js +16 -31
- package/dist/integrations/agent/execution.js +46 -21
- package/dist/integrations/agent/index.js +1 -1
- package/dist/integrations/agent/model-map.js +16 -15
- package/dist/integrations/agent/prompts.js +53 -76
- package/dist/integrations/agent/request-lowering.js +20 -9
- package/dist/integrations/agent/runner-dispatch.js +103 -4
- package/dist/integrations/agent/runner.js +8 -2
- package/dist/integrations/harnesses/aider/agent-builder.js +11 -23
- package/dist/integrations/harnesses/aider/index.js +0 -5
- package/dist/integrations/harnesses/amazonq/agent-builder.js +10 -21
- package/dist/integrations/harnesses/amazonq/index.js +0 -5
- package/dist/integrations/harnesses/claude/agent-builder.js +61 -38
- package/dist/integrations/harnesses/claude/index.js +0 -14
- package/dist/integrations/harnesses/codex/agent-builder.js +4 -4
- package/dist/integrations/harnesses/codex/index.js +0 -4
- package/dist/integrations/harnesses/copilot/agent-builder.js +4 -13
- package/dist/integrations/harnesses/copilot/index.js +2 -7
- package/dist/integrations/harnesses/gemini/agent-builder.js +4 -13
- package/dist/integrations/harnesses/gemini/index.js +0 -5
- package/dist/integrations/harnesses/ids.js +18 -10
- package/dist/integrations/harnesses/opencode/agent-builder.js +52 -10
- package/dist/integrations/harnesses/opencode/index.js +0 -8
- package/dist/integrations/harnesses/opencode/model-config.js +80 -0
- package/dist/integrations/harnesses/opencode/model-work-agent.js +80 -0
- package/dist/integrations/harnesses/opencode-sdk/harness.js +6 -7
- package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +168 -24
- package/dist/integrations/harnesses/openhands/agent-builder.js +8 -21
- package/dist/integrations/harnesses/openhands/index.js +0 -5
- package/dist/integrations/harnesses/pi/agent-builder.js +8 -21
- package/dist/integrations/harnesses/pi/index.js +0 -5
- package/dist/llm/client.js +5 -0
- package/dist/llm/feature-gate.js +10 -4
- package/dist/llm/index-passes.js +3 -6
- package/dist/llm/memory-infer.js +6 -5
- package/dist/llm/structured-call.js +33 -11
- package/dist/scripts/akm-migrate-node.js +298 -204
- package/dist/scripts/akm-migrate.js +298 -204
- package/dist/workflows/exec/step-work.js +6 -5
- package/dist/workflows/freeze/step-values.js +1 -1
- package/docs/reference/cli.md +34 -14
- package/docs/reference/configuration.md +171 -12
- package/docs/reference/workflow-schema.md +13 -9
- package/package.json +1 -1
- package/schemas/akm-config.json +36 -0
|
@@ -13,6 +13,7 @@ import unitPreambleTemplate from "../../assets/prompts/workflow-unit-preamble.md
|
|
|
13
13
|
import { UsageError } from "../../core/errors.js";
|
|
14
14
|
import { validateJsonSchemaSubset } from "../../core/json-schema.js";
|
|
15
15
|
import { parseEmbeddedJsonResponse } from "../../core/parse.js";
|
|
16
|
+
import { withSchemaInstruction } from "../../core/structured.js";
|
|
16
17
|
import { canonicalInputJson, validateInputs } from "../../execution/input-contract.js";
|
|
17
18
|
import { withWorkflowRunsRepo, } from "../../storage/repositories/workflow-runs-repository.js";
|
|
18
19
|
import { canonicalJson } from "../ir/plan-hash.js";
|
|
@@ -248,7 +249,9 @@ function buildStepWorkUnit(ctx, unitId, item, index) {
|
|
|
248
249
|
// `inputBindings`, when non-empty — see StepWorkUnitContext.taskInputs.
|
|
249
250
|
...(taskInputs && Object.keys(taskInputs).length > 0 ? { taskInputs } : {}),
|
|
250
251
|
...(input.gateFeedback ? { gateFeedback: input.gateFeedback } : {}),
|
|
251
|
-
|
|
252
|
+
// An agent or SDK transport's request lowering appends the schema
|
|
253
|
+
// instruction itself; a direct-LLM unit carries it in its prompt.
|
|
254
|
+
...(template.schema && ctx.runner === "llm" ? { schema: template.schema } : {}),
|
|
252
255
|
instructions: template.instructions,
|
|
253
256
|
});
|
|
254
257
|
const inputHash = computeUnitInputHash(ctx, item);
|
|
@@ -373,10 +376,8 @@ export function buildUnitPrompt(input) {
|
|
|
373
376
|
? `\nUnmet criteria:\n${gateFeedback.missing.map((m) => `- ${m}`).join("\n")}`
|
|
374
377
|
: "")
|
|
375
378
|
: "";
|
|
376
|
-
const
|
|
377
|
-
|
|
378
|
-
: "";
|
|
379
|
-
return `${preamble}\n${instructions}${itemBlock}${inputsBlock}${taskInputsBlock}${gateBlock}${schemaDirective}`;
|
|
379
|
+
const prompt = `${preamble}\n${instructions}${itemBlock}${inputsBlock}${taskInputsBlock}${gateBlock}`;
|
|
380
|
+
return schema ? withSchemaInstruction(prompt, schema) : prompt;
|
|
380
381
|
}
|
|
381
382
|
/**
|
|
382
383
|
* Content-derived unit identity (module doc): `<node_id>:<sha256>` for a
|
|
@@ -33,7 +33,7 @@ export function targetConcurrency(runner, config) {
|
|
|
33
33
|
if (runner.kind !== "sdk" || !runner.fallbackConnection)
|
|
34
34
|
return undefined;
|
|
35
35
|
const selected = typeof runner.engine === "string" ? config.engines?.[runner.engine] : undefined;
|
|
36
|
-
const fallbackName = selected?.kind === "agent" ?
|
|
36
|
+
const fallbackName = selected?.kind === "agent" ? selected.llmEngine : undefined;
|
|
37
37
|
const fallback = fallbackName ? config.engines?.[fallbackName] : undefined;
|
|
38
38
|
return defaultLlmEngineConcurrency(runner.fallbackConnection.endpoint, fallback?.kind === "llm" ? fallback.concurrency : undefined);
|
|
39
39
|
}
|
package/docs/reference/cli.md
CHANGED
|
@@ -2326,7 +2326,8 @@ or inference payload was selected.
|
|
|
2326
2326
|
|
|
2327
2327
|
**Platform-specific dispatch:** akm uses a platform builder to construct the
|
|
2328
2328
|
CLI argv for each engine's harness platform. `platform: "opencode"` engines emit:
|
|
2329
|
-
`opencode run [--
|
|
2329
|
+
`opencode run [--model opencode/claude-opus-4-7] "<prompt>"`. `opencode run` has
|
|
2330
|
+
no system-prompt option, so akm composes a persona into the prompt.
|
|
2330
2331
|
`platform: "claude"` engines emit:
|
|
2331
2332
|
`claude [--system-prompt "..."] [--model claude-opus-4-7] --print -- "<prompt>"`.
|
|
2332
2333
|
Agent engines may set `bin`, `args`, `workspace`, `model`, and `timeoutMs` in
|
|
@@ -2446,7 +2447,7 @@ akm improve report --since 7d # ...aggregated over every real run start
|
|
|
2446
2447
|
| `--strategy <name>` | Override the active improve strategy (a built-in or entry under `improve.strategies`) |
|
|
2447
2448
|
| `--json-to-stdout` | Also emit the full persisted JSON result on stdout for a live run. Without this flag, stdout stays empty. Dry-runs always emit their result and are never persisted. |
|
|
2448
2449
|
| `--skip-if-locked` | If another improve run already holds the lock, skip gracefully (exit 0) instead of failing with "already running" (exit 75, `TransientError`, code `IMPROVE_LOCK_HELD` — field follow-up to #948: two legitimate `improve` invocations colliding on this lock is ordinary, retryable contention, not a broken config file). Use for high-frequency scheduled runs so they don't pile up failures while a longer run is in progress. |
|
|
2449
|
-
| `--require-engines` | Abort (exit 78, before any indexing, lock, or log side effect) if the active strategy would enable a process whose engine or credential cannot be resolved in this process's environment, OR whose endpoint fails a bounded reachability probe — the same probe `akm health`'s `default-llm-engine`/`configured-engines` checks run, once per distinct endpoint. Without this flag, improve degrades gracefully: it skips the affected processes and reports them in the result's `skippedProcesses`. Recommended alongside `--skip-if-locked` for scheduled runs, since the operator's own shell can pass config validation while a scheduler's stripped-down environment (see #953) cannot. |
|
|
2450
|
+
| `--require-engines` | Abort (exit 78, before any indexing, lock, or log side effect) if the active strategy would enable a process whose engine or credential cannot be resolved in this process's environment, OR whose endpoint fails a bounded reachability probe — the same probe `akm health`'s `default-llm-engine`/`configured-engines` checks run, once per distinct endpoint. An agent engine's check is that its binary is on PATH, and an `opencode-sdk` engine's is both its binary and, when it sets `llmEngine`, that LLM fallback's endpoint. Without this flag, improve degrades gracefully: it skips the affected processes and reports them in the result's `skippedProcesses`. Recommended alongside `--skip-if-locked` for scheduled runs, since the operator's own shell can pass config validation while a scheduler's stripped-down environment (see #953) cannot. |
|
|
2450
2451
|
| `--show-prompt` | Print the composed reflect prompt (#952) for one asset and exit — before any lock, index write, or engine dispatch. Requires a fully-qualified asset ref as the scope (`akm improve lessons/my-lesson --show-prompt`); rejected with a type or whole-bundle scope. The default output format is JSON, which carries the prompt as a `prompt` field (escaped into one line) alongside the resolved `engine`/`engineKind`; pass `--format text` to print the prompt itself, unwrapped and readable by eye. |
|
|
2451
2452
|
| `--sync` / `--no-sync` | Commit (and optionally push) the git-backed primary bundle when the run finishes. Default: on for git-backed bundles (per profile config). |
|
|
2452
2453
|
| `--push` / `--no-push` | Push after the end-of-run sync commit when writable with a remote configured. `--no-push` commits only, skipping the push. Default: per profile config (`true`). `sync.push` stays outside the autonomy gate — this is a per-run opt-out, not a default change. |
|
|
@@ -2582,7 +2583,7 @@ table: one row per improve process (`reflect`, `distill`, `consolidate`,
|
|
|
2582
2583
|
`memoryInference`, `extract`, `validation`, `triage`,
|
|
2583
2584
|
`proactiveMaintenance`), plus a `triage.judgment` row when the strategy
|
|
2584
2585
|
configures a judgment engine. Each row carries `enabled`, the resolved
|
|
2585
|
-
`engine
|
|
2586
|
+
`engine`, its `model` (when the engine has an LLM connection) and `engineKind`, this process's
|
|
2586
2587
|
own lowering `notices`, and — for reflect/distill/consolidate only —
|
|
2587
2588
|
`eligibleRefs`, the count of this run's `effectiveRefs` the process would act
|
|
2588
2589
|
on (`shouldSkipRef`'s allowedTypes/excludeRefPrefixes (reflect only)/
|
|
@@ -2607,7 +2608,9 @@ default probe-on behavior) to check whether a named engine actually answers.
|
|
|
2607
2608
|
builds the exact prompt reflect would send for one asset — the same source
|
|
2608
2609
|
resolution, runner selection, feedback/schema-hint/related-lesson/rejected-
|
|
2609
2610
|
proposal gathering `akm improve`'s live reflect step uses — and prints it
|
|
2610
|
-
without reading a credential, so it never calls an engine.
|
|
2611
|
+
without reading a credential, so it never calls an engine. An LLM engine
|
|
2612
|
+
receives the reply's JSON Schema as `response_format`, and an agent engine as
|
|
2613
|
+
an instruction that dispatch appends to this prompt. Add
|
|
2611
2614
|
`--format text` (the default JSON/yaml envelope escapes the prompt into one
|
|
2612
2615
|
line, which defeats a by-eye read) to confirm by eye that recent feedback is
|
|
2613
2616
|
framed as an unverified report to investigate (never a fact to insert
|
|
@@ -2619,7 +2622,7 @@ destination than `memory`.
|
|
|
2619
2622
|
|
|
2620
2623
|
#### improve report
|
|
2621
2624
|
|
|
2622
|
-
`akm improve report` (#944) answers "which engine did each
|
|
2625
|
+
`akm improve report` (#944) answers "which engine did each model-calling process
|
|
2623
2626
|
use this run, how much did it cost, and which enabled processes made zero
|
|
2624
2627
|
calls (and why)" without hand-written SQLite against `state.db`. It is a
|
|
2625
2628
|
`scope` value, not a subcommand — `report` is not, and will never be, a real
|
|
@@ -2632,7 +2635,7 @@ field on the result (`result_json` in `improve_runs`, and in the
|
|
|
2632
2635
|
`byProcessEngineModel` is a cross-tab of this run's own `llm_usage` events
|
|
2633
2636
|
(#576) — one row per distinct `(process, engine, model)` triple, each with
|
|
2634
2637
|
`calls`, `failures`, `promptTokens`, `completionTokens`, `totalTokens`,
|
|
2635
|
-
`reasoningTokens`, and `totalDurationMs`. `noCalls` lists every
|
|
2638
|
+
`reasoningTokens`, and `totalDurationMs`. `noCalls` lists every model-calling
|
|
2636
2639
|
process (`reflect`, `distill`, `consolidate`, `memoryInference`,
|
|
2637
2640
|
`extract`, `validation` — not `triage`/`proactiveMaintenance`,
|
|
2638
2641
|
which never make an attributable LLM call themselves) the active strategy
|
|
@@ -2706,7 +2709,7 @@ akm proposal extract --type claude --location /custom/path --session-id <id>
|
|
|
2706
2709
|
| `--dry-run` | Show candidates without queuing proposals. |
|
|
2707
2710
|
| `--force` | Re-process sessions even if they were already extracted and have no new events. Default: skip already-seen sessions. |
|
|
2708
2711
|
| `--timeout-ms <ms>` | Per-session LLM timeout in ms (default `600000`). |
|
|
2709
|
-
| `--engine <name>` | Named
|
|
2712
|
+
| `--engine <name>` | Named engine for this invocation: an LLM engine, or a `claude`, `opencode` or `opencode-sdk` agent engine. Mutually exclusive with `--strategy`. |
|
|
2710
2713
|
| `--strategy <name>` | Improve strategy supplying extract behavior and engine. Mutually exclusive with `--engine`. |
|
|
2711
2714
|
|
|
2712
2715
|
`--type` and `--auto` are mutually exclusive; one of them is required.
|
|
@@ -2721,8 +2724,10 @@ dropped — a foreground polling daemon in a one-shot CLI); the shipped
|
|
|
2721
2724
|
`core/extract.yml` cron template (`akm proposal extract --auto` on a
|
|
2722
2725
|
schedule) is the answer.
|
|
2723
2726
|
|
|
2724
|
-
Requires an
|
|
2725
|
-
`
|
|
2727
|
+
Requires an engine that can run unattended model work (an LLM engine, or a
|
|
2728
|
+
`claude`, `opencode` or `opencode-sdk` agent): pass `--engine`, select a
|
|
2729
|
+
`--strategy` whose `processes.extract.engine` is set, or configure
|
|
2730
|
+
`defaults.llmEngine`.
|
|
2726
2731
|
|
|
2727
2732
|
**Output.** `ok` means the command ran to completion — it is `true` even when
|
|
2728
2733
|
every session was skipped (an unreachable LLM engine included); it does not
|
|
@@ -2732,7 +2737,7 @@ harvest" branch on `skipReasons`, `warnings`, or `sessionsProcessed` /
|
|
|
2732
2737
|
|
|
2733
2738
|
| Field | Description |
|
|
2734
2739
|
| --- | --- |
|
|
2735
|
-
| `engine` | Resolved
|
|
2740
|
+
| `engine` | Resolved engine name for this run. Absent only when extract is disabled by the selected improve strategy (the run returns before an engine is resolved). |
|
|
2736
2741
|
| `engineKind` | `"llm"`, `"sdk"`, or `"agent"` — the kind of runner `engine` resolved to. Same absence condition as `engine`. |
|
|
2737
2742
|
| `skipReasons` | Per-`skipReason` count across `sessions[]` (e.g. `{ "llm_unavailable": 25 }`). Present only when `sessionsSkipped > 0`. |
|
|
2738
2743
|
| `warnings` | Includes one aggregate line per infrastructure skip reason that fired (`llm_unavailable`, `read_failed`, `exception`, `locked_concurrent`) — e.g. `25 of 25 sessions skipped: llm_unavailable (engine "default")` — so an engine outage is visible without inspecting `sessions[]`. Session-content skips (`already_extracted`, `too_short`, `triaged_out`) are counted in `skipReasons` but never produce a warning line. |
|
|
@@ -2755,11 +2760,24 @@ akm proposal new skill code-review --path team --task "PR-style review skill" #
|
|
|
2755
2760
|
| `--path` | Relative subdirectory under the type dir to place the proposed asset in (e.g. `release`). The filename comes from `<name>`. |
|
|
2756
2761
|
| `--task` | Inline task text |
|
|
2757
2762
|
| `--file` | Read task text from a UTF-8 file |
|
|
2758
|
-
| `--engine` | Override the default execution engine |
|
|
2763
|
+
| `--engine` | Override the default execution engine. Any kind works: an LLM engine, an agent CLI or `opencode-sdk` |
|
|
2759
2764
|
| `--timeout-ms` | Override the selected engine timeout for this call |
|
|
2760
2765
|
|
|
2761
2766
|
Exactly one of `--task` or `--file` is required. Emits `propose_invoked`.
|
|
2762
2767
|
|
|
2768
|
+
Every engine kind returns the proposal the same way: as one JSON object on
|
|
2769
|
+
stdout with the asset's `ref`, its full `content` and a self-rated
|
|
2770
|
+
`confidence`. akm sends the object's JSON Schema with the request, as
|
|
2771
|
+
`response_format` to an LLM engine and as an instruction at the end of the
|
|
2772
|
+
prompt to an agent engine (codex also gets it as `--output-schema`). akm
|
|
2773
|
+
captures the reply; an agent CLI runs headless, with no live terminal session.
|
|
2774
|
+
A harness's own JSON envelope, such as claude's
|
|
2775
|
+
`--output-format json` result, is unwrapped first.
|
|
2776
|
+
A reply that is not a valid proposal gets one corrective retry that says what
|
|
2777
|
+
was wrong. If that reply is not valid either, the command exits 1 with
|
|
2778
|
+
`reason: "parse_error"` and an error that names the engine, for example
|
|
2779
|
+
`Engine "local" reply was not valid proposal JSON after 2 attempts: …`.
|
|
2780
|
+
|
|
2763
2781
|
**Prompt-task `timeoutMs`:** a version-2 prompt task may set `timeoutMs` to
|
|
2764
2782
|
override its selected engine timeout. Set it to `null` to disable the timer, or
|
|
2765
2783
|
to a positive integer (milliseconds) to apply a task-specific limit.
|
|
@@ -3039,9 +3057,11 @@ Drain the standing pending-proposal backlog instead of adjudicating proposals
|
|
|
3039
3057
|
one at a time. One rule decides each proposal: a proposal whose quality judge
|
|
3040
3058
|
passed on its current content is accepted (unless its target changed since it
|
|
3041
3059
|
was minted — that one is auto-rejected as `stale-target`); an empty diff is
|
|
3042
|
-
rejected;
|
|
3043
|
-
|
|
3044
|
-
|
|
3060
|
+
rejected; a proposal that reflect or distill deferred for review is left for a
|
|
3061
|
+
person; everything else goes to the judgment tier when one is enabled, and
|
|
3062
|
+
is otherwise left for review. A reflect revision that changes the body is
|
|
3063
|
+
deferred for review even when its judge passes it. Default mode stages
|
|
3064
|
+
decisions (queue mode); pass `--promote` to actually accept.
|
|
3045
3065
|
|
|
3046
3066
|
```sh
|
|
3047
3067
|
akm proposal drain --dry-run # Preview without writing
|
|
@@ -116,9 +116,32 @@ settable via `extraParams`. A response with reasoning tokens despite
|
|
|
116
116
|
`enableThinking: false` triggers a runtime warning and the `akm health`
|
|
117
117
|
`thinking-control` advisory.
|
|
118
118
|
|
|
119
|
-
|
|
119
|
+
When a call asks for JSON that matches a schema, an LLM engine sends the schema
|
|
120
|
+
as `response_format` (`json_schema`, strict) unless the engine sets
|
|
121
|
+
`supportsJsonSchema: false`. If the endpoint rejects it with a 4xx other than
|
|
122
|
+
429, AKM retries once without `response_format` and stops sending it to that
|
|
123
|
+
endpoint and model for the rest of the process. An agent engine receives the
|
|
124
|
+
schema as one instruction at the end of its prompt (`Respond with ONLY a JSON
|
|
125
|
+
value matching this JSON Schema (no prose, no code fences):` followed by the
|
|
126
|
+
schema), plus the harness's own schema channel where it has one (codex
|
|
127
|
+
`--output-schema`).
|
|
128
|
+
|
|
129
|
+
An agent engine may set `bin`, `args`, `workspace`, `model`, and `timeoutMs`,
|
|
130
|
+
and the inference fields its platform translates (see
|
|
131
|
+
[Inference on an agent engine](#inference-on-an-agent-engine)).
|
|
120
132
|
Only `platform: "opencode-sdk"` may set `llmEngine`; it names
|
|
121
|
-
the LLM engine used as that SDK engine's fallback connection.
|
|
133
|
+
the LLM engine used as that SDK engine's fallback connection. With no
|
|
134
|
+
`llmEngine`, an SDK engine has no fallback connection and opencode resolves
|
|
135
|
+
provider, model and auth from its own configuration. `defaults.llmEngine` is
|
|
136
|
+
not a substitute.
|
|
137
|
+
|
|
138
|
+
On an engine without `timeoutMs`, model work (an improve process, a quality or
|
|
139
|
+
triage judge, or an index pass) stops after 600 seconds, whatever the engine's
|
|
140
|
+
kind; other work on an agent engine runs until it finishes. An improve stage's
|
|
141
|
+
reply that does not match the stage's JSON Schema gets one corrective retry
|
|
142
|
+
before the stage reads it. Reflect holds its reply to its own contract the same
|
|
143
|
+
way on every engine kind: a reply that is not the JSON object of reflect's
|
|
144
|
+
schema gets one repair turn, and one still invalid fails with `parse_error`.
|
|
122
145
|
|
|
123
146
|
Executable assets may request tools, but the request is not authority. Configure
|
|
124
147
|
the host-local `execution.allowedTools` list to define the ceiling; `"*"` is an
|
|
@@ -132,6 +155,126 @@ client with no dependencies — it spawns `opencode serve` and talks to it — s
|
|
|
132
155
|
the npm dependency alone does not make the platform usable. Install the binary
|
|
133
156
|
with `npm i -g opencode-ai` or opencode's own installer.
|
|
134
157
|
|
|
158
|
+
### Inference on an agent engine
|
|
159
|
+
|
|
160
|
+
A request's inference (`temperature`, `maxTokens`, `contextLength`,
|
|
161
|
+
`enableThinking`, `reasoningEffort`) reaches an agent engine's harness from
|
|
162
|
+
every place it can come from: the engine's own settings, an improve process's
|
|
163
|
+
`llm` overlay, a task, command or agent asset's `inference`, a workflow's
|
|
164
|
+
`llm:`, and a `models.json` alias. The nearest layer wins, field by field, as
|
|
165
|
+
for an LLM engine. With no setting anywhere akm sends nothing of its own, so
|
|
166
|
+
the model's own default applies: a `reasoningEffort: "none"` that your opencode
|
|
167
|
+
config sets on a model stays in force until a layer overrides it.
|
|
168
|
+
|
|
169
|
+
Each platform translates what it can carry:
|
|
170
|
+
|
|
171
|
+
| Platform | Translates | How |
|
|
172
|
+
|---|---|---|
|
|
173
|
+
| `claude` | `reasoningEffort` | `--effort <level>`, in the harness's own levels (`low`, `medium`, `high`, `xhigh`, `max` in Claude Code 2.1.283). The value is passed as given, so one Claude Code rejects fails the dispatch. |
|
|
174
|
+
| `opencode`, `opencode-sdk` | `temperature`, `reasoningEffort`, `enableThinking`, and `maxTokens` with `contextLength` | Injected opencode config, below. |
|
|
175
|
+
| every other platform | nothing | |
|
|
176
|
+
|
|
177
|
+
An agent engine that sets a field its platform does not translate fails to
|
|
178
|
+
load, with an error that names the platform and the fields it does translate.
|
|
179
|
+
Inference that reaches the engine from an asset or a caller, and that the
|
|
180
|
+
platform does not translate, is reported as an `untranslated-field` notice and
|
|
181
|
+
dispatch continues, because the same asset may run on any engine.
|
|
182
|
+
|
|
183
|
+
On `opencode` and `opencode-sdk`, inference goes into opencode's config for the
|
|
184
|
+
model the dispatch names: the request's model, or the `--model` an `opencode`
|
|
185
|
+
engine's `args` name. That must be a `provider/model`, except for an
|
|
186
|
+
`opencode-sdk` engine with an `llmEngine` fallback, which routes the
|
|
187
|
+
fallback's model through its own provider. Without a model to attach to the
|
|
188
|
+
fields are reported as untranslated, except that model work's agent (below)
|
|
189
|
+
carries its options whichever model opencode picks, so an `opencode-sdk` engine
|
|
190
|
+
with no `model` and no `llmEngine` still gets them. The injected entry merges
|
|
191
|
+
over your own opencode config for the same provider and model, so what the
|
|
192
|
+
request does not set stays as you wrote it. A dispatch with no inference injects
|
|
193
|
+
nothing. akm
|
|
194
|
+
checked these shapes against opencode 1.18.25 and an OpenAI-compatible provider
|
|
195
|
+
(`@ai-sdk/openai-compatible`); another provider package is given the same
|
|
196
|
+
options and may ignore one.
|
|
197
|
+
|
|
198
|
+
- `temperature` becomes `options.temperature`, and `reasoningEffort` becomes
|
|
199
|
+
`options.reasoningEffort`. opencode reads the camelCase name: its
|
|
200
|
+
`reasoning_effort` spelling is dropped, so a model option written that way in
|
|
201
|
+
`opencode.jsonc` does nothing.
|
|
202
|
+
- `enableThinking` becomes `options.chat_template_kwargs.enable_thinking` and
|
|
203
|
+
`options.enable_thinking`, the two forms an LLM engine sends.
|
|
204
|
+
- `maxTokens` becomes `limit.output` and `contextLength` becomes
|
|
205
|
+
`limit.context`. Set them together. Without `limit.output` opencode asks for
|
|
206
|
+
`max_tokens: 32000`, which a small-context server rejects, and a wrong
|
|
207
|
+
`limit.context` lets opencode build requests longer than the server's window.
|
|
208
|
+
opencode refuses a `limit` with only one of them, and a half would overwrite
|
|
209
|
+
the other half of a limit you declared for the model, so akm declares `limit`
|
|
210
|
+
only when it has both and reports a lone `maxTokens` or `contextLength` as
|
|
211
|
+
untranslated.
|
|
212
|
+
- For model work (below) the options go on the `akm-model-work` agent that runs
|
|
213
|
+
the dispatch, so opencode's own call on the same model, a title for the
|
|
214
|
+
session, keeps the model's defaults. Any other dispatch runs your own agent,
|
|
215
|
+
whose name akm cannot rely on, so the options go on the model and the title
|
|
216
|
+
call sees them too. `opencode-sdk` names its session, so it makes no title
|
|
217
|
+
call.
|
|
218
|
+
- `opencode-sdk` declares the model of its `llmEngine` fallback with the
|
|
219
|
+
fallback's own inference, under the request's, field by field. Each distinct
|
|
220
|
+
set of inference starts its own `opencode serve`, as a different model does.
|
|
221
|
+
|
|
222
|
+
Reasoning effort has one word in a request, `reasoningEffort`. `effort`, as a
|
|
223
|
+
`models.json` alias or an asset's `effort:` frontmatter spells it, is the same
|
|
224
|
+
setting and is read as `reasoningEffort` wherever layers are merged, so the
|
|
225
|
+
nearest layer wins whichever word it used. On an LLM engine an alias's or
|
|
226
|
+
asset's `effort` is therefore sent as `reasoning_effort`; before, it was
|
|
227
|
+
reported as untranslated.
|
|
228
|
+
|
|
229
|
+
### Engines for unattended model work
|
|
230
|
+
|
|
231
|
+
Unattended model work is the work akm hands a model with no one watching:
|
|
232
|
+
- the improve processes (reflect, distill, consolidate, memory inference,
|
|
233
|
+
extract, and validation's repair);
|
|
234
|
+
- the quality, triage and retrieval-gate judges;
|
|
235
|
+
- index passes;
|
|
236
|
+
- `akm remember --enrich`.
|
|
237
|
+
|
|
238
|
+
It runs on any engine kind under one tool policy. The model may read, edit
|
|
239
|
+
files only inside a scratch working directory that akm creates for the
|
|
240
|
+
dispatch and removes after it, and run `akm search` and `akm show`. The
|
|
241
|
+
stash stays read-only to it: what model work changes reaches the stash only
|
|
242
|
+
as a proposal, through the review queue.
|
|
243
|
+
|
|
244
|
+
Each engine enforces as much of the policy as it can, and grants nothing it
|
|
245
|
+
cannot enforce:
|
|
246
|
+
|
|
247
|
+
| Engine | What the model gets |
|
|
248
|
+
|---|---|
|
|
249
|
+
| LLM | No tools. |
|
|
250
|
+
| `claude` | Read and Edit inside the working directory, and Bash for `akm search` and `akm show` only. akm runs it with `--restricted`, so your user, project and local settings cannot widen that. |
|
|
251
|
+
| `opencode`, `opencode-sdk` | Read and edit inside the working directory, through an injected `akm-model-work` agent. No bash, so no `akm search` or `akm show`, because opencode cannot stop a redirect such as `akm show x > file` from writing elsewhere. |
|
|
252
|
+
| `codex`, `copilot`, `pi`, `gemini`, `aider`, `amazonq`, `openhands` | Cannot run model work: akm refuses the request before it starts. |
|
|
253
|
+
|
|
254
|
+
**The one rule.** Every key model work reads its engine from must name an
|
|
255
|
+
LLM engine or a `claude`, `opencode` or `opencode-sdk` agent engine. Those
|
|
256
|
+
keys are:
|
|
257
|
+
- `defaults.llmEngine`;
|
|
258
|
+
- `index.defaults.engine` and `index.<pass>.engine`;
|
|
259
|
+
- `improve.strategies.<name>.engine`;
|
|
260
|
+
- `improve.strategies.<name>.processes.<process>.engine`;
|
|
261
|
+
- an enabled `processes.triage.judgment.engine`;
|
|
262
|
+
- `processes.<process>.qualityGate.engine`.
|
|
263
|
+
|
|
264
|
+
A config that breaks the rule fails to load, with an error that names the
|
|
265
|
+
key, the engine and its platform. Other engine keys, `defaults.engine` and
|
|
266
|
+
`workflow.judgeEngine`, may name any configured engine.
|
|
267
|
+
|
|
268
|
+
**Further details:**
|
|
269
|
+
- A model-work dispatch builds its own command, so the engine's `args` do not
|
|
270
|
+
apply to it, except a `--model` they name, and neither does its `workspace`.
|
|
271
|
+
- The temporary directory must not be inside a git repository, because
|
|
272
|
+
opencode would treat the whole repository as its working directory. Point
|
|
273
|
+
`TMPDIR` elsewhere if it is.
|
|
274
|
+
- `llm` overrides that reach an agent engine are translated when its platform
|
|
275
|
+
translates them (see [Inference on an agent engine](#inference-on-an-agent-engine))
|
|
276
|
+
and reported as `untranslated-field` notices, not errors, when it does not.
|
|
277
|
+
|
|
135
278
|
### Model-map files
|
|
136
279
|
|
|
137
280
|
AKM ships an immutable `models.json` package asset with three intent aliases:
|
|
@@ -176,7 +319,10 @@ profile may omit `model` when the installed layer already supplies it, as the
|
|
|
176
319
|
partial Claude override above does. After overlay, every alias/engine entry
|
|
177
320
|
must have a usable model. Unknown profile fields are rejected; JSON-safe
|
|
178
321
|
fields inside `inference` are preserved for engine adapters to lower
|
|
179
|
-
optimistically.
|
|
322
|
+
optimistically. An `inference.effort` is read as `reasoningEffort`
|
|
323
|
+
(see [Inference on an agent engine](#inference-on-an-agent-engine)), so the
|
|
324
|
+
starter's `reasoning` alias sets `--effort high` on `claude` and
|
|
325
|
+
`options.reasoningEffort: "high"` on `opencode` and `opencode-sdk`.
|
|
180
326
|
|
|
181
327
|
A profile's `engine` field (0.9.15, #946) borrows a column's `model` (and, for
|
|
182
328
|
an `llm`-kind engine, its inference defaults) from a configured
|
|
@@ -265,16 +411,24 @@ the embedded copy, and release tests pin copied bytes to `src/assets/models.json
|
|
|
265
411
|
The health check passes when the optional user file is absent and warns with
|
|
266
412
|
its path and JSON location when the user file is unreadable or invalid.
|
|
267
413
|
|
|
268
|
-
`defaults.engine` names an LLM or agent engine. `defaults.llmEngine`
|
|
269
|
-
|
|
414
|
+
`defaults.engine` names an LLM or agent engine. `defaults.llmEngine` names the
|
|
415
|
+
default engine for unattended model work, so it follows
|
|
416
|
+
[the one rule](#engines-for-unattended-model-work). There is no first-engine
|
|
417
|
+
fallback: an unset `defaults.engine`
|
|
270
418
|
never resolves to some arbitrary entry in `engines`. It resolves instead to a
|
|
271
419
|
synthesized, config-free `opencode-sdk` engine when the `opencode` binary is on
|
|
272
420
|
PATH — announced once per run, and preempted by any `opencode-sdk` engine you
|
|
273
421
|
configure yourself. Naming an engine that is not configured is always an error
|
|
274
422
|
and is never rescued by that fallback.
|
|
275
423
|
|
|
424
|
+
`defaults.llmEngine` is not an `opencode-sdk` engine's fallback connection. An
|
|
425
|
+
SDK engine gets an LLM fallback only from its own `llmEngine`, so the
|
|
426
|
+
synthesized engine, which sets none, runs on opencode's own provider, model and
|
|
427
|
+
auth.
|
|
428
|
+
|
|
276
429
|
Index passes select engines through `index.defaults.engine` or
|
|
277
|
-
`index.<pass>.engine
|
|
430
|
+
`index.<pass>.engine`, which follow
|
|
431
|
+
[the one rule](#engines-for-unattended-model-work). Per-pass `model`, `timeoutMs`, and `llm` fields are
|
|
278
432
|
invocation overrides; `enabled: false` disables that pass. Connection fields
|
|
279
433
|
such as `endpoint`, `provider`, `apiKey`, and `apiKeyFile` belong only on
|
|
280
434
|
named engines.
|
|
@@ -323,7 +477,8 @@ can select `engine`, `model`, `timeoutMs`, and LLM request overrides:
|
|
|
323
477
|
}
|
|
324
478
|
```
|
|
325
479
|
|
|
326
|
-
|
|
480
|
+
An improve process's engine follows
|
|
481
|
+
[the one rule](#engines-for-unattended-model-work); an explicit invalid or
|
|
327
482
|
incompatible engine never falls back to another engine. Built-in strategies
|
|
328
483
|
are complete presets. User-defined strategies inherit omitted fields from the
|
|
329
484
|
built-in `default` strategy before applying their own overrides.
|
|
@@ -356,12 +511,16 @@ guidance. When enabled, engine selection is judgment → triage → strategy →
|
|
|
356
511
|
|
|
357
512
|
`processes.reflect.qualityGate` and `processes.distill.qualityGate` control
|
|
358
513
|
each process's LLM-as-judge quality gate. Each is on unless it sets
|
|
359
|
-
`enabled: false`, and each follows only its own switch.
|
|
360
|
-
|
|
361
|
-
|
|
514
|
+
`enabled: false`, and each follows only its own switch. A reflect revision
|
|
515
|
+
that changes the body is never auto-accepted; when the judge passes it, it
|
|
516
|
+
waits for review. With the gate off, it waits for review too. The judge is the
|
|
517
|
+
process's own engine when that is an LLM engine, or the `defaults.llmEngine`
|
|
518
|
+
engine when an agent generates. `engine`, `model`, `timeoutMs` and `llm` give the gate a judge of its own,
|
|
362
519
|
resolved over the process's settings the way `triage.judgment` resolves over
|
|
363
|
-
triage's.
|
|
364
|
-
|
|
520
|
+
triage's. The judge may be any engine that follows
|
|
521
|
+
[the one rule](#engines-for-unattended-model-work). A gate whose settings
|
|
522
|
+
resolve to no engine fails before anything is generated; it never falls back
|
|
523
|
+
to another engine. The judge runs at
|
|
365
524
|
temperature 0 with thinking off unless its engine sets `enableThinking: true`.
|
|
366
525
|
Thinking is slow: on a 27B llama.cpp server, a thinking judgment took a median
|
|
367
526
|
of 30–67 s and up to about 3 minutes, against about 5 s without.
|
|
@@ -297,13 +297,17 @@ families) plus the orchestration keys:
|
|
|
297
297
|
- `params` — name → `{ type, description }` (JSON-Schema-typed, unlike a bare
|
|
298
298
|
description string).
|
|
299
299
|
- `defaults` — run-level dispatch defaults (`engine`, `model`, `llm`,
|
|
300
|
-
`timeout`, `on_error`), overridable per unit. `
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
300
|
+
`timeout`, `on_error`), overridable per unit. `llm:` tuning reaches an engine
|
|
301
|
+
of any kind, and a document-level `llm:` reaches EVERY step. An LLM engine
|
|
302
|
+
sends it; an agent engine translates what its platform can carry
|
|
303
|
+
(`temperature`, `reasoning_effort`, `enable_thinking`, and `max_tokens` with
|
|
304
|
+
`context_length` on `opencode` and `opencode-sdk`; `reasoning_effort` on
|
|
305
|
+
`claude`; see
|
|
306
|
+
[Inference on an agent engine](configuration.md#inference-on-an-agent-engine))
|
|
307
|
+
and reports the rest as an `untranslated-field` notice on that step. There is
|
|
308
|
+
no per-step opt-out (`llm: {}` is a no-op and `llm: null` is a parse error),
|
|
309
|
+
so in a mixed document put `llm:` on the `unit:` of each step it is meant for
|
|
310
|
+
instead of in `defaults:`.
|
|
307
311
|
- `outputs` — name → `{ from, schema? }`, a run-level export projected from
|
|
308
312
|
a step's own artifact (Markdown-only; see [Workflow
|
|
309
313
|
outputs](#workflow-outputs) below).
|
|
@@ -1179,8 +1183,8 @@ and return HTTP 500 — a hard failure, so loopback endpoints stay at 1 and a
|
|
|
1179
1183
|
`engines.<name>.concurrency` yourself. Remote providers fail softly (a
|
|
1180
1184
|
retryable 429), and four concurrent completions is well inside any hosted
|
|
1181
1185
|
provider's entry tier. Agent engines carry no concurrency limit of their own —
|
|
1182
|
-
except an `opencode-sdk` engine
|
|
1183
|
-
|
|
1186
|
+
except an `opencode-sdk` engine that sets `llmEngine`, which inherits that
|
|
1187
|
+
fallback engine's limit.
|
|
1184
1188
|
|
|
1185
1189
|
#### What counts as a loopback endpoint
|
|
1186
1190
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "akm-cli",
|
|
3
|
-
"version": "0.9.
|
|
3
|
+
"version": "0.9.25-alpha.1",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "akm (Agent Knowledge Manager) — a portable, local-first capability library for AI agents. Discover, load, share, and improve reusable skills, scripts, workflows, and knowledge across any shell-capable coding agent, including Claude Code, OpenCode, and Cursor.",
|
|
6
6
|
"keywords": [
|
package/schemas/akm-config.json
CHANGED
|
@@ -146,6 +146,24 @@
|
|
|
146
146
|
"type": "string",
|
|
147
147
|
"maxLength": 63,
|
|
148
148
|
"pattern": "^(?!akm-)[a-z][a-z0-9]*(?:-[a-z0-9]+)*$"
|
|
149
|
+
},
|
|
150
|
+
"temperature": {
|
|
151
|
+
"type": "number"
|
|
152
|
+
},
|
|
153
|
+
"maxTokens": {
|
|
154
|
+
"type": "integer",
|
|
155
|
+
"exclusiveMinimum": 0
|
|
156
|
+
},
|
|
157
|
+
"contextLength": {
|
|
158
|
+
"type": "integer",
|
|
159
|
+
"exclusiveMinimum": 0
|
|
160
|
+
},
|
|
161
|
+
"enableThinking": {
|
|
162
|
+
"type": "boolean"
|
|
163
|
+
},
|
|
164
|
+
"reasoningEffort": {
|
|
165
|
+
"type": "string",
|
|
166
|
+
"minLength": 1
|
|
149
167
|
}
|
|
150
168
|
},
|
|
151
169
|
"required": [
|
|
@@ -1798,6 +1816,24 @@
|
|
|
1798
1816
|
"type": "string",
|
|
1799
1817
|
"maxLength": 63,
|
|
1800
1818
|
"pattern": "^(?!akm-)[a-z][a-z0-9]*(?:-[a-z0-9]+)*$"
|
|
1819
|
+
},
|
|
1820
|
+
"temperature": {
|
|
1821
|
+
"type": "number"
|
|
1822
|
+
},
|
|
1823
|
+
"maxTokens": {
|
|
1824
|
+
"type": "integer",
|
|
1825
|
+
"exclusiveMinimum": 0
|
|
1826
|
+
},
|
|
1827
|
+
"contextLength": {
|
|
1828
|
+
"type": "integer",
|
|
1829
|
+
"exclusiveMinimum": 0
|
|
1830
|
+
},
|
|
1831
|
+
"enableThinking": {
|
|
1832
|
+
"type": "boolean"
|
|
1833
|
+
},
|
|
1834
|
+
"reasoningEffort": {
|
|
1835
|
+
"type": "string",
|
|
1836
|
+
"minLength": 1
|
|
1801
1837
|
}
|
|
1802
1838
|
},
|
|
1803
1839
|
"required": [
|