akm-cli 0.9.24 → 0.9.25-alpha.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +294 -0
- package/dist/commands/health/checks.js +10 -11
- package/dist/commands/improve/consolidate/pair-pass.js +3 -2
- package/dist/commands/improve/consolidate.js +3 -2
- package/dist/commands/improve/execution.js +4 -10
- package/dist/commands/improve/extract-prompt.js +4 -4
- package/dist/commands/improve/extract.js +10 -13
- package/dist/commands/improve/improve-cli.js +32 -33
- package/dist/commands/improve/improve-strategies.js +49 -43
- package/dist/commands/improve/improve-usage-report.js +8 -17
- package/dist/commands/improve/preparation.js +3 -1
- package/dist/commands/improve/reflect.js +108 -172
- package/dist/commands/improve/stage.js +69 -18
- package/dist/commands/proposal/drain.js +14 -2
- package/dist/commands/proposal/proposal-cli.js +1 -5
- package/dist/commands/proposal/propose-cli.js +2 -2
- package/dist/commands/proposal/propose.js +80 -83
- package/dist/commands/remember.js +3 -3
- package/dist/commands/sources/schema-repair.js +1 -1
- package/dist/core/config/config-schema.js +37 -60
- package/dist/core/config/engine-semantics.js +15 -11
- package/dist/core/config/schema/engines.js +33 -15
- package/dist/core/config/schema/improve-processes.js +2 -2
- package/dist/core/improve-result.js +3 -3
- package/dist/core/structured.js +10 -0
- package/dist/execution/source.js +14 -0
- package/dist/indexer/passes/memory-inference.js +2 -1
- package/dist/integrations/agent/builder-shared.js +15 -0
- package/dist/integrations/agent/config.js +2 -0
- package/dist/integrations/agent/engine-resolution.js +16 -31
- package/dist/integrations/agent/execution.js +46 -21
- package/dist/integrations/agent/index.js +1 -1
- package/dist/integrations/agent/model-map.js +16 -15
- package/dist/integrations/agent/prompts.js +53 -76
- package/dist/integrations/agent/request-lowering.js +20 -9
- package/dist/integrations/agent/runner-dispatch.js +103 -4
- package/dist/integrations/agent/runner.js +8 -2
- package/dist/integrations/harnesses/aider/agent-builder.js +11 -23
- package/dist/integrations/harnesses/aider/index.js +0 -5
- package/dist/integrations/harnesses/amazonq/agent-builder.js +10 -21
- package/dist/integrations/harnesses/amazonq/index.js +0 -5
- package/dist/integrations/harnesses/claude/agent-builder.js +61 -38
- package/dist/integrations/harnesses/claude/index.js +0 -14
- package/dist/integrations/harnesses/codex/agent-builder.js +4 -4
- package/dist/integrations/harnesses/codex/index.js +0 -4
- package/dist/integrations/harnesses/copilot/agent-builder.js +4 -13
- package/dist/integrations/harnesses/copilot/index.js +2 -7
- package/dist/integrations/harnesses/gemini/agent-builder.js +4 -13
- package/dist/integrations/harnesses/gemini/index.js +0 -5
- package/dist/integrations/harnesses/ids.js +18 -10
- package/dist/integrations/harnesses/opencode/agent-builder.js +52 -10
- package/dist/integrations/harnesses/opencode/index.js +0 -8
- package/dist/integrations/harnesses/opencode/model-config.js +80 -0
- package/dist/integrations/harnesses/opencode/model-work-agent.js +80 -0
- package/dist/integrations/harnesses/opencode-sdk/harness.js +6 -7
- package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +168 -24
- package/dist/integrations/harnesses/openhands/agent-builder.js +8 -21
- package/dist/integrations/harnesses/openhands/index.js +0 -5
- package/dist/integrations/harnesses/pi/agent-builder.js +8 -21
- package/dist/integrations/harnesses/pi/index.js +0 -5
- package/dist/llm/client.js +5 -0
- package/dist/llm/feature-gate.js +10 -4
- package/dist/llm/index-passes.js +3 -6
- package/dist/llm/memory-infer.js +6 -5
- package/dist/llm/structured-call.js +33 -11
- package/dist/scripts/akm-migrate-node.js +298 -204
- package/dist/scripts/akm-migrate.js +298 -204
- package/dist/workflows/exec/step-work.js +6 -5
- package/dist/workflows/freeze/step-values.js +1 -1
- package/docs/reference/cli.md +29 -11
- package/docs/reference/configuration.md +168 -11
- package/docs/reference/workflow-schema.md +13 -9
- package/package.json +1 -1
- package/schemas/akm-config.json +36 -0
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,300 @@ All notable changes to this project will be documented in this file.
|
|
|
4
4
|
|
|
5
5
|
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
|
|
6
6
|
|
|
7
|
+
## [Unreleased]
|
|
8
|
+
|
|
9
|
+
## [0.9.25-alpha.1] - 2026-10-02
|
|
10
|
+
|
|
11
|
+
### Changed
|
|
12
|
+
|
|
13
|
+
- **One schema instruction, appended by the shared agent request lowering.**
|
|
14
|
+
Seven harness builders and the workflow engine each kept a copy of
|
|
15
|
+
"Respond with ONLY a JSON value matching this JSON Schema (no prose, no code
|
|
16
|
+
fences)". The lowering now appends it to every agent engine's prompt when a
|
|
17
|
+
schema is requested, so `codex` gets it beside `--output-schema` outside
|
|
18
|
+
workflows too. A workflow unit sends the same prompt bytes as before, less
|
|
19
|
+
the duplicate fixed below; a direct-LLM unit still carries the instruction
|
|
20
|
+
in its own prompt.
|
|
21
|
+
- **Model work is bounded at 600 seconds on every engine kind.** An improve
|
|
22
|
+
process, a quality or triage judge, or an index pass whose engine sets no
|
|
23
|
+
`timeoutMs` now stops after 600 seconds. Before, an agent or `opencode-sdk`
|
|
24
|
+
engine ran until it finished, and so did memory inference and consolidation
|
|
25
|
+
on an LLM engine. An engine's own `timeoutMs` still applies.
|
|
26
|
+
- **An improve stage's reply that fails its JSON Schema gets one corrective
|
|
27
|
+
retry.** The stage retries once with the validation errors, then reads the
|
|
28
|
+
last reply with its own parser as before. `akm extract` now allows its
|
|
29
|
+
corrective retry on every engine, not only on one without JSON Schema
|
|
30
|
+
support. Reflect keeps its own repair turn.
|
|
31
|
+
- **Agent and `opencode-sdk` dispatches leave a usage record.** Each one now
|
|
32
|
+
writes one `llm_usage` record through the same sink and stage attribution as
|
|
33
|
+
the LLM path, with the request's model and the tokens the runner reports. An
|
|
34
|
+
LLM engine still records each HTTP attempt.
|
|
35
|
+
- **`akm proposal new` gets the proposal as JSON from every engine kind, with
|
|
36
|
+
no live session.** It told every engine to write the asset to a draft file
|
|
37
|
+
and print no JSON, then parsed stdout as JSON. An LLM engine that followed
|
|
38
|
+
the instruction could not succeed, and an agent engine succeeded only by
|
|
39
|
+
writing the file: an agent CLI ran in a live terminal session whose output
|
|
40
|
+
akm could not read, and otherwise failed with an "interactive mode" error.
|
|
41
|
+
Every engine now returns the proposal as one JSON object on stdout, and the
|
|
42
|
+
request carries its JSON Schema: as `response_format` for an LLM engine, as
|
|
43
|
+
the schema instruction for an agent engine. akm captures the reply, unwraps
|
|
44
|
+
a harness envelope such as claude's `--output-format json` result, and
|
|
45
|
+
validates it. A reply that is not a proposal gets one corrective retry, then
|
|
46
|
+
fails with an error that names the engine. An agent CLI now runs headless:
|
|
47
|
+
you see the queued proposal, not the agent at work, and no draft file is
|
|
48
|
+
written.
|
|
49
|
+
- **Unattended model work has one tool policy, which each engine confines or
|
|
50
|
+
refuses when the request is built.** The policy allows reading, editing only
|
|
51
|
+
inside a scratch working directory that akm creates for the dispatch and
|
|
52
|
+
removes after it, and running `akm search` and `akm show`. The stash stays
|
|
53
|
+
read-only. An LLM engine has no tools. `claude` runs it with `--restricted`
|
|
54
|
+
(its user, project and local settings, whose allow rules could pre-approve
|
|
55
|
+
any command or path, are ignored, and its file tools stay in the working
|
|
56
|
+
directory), `--strict-mcp-config`, `--tools Read,Edit,Bash`, `--allowedTools`
|
|
57
|
+
for Read, Edit and the two `akm` commands, and `--permission-mode dontAsk`.
|
|
58
|
+
`opencode` and `opencode-sdk` run an injected `akm-model-work` agent that
|
|
59
|
+
can read and edit only inside the working directory and has no bash:
|
|
60
|
+
opencode checks a bash rule against the command's words only, so
|
|
61
|
+
`akm show x > ~/stash/asset.md` would pass an `akm show *` rule. The agent
|
|
62
|
+
has its own short prompt in place of opencode's coding-assistant prompt, an
|
|
63
|
+
8-step limit, and no automatic compaction. opencode only asks the model to
|
|
64
|
+
stop at that limit, so `opencode-sdk` aborts the session two steps past it,
|
|
65
|
+
and aborts a session the dispatch times out on; on `opencode` the dispatch
|
|
66
|
+
timeout is the bound. Every other harness refuses the policy. Such a
|
|
67
|
+
dispatch ignores the engine's `args` and `workspace`, refuses to start when
|
|
68
|
+
the temporary directory is inside a git repository, which opencode would
|
|
69
|
+
treat as its working directory, and fails with `parse_error` when the agent
|
|
70
|
+
ends with no answer, as opencode can at its step limit.
|
|
71
|
+
- **Unattended model work runs on any engine that confines the model-work
|
|
72
|
+
tool policy, and config checks that with one rule.**
|
|
73
|
+
- **Who sends the policy.** The improve processes, the quality, triage and
|
|
74
|
+
retrieval-gate judges, index passes and `akm remember --enrich` now send
|
|
75
|
+
it, so they may run on an LLM engine or on a `claude`, `opencode` or
|
|
76
|
+
`opencode-sdk` agent engine, where they could only use an LLM engine
|
|
77
|
+
before.
|
|
78
|
+
- **The one rule.** Every key model work reads its engine from must name
|
|
79
|
+
such an engine. The keys are `defaults.llmEngine`, `index.defaults.engine`,
|
|
80
|
+
`index.<pass>.engine`, a strategy's `engine`, a process's `engine`, an
|
|
81
|
+
enabled triage `judgment.engine`, and a `qualityGate.engine`. A config
|
|
82
|
+
that breaks the rule fails to load, naming the key, the engine and its
|
|
83
|
+
platform. The rule replaces six checks that each required an LLM engine
|
|
84
|
+
for some of those keys, or let the triage judgment use any agent.
|
|
85
|
+
- **What else is gone.** The plan, reflect, the quality gate and index
|
|
86
|
+
passes no longer turn an agent engine away. A triage judgment on an agent
|
|
87
|
+
engine takes a strategy's `llm` overrides as `untranslated-field`
|
|
88
|
+
notices, where it was refused before.
|
|
89
|
+
- **Reflect on an agent engine.** The agent returns its proposal as JSON on
|
|
90
|
+
stdout instead of writing a draft file, which the policy's scratch working
|
|
91
|
+
directory would not keep.
|
|
92
|
+
- **The answer is unwrapped.** A model-work reply from an agent engine goes
|
|
93
|
+
through its harness's result extractor, so a stage call on `claude` gets
|
|
94
|
+
the answer, not the `--output-format json` envelope around it.
|
|
95
|
+
- **`--require-engines` checks agent engines too.** It checks an agent
|
|
96
|
+
engine by its binary on PATH, and an `opencode-sdk` engine by its binary
|
|
97
|
+
and its LLM fallback's endpoint. It probed only LLM connections before.
|
|
98
|
+
The usage report and `akm health` count a process's calls whatever its
|
|
99
|
+
engine's kind.
|
|
100
|
+
- **An `opencode-sdk` engine gets its LLM fallback connection only from its own
|
|
101
|
+
`llmEngine`. `defaults.llmEngine` no longer supplies one.** An SDK engine
|
|
102
|
+
that set no `llmEngine` used to borrow `defaults.llmEngine`: its connection,
|
|
103
|
+
its model (unless the engine set its own) and its timeout. `defaults.llmEngine`
|
|
104
|
+
names the default engine for unattended model work, and since that work may
|
|
105
|
+
now run on an agent engine it can be one too, so it is no longer also a
|
|
106
|
+
connection that every SDK engine shares. **If you relied on the inheritance,
|
|
107
|
+
set `llmEngine` on the SDK engine**, for example
|
|
108
|
+
`"sdk": { "kind": "agent", "platform": "opencode-sdk", "llmEngine": "fast" }`.
|
|
109
|
+
Without one, the SDK engine runs on opencode's own provider and auth, and on
|
|
110
|
+
its own `model` if it sets one: akm sends it no connection. An SDK engine
|
|
111
|
+
that sets `llmEngine` is unchanged, and so is a config with no `opencode-sdk`
|
|
112
|
+
engine. The rule holds everywhere the fallback is read: dispatch, a
|
|
113
|
+
workflow's frozen concurrency cap, `akm health`, and
|
|
114
|
+
`akm improve --require-engines`, which now check an SDK engine's fallback
|
|
115
|
+
endpoint only when it sets `llmEngine`.
|
|
116
|
+
- **Inference reaches `opencode`, `opencode-sdk` and `claude` engines.** A
|
|
117
|
+
request's `temperature`, `maxTokens`, `contextLength`, `enableThinking` and
|
|
118
|
+
`reasoningEffort` came from the engine's own settings, an improve process's
|
|
119
|
+
`llm` overlay, a task, command or agent asset's `inference`, a workflow's
|
|
120
|
+
`llm:` and a `models.json` alias, and every agent engine dropped all of them
|
|
121
|
+
with an `untranslated-field` notice. Each platform now translates what it can
|
|
122
|
+
carry, and the nearest layer wins, field by field, as on an LLM engine. With
|
|
123
|
+
no setting anywhere akm sends nothing of its own, so the model's configured
|
|
124
|
+
default still applies, such as a `reasoningEffort: "none"` in your opencode
|
|
125
|
+
config.
|
|
126
|
+
- **`claude`** gets `reasoningEffort` as `--effort <level>`, passed as given
|
|
127
|
+
(Claude Code 2.1.283 takes `low`, `medium`, `high`, `xhigh` and `max`).
|
|
128
|
+
Its other four fields are still reported as untranslated.
|
|
129
|
+
- **`opencode` and `opencode-sdk`** get inference as opencode config, which
|
|
130
|
+
merges over your own opencode config for the same provider and model:
|
|
131
|
+
`temperature` and `reasoningEffort` become `options.temperature` and
|
|
132
|
+
`options.reasoningEffort` (opencode drops the snake_case spelling),
|
|
133
|
+
`enableThinking` becomes both wire forms an LLM engine sends, and
|
|
134
|
+
`maxTokens` with `contextLength` becomes `limit.output` and
|
|
135
|
+
`limit.context`. Set both limit fields: opencode refuses half a limit, and
|
|
136
|
+
a half would overwrite the other half of one you declared, so a lone one
|
|
137
|
+
is reported as untranslated. Without `limit.output` opencode asks for
|
|
138
|
+
`max_tokens: 32000`, which a small-context server rejects, and a wrong
|
|
139
|
+
`limit.context` lets it build requests past the server's window. The fields
|
|
140
|
+
need a `provider/model`: the request's model, or the `--model` an
|
|
141
|
+
`opencode` engine's `args` name. Model work needs none for its options, so
|
|
142
|
+
an `opencode-sdk` engine with no `model` and no `llmEngine` gets them too.
|
|
143
|
+
- **Where it goes.** Model work puts the options on the `akm-model-work`
|
|
144
|
+
agent that runs it, so opencode's own title call on the same model keeps
|
|
145
|
+
the model's defaults. Any other dispatch runs your own agent, whose name
|
|
146
|
+
akm cannot rely on, so its options go on the model and the title call sees
|
|
147
|
+
them too. `opencode-sdk` makes no title call. An `opencode-sdk` engine
|
|
148
|
+
declares its `llmEngine` fallback's model with the fallback's own
|
|
149
|
+
inference, under the engine's and the request's; each distinct set starts
|
|
150
|
+
its own `opencode serve`, as a different model does.
|
|
151
|
+
- **An agent engine may set the inference fields its platform translates.**
|
|
152
|
+
`engines.<name>` of `kind: "agent"` took none, so an engine could not carry a
|
|
153
|
+
`temperature` or a `reasoningEffort` of its own. `opencode` and
|
|
154
|
+
`opencode-sdk` may set `temperature`, `maxTokens`, `contextLength`,
|
|
155
|
+
`enableThinking` and `reasoningEffort`, `claude` may set `reasoningEffort`,
|
|
156
|
+
and every other platform none. A field its platform does not translate fails
|
|
157
|
+
to load, naming the platform and the fields it does translate. `provider`,
|
|
158
|
+
`endpoint`, `apiKey`, `apiKeyFile`, `concurrency` and `extraParams` stay
|
|
159
|
+
invalid on an agent engine.
|
|
160
|
+
- **Reasoning effort has one word in a request, `reasoningEffort`.** The
|
|
161
|
+
starter `reasoning` alias, an alias in your `models.json` and an asset's
|
|
162
|
+
`effort:` frontmatter said `effort`, and engines, opencode and the LLM
|
|
163
|
+
request said `reasoningEffort`. `effort` is now read as `reasoningEffort`
|
|
164
|
+
where layers are merged, so the nearest layer wins whichever word it used.
|
|
165
|
+
On an LLM engine an alias's or asset's `effort` is therefore sent as
|
|
166
|
+
`reasoning_effort`; it was reported as untranslated and dropped before. An
|
|
167
|
+
LLM engine's own request is unchanged.
|
|
168
|
+
- **Reflect asks every engine kind for the same JSON reply, checks it the same
|
|
169
|
+
way and repairs it once.** An LLM engine was sent the reply's JSON Schema,
|
|
170
|
+
and its reply was held to exact fields and repaired once. An agent or
|
|
171
|
+
`opencode-sdk` engine got a looser contract in its prompt (`ref`, `content`
|
|
172
|
+
and an optional `frontmatter`) with no schema and no repair, so one invalid
|
|
173
|
+
reply failed the run. Every engine kind now runs the same iteration:
|
|
174
|
+
- **The request.** The prompt carries the same output contract, and the
|
|
175
|
+
reply's JSON Schema is the request's output schema: `response_format` for
|
|
176
|
+
an LLM engine, as before, and the schema instruction at the end of the
|
|
177
|
+
prompt for an agent engine. `claude` also gets `--output-format json`,
|
|
178
|
+
which akm unwraps. An LLM endpoint that rejects JSON Schema still gets the
|
|
179
|
+
framed-markdown contract.
|
|
180
|
+
- **The reply.** An agent now returns `content`, `confidence` and a
|
|
181
|
+
`frontmatterPatch` of `description` and `when_to_use`, and `ref` too when
|
|
182
|
+
no asset was named, as an LLM does. akm derives a named asset's ref and
|
|
183
|
+
merges the patch with the source's frontmatter, so an agent can no longer
|
|
184
|
+
set other frontmatter keys or retarget the proposal.
|
|
185
|
+
- **The repair.** A reply that fails the contract gets one repair turn that
|
|
186
|
+
carries the first reply, shared across self-refine passes. A reply that is
|
|
187
|
+
still invalid fails with `parse_error` and queues nothing. On an agent or
|
|
188
|
+
`opencode-sdk` engine the error names the engine, as in
|
|
189
|
+
`Engine "oc" reply was not a valid reflect proposal after 2 attempts: …`.
|
|
190
|
+
An LLM engine's message is unchanged, because improve feeds it into later
|
|
191
|
+
prompts as a pattern to avoid. A failed agent dispatch is still reported
|
|
192
|
+
with its exit code and stderr.
|
|
193
|
+
- **What reflect sends and reports on an agent engine.** Reflect asks every
|
|
194
|
+
engine kind for no visible chain of thought (`enableThinking: false`).
|
|
195
|
+
`opencode` and `opencode-sdk` carry it, as the inference entry above
|
|
196
|
+
says; `claude` reports it as an `untranslated-field` notice, as it does
|
|
197
|
+
for every other stage. `reflect_completed` carries `outputMode` and
|
|
198
|
+
`repairAttempts` for every engine kind.
|
|
199
|
+
- **Unchanged.** An LLM engine's requests are byte-identical to before:
|
|
200
|
+
reflect's generation and repair, and its quality judge. So are the
|
|
201
|
+
refine passes, the content budget (an LLM engine's context length), the
|
|
202
|
+
protected frontmatter fields, the review routing and the judge selection.
|
|
203
|
+
|
|
204
|
+
### Removed
|
|
205
|
+
|
|
206
|
+
- **Two harness metadata fields that nothing read.** Each of the ten harness
|
|
207
|
+
descriptors carried an execution `pattern` and a `structuredOutput` tier.
|
|
208
|
+
No code branched on either: the shared request lowering appends the schema
|
|
209
|
+
instruction to every agent prompt, and a harness's own argv builder adds its
|
|
210
|
+
native channel, as codex does with `--output-schema`. The fields, their two
|
|
211
|
+
types and the tests that pinned their values are gone. Nothing you run
|
|
212
|
+
changes.
|
|
213
|
+
- **`resolveLlmEngineUse`'s swap of an agent engine for an LLM engine.** Given
|
|
214
|
+
an agent engine, it used the engine's `llmEngine`, then `defaults.llmEngine`,
|
|
215
|
+
and warned. Only the implicit SDK fallback above could reach it, so nothing
|
|
216
|
+
you run changes. An agent engine passed to it is now an error.
|
|
217
|
+
- **An `effort` hint on the agent dispatch request that nothing read.** The
|
|
218
|
+
lowering set it from `inference.effort`, reserved for a workflow field, and
|
|
219
|
+
no builder consumed it. `reasoningEffort` in the request's inference is read
|
|
220
|
+
where it is translated, and the field is gone.
|
|
221
|
+
- **The agent file-write contract.** An agent was once told to write its
|
|
222
|
+
proposal to a draft file and print `DRAFT_WRITTEN confidence=<n>`. Reflect and
|
|
223
|
+
`akm proposal new` had already stopped sending that instruction, because the
|
|
224
|
+
model-work scratch directory does not outlive a dispatch. The instruction, the
|
|
225
|
+
function that read the `DRAFT_WRITTEN` line and the `draftFilePath` prompt
|
|
226
|
+
inputs that nothing passed any more are gone, with their tests. So is
|
|
227
|
+
reflect's `ref_mismatch` check: a reflect that names an asset derives the
|
|
228
|
+
proposal's ref, so an engine can no longer name another one. Nothing you run
|
|
229
|
+
changes.
|
|
230
|
+
|
|
231
|
+
### Fixed
|
|
232
|
+
|
|
233
|
+
- **An `opencode-sdk` engine with an LLM fallback now reaches its endpoint,
|
|
234
|
+
and a failed dispatch is reported as a failure (#1015).** The `akm-custom`
|
|
235
|
+
provider that akm generates for the fallback listed no models, so opencode
|
|
236
|
+
could not find the model and failed every dispatch. akm now declares the
|
|
237
|
+
routed model under the provider's `models`. The runner also ignored the
|
|
238
|
+
error that the SDK client returns for an HTTP error, and the error opencode
|
|
239
|
+
puts on a reply when the provider rejects a request, and reported
|
|
240
|
+
`ok: true` with empty output. Both now give `ok: false`, with opencode's
|
|
241
|
+
error name and message in `error` and `stderr`. An aborted message is
|
|
242
|
+
`aborted`, a reply cut off at the output limit is `parse_error`, and any
|
|
243
|
+
other error is `non_zero_exit`. A reply with several text parts now returns
|
|
244
|
+
the last one, which is the answer, instead of the first.
|
|
245
|
+
- **An LLM engine reports a provider error sent with HTTP 200 as a failure.**
|
|
246
|
+
OpenRouter, for one, answers a request whose provider fails after the
|
|
247
|
+
response has started with HTTP 200 and a body that holds only an `error`
|
|
248
|
+
object and no `choices`. akm returned that as an empty reply with
|
|
249
|
+
`ok: true`. It is now a provider error with the body in its message, as an
|
|
250
|
+
error status is.
|
|
251
|
+
- **An `opencode` engine runs a persona instead of failing.** akm passed the
|
|
252
|
+
persona to `opencode run` as `--system-prompt`, which opencode 1.18 does
|
|
253
|
+
not accept, so opencode printed its usage and exited 1 on every dispatch
|
|
254
|
+
that carried a persona, including `akm agent <agent-ref> --engine opencode`.
|
|
255
|
+
akm now composes the persona into the prompt in an `<AKM_PERSONA>` block,
|
|
256
|
+
as it does for harnesses with no system-prompt option.
|
|
257
|
+
- **`opencode` and `opencode-sdk` engines receive a requested output schema.**
|
|
258
|
+
They dropped it with only an `untranslated-field` warning, so a schema from
|
|
259
|
+
`akm agent`, `akm command run`, a command's frontmatter or a task's
|
|
260
|
+
`output:` never reached the model. They now get it as the same instruction
|
|
261
|
+
every other agent engine gets.
|
|
262
|
+
- **An LLM engine sends a requested schema unless it opts out.** It sent
|
|
263
|
+
`response_format` only when the engine set `supportsJsonSchema: true`, so an
|
|
264
|
+
engine that left the flag unset never had its output constrained. It now
|
|
265
|
+
sends it unless the engine sets `supportsJsonSchema: false`. An endpoint
|
|
266
|
+
that rejects it with a 4xx is still retried once without it.
|
|
267
|
+
- **A workflow unit on `claude`, `copilot`, `gemini`, `pi`, `aider`,
|
|
268
|
+
`amazonq` or `openhands` gets the schema instruction once.** The unit prompt
|
|
269
|
+
carried it and the harness builder appended a second copy.
|
|
270
|
+
- **A feature gate's timeout now stops the call it bounds.** When an improve
|
|
271
|
+
stage's gate timed out (600 seconds unless the call sets its own), the model
|
|
272
|
+
call kept running in the background, on an engine with `timeoutMs: 900000`
|
|
273
|
+
for up to five more minutes. The gate now aborts it.
|
|
274
|
+
- **A stage call reports a timeout or abort by the dispatch's own reason.** A
|
|
275
|
+
timed-out or aborted agent or `opencode-sdk` dispatch, and an LLM timeout
|
|
276
|
+
inside a feature gate, came back as `error`. They now come back as `timeout`
|
|
277
|
+
or `aborted`.
|
|
278
|
+
|
|
279
|
+
- **`akm proposal new` keeps the reply's confidence.** The engine's
|
|
280
|
+
self-rated `confidence` was parsed and then dropped, so a proposal from
|
|
281
|
+
`proposal new` never carried the field the reference says it has.
|
|
282
|
+
- **Model work on an agent or `opencode-sdk` engine that sets no `timeoutMs` now
|
|
283
|
+
stops after 600 seconds, as documented.** Such an engine resolved to an
|
|
284
|
+
explicit "no timeout", so the 600-second bound for model work never applied
|
|
285
|
+
to it. An engine's own `timeoutMs`, `null` included, still applies, and
|
|
286
|
+
other work on an agent engine still runs until it finishes.
|
|
287
|
+
- **The announcement of the implicit `opencode-sdk` fallback is now true.** It
|
|
288
|
+
says provider, model and auth come from opencode's own configuration, but the
|
|
289
|
+
fallback engine borrowed `defaults.llmEngine`'s connection whenever one was
|
|
290
|
+
set, so opencode got an akm-generated provider and model instead. It now runs
|
|
291
|
+
on opencode's own configuration, as announced.
|
|
292
|
+
- **`model: reasoning` set no effort on `claude`, `opencode` or `opencode-sdk`.**
|
|
293
|
+
The starter alias supplies `effort: high` for all three, and each dropped it
|
|
294
|
+
with an `untranslated-field` notice, so the alias chose a stronger model and
|
|
295
|
+
nothing more. It now sets `--effort high` on `claude` and
|
|
296
|
+
`options.reasoningEffort` on opencode (see "Inference reaches `opencode`,
|
|
297
|
+
`opencode-sdk` and `claude` engines" above). An improve process's
|
|
298
|
+
`llm.reasoningEffort` and `llm.temperature` overlay now reaches those
|
|
299
|
+
engines the same way.
|
|
300
|
+
|
|
7
301
|
## [0.9.24] - 2026-10-02
|
|
8
302
|
|
|
9
303
|
### Changed
|
|
@@ -3,7 +3,6 @@
|
|
|
3
3
|
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
4
|
import { spawnSync } from "node:child_process";
|
|
5
5
|
import { loadConfig } from "../../core/config/config.js";
|
|
6
|
-
import { IMPROVE_PROCESS_ENGINE_CAPABILITIES } from "../../core/config/engine-semantics.js";
|
|
7
6
|
import { listEnvsRecursive } from "../../core/env-secret-ref.js";
|
|
8
7
|
import { ConfigError } from "../../core/errors.js";
|
|
9
8
|
import { EXTRACT_INFRASTRUCTURE_SKIP_REASONS } from "../../core/improve-types.js";
|
|
@@ -15,7 +14,7 @@ import { loadModelMap, mergeModelMapLayers, parseModelMapLayer, readInstalledMod
|
|
|
15
14
|
import { probeEndpointOnce } from "../../llm/client.js";
|
|
16
15
|
import { STATE_DB_FREELIST_WARN_RATIO, } from "../../storage/state-db-integrity.js";
|
|
17
16
|
import { listKeys } from "../env/env.js";
|
|
18
|
-
import { resolveImprovePlan } from "../improve/improve-strategies.js";
|
|
17
|
+
import { MODEL_CALLING_PROCESSES, resolveImprovePlan } from "../improve/improve-strategies.js";
|
|
19
18
|
import { ENGINE_LAST_USED_LOOKBACK_DAYS } from "./engine-usage.js";
|
|
20
19
|
import { ACTIVE_RUN_WARN_MS, TASK_FAIL_RATE_WARN, } from "./types.js";
|
|
21
20
|
/** Probe one connection's reachability, once per endpoint; `undefined` when no probe seam is supplied. */
|
|
@@ -142,7 +141,7 @@ async function runConfiguredEngineProbe(checkName, engineName, config, deps, rea
|
|
|
142
141
|
evidence: { engine: engineName, runtimeKind: "sdk", binaryAvailable: false },
|
|
143
142
|
};
|
|
144
143
|
}
|
|
145
|
-
const fallbackEngine = configuredEngine.llmEngine
|
|
144
|
+
const fallbackEngine = configuredEngine.llmEngine;
|
|
146
145
|
let fallback;
|
|
147
146
|
let fallbackCredential;
|
|
148
147
|
let fallbackApiKeyFile;
|
|
@@ -523,17 +522,17 @@ export function probeActiveImproveStrategy(deps = {}) {
|
|
|
523
522
|
.sort(([a], [b]) => a.localeCompare(b))
|
|
524
523
|
.map(([process, engine]) => `${process}: "${engine}"`)
|
|
525
524
|
.join(", ");
|
|
526
|
-
// #957: fail only when the strategy's
|
|
527
|
-
//
|
|
528
|
-
//
|
|
529
|
-
//
|
|
525
|
+
// #957: fail only when the strategy's model work would be a total no-op —
|
|
526
|
+
// every process the strategy actually enabled among the ones that call a
|
|
527
|
+
// model themselves (MODEL_CALLING_PROCESSES, any engine kind) ended up
|
|
528
|
+
// unavailable. A partial failure (some processes still have a working
|
|
530
529
|
// engine) stays a `warn`, matching #914's "a credential warn stays a warn"
|
|
531
530
|
// policy for the general per-engine probes; this is the strategy-scoped
|
|
532
531
|
// "is the whole run a no-op" question instead.
|
|
533
|
-
const
|
|
534
|
-
const
|
|
535
|
-
const
|
|
536
|
-
const allRequiredUnavailable =
|
|
532
|
+
const modelProcessNames = [...MODEL_CALLING_PROCESSES];
|
|
533
|
+
const requiredModelProcessNames = modelProcessNames.filter((name) => plan.processes[name].enabled || plan.engineUnavailable.some((item) => item.process === name));
|
|
534
|
+
const availableModelProcessNames = modelProcessNames.filter((name) => plan.processes[name].enabled);
|
|
535
|
+
const allRequiredUnavailable = requiredModelProcessNames.length > 0 && availableModelProcessNames.length === 0;
|
|
537
536
|
const status = unavailableProcesses.length === 0 ? "pass" : allRequiredUnavailable ? "fail" : "warn";
|
|
538
537
|
return {
|
|
539
538
|
check: {
|
|
@@ -36,6 +36,7 @@ import { concurrentMap } from "../../../core/concurrent.js";
|
|
|
36
36
|
import { parseEmbeddedJsonResponse } from "../../../core/parse.js";
|
|
37
37
|
import { DERIVED_SUFFIX } from "../../../core/recognition-util.js";
|
|
38
38
|
import { warnOnce } from "../../../core/warn.js";
|
|
39
|
+
import { runnerLlmConnection } from "../../../integrations/agent/runner.js";
|
|
39
40
|
import { assertRunnerCredentials } from "../../../integrations/agent/runner-dispatch.js";
|
|
40
41
|
import { runGit } from "../../../sources/providers/git-install.js";
|
|
41
42
|
import { closeDatabase, openExistingDatabase, openReadonlyExistingDatabase, } from "../../../storage/repositories/index-connection.js";
|
|
@@ -501,7 +502,7 @@ async function judgeOne(ctx, candidate) {
|
|
|
501
502
|
request: {
|
|
502
503
|
responseSchema: PAIR_JUDGE_JSON_SCHEMA,
|
|
503
504
|
enableThinking: false,
|
|
504
|
-
timeoutMs: ctx.llmRunner.timeoutMs,
|
|
505
|
+
...(Object.hasOwn(ctx.llmRunner, "timeoutMs") ? { timeoutMs: ctx.llmRunner.timeoutMs } : {}),
|
|
505
506
|
signal: ctx.opts.signal,
|
|
506
507
|
...(ctx.chat ? { chat: ctx.chat } : {}),
|
|
507
508
|
},
|
|
@@ -747,7 +748,7 @@ seams = {}) {
|
|
|
747
748
|
// entirely and needs no credential.
|
|
748
749
|
if (!seams.chat)
|
|
749
750
|
assertRunnerCredentials(llmRunner);
|
|
750
|
-
const results = await concurrentMap(judgeable, (candidate) => judgeOne(ctx, candidate), llmRunner
|
|
751
|
+
const results = await concurrentMap(judgeable, (candidate) => judgeOne(ctx, candidate), runnerLlmConnection(llmRunner)?.concurrency ?? 1, { signal: opts.signal });
|
|
751
752
|
results.forEach((r, idx) => {
|
|
752
753
|
// Must-fix 3: `concurrentMap` leaves an entry `undefined` for a call an
|
|
753
754
|
// aborted run never sent at all — `r?.failed === true` reads that as
|
|
@@ -36,6 +36,7 @@ import { resolveWriteTarget } from "../../core/write-source.js";
|
|
|
36
36
|
import { deriveInstallations } from "../../indexer/installations.js";
|
|
37
37
|
import { resolveSourceEntries } from "../../indexer/search/search-source.js";
|
|
38
38
|
import { USAGE_EVENT_RETENTION_DAYS } from "../../indexer/usage/usage-events.js";
|
|
39
|
+
import { runnerLlmConnection } from "../../integrations/agent/runner.js";
|
|
39
40
|
import { assertRunnerCredentials } from "../../integrations/agent/runner-dispatch.js";
|
|
40
41
|
import { cosineSimilarity, embedBatch, resolveEmbeddingModelId } from "../../llm/embedder.js";
|
|
41
42
|
import { getBodyEmbeddings, upsertBodyEmbeddings } from "../../storage/repositories/embeddings-repository.js";
|
|
@@ -566,7 +567,7 @@ async function judgeConsolidationChunks(args) {
|
|
|
566
567
|
request: {
|
|
567
568
|
responseSchema: CONSOLIDATE_PLAN_JSON_SCHEMA,
|
|
568
569
|
enableThinking: false,
|
|
569
|
-
timeoutMs: llmRunner.timeoutMs,
|
|
570
|
+
...(Object.hasOwn(llmRunner, "timeoutMs") ? { timeoutMs: llmRunner.timeoutMs } : {}),
|
|
570
571
|
signal: opts.signal,
|
|
571
572
|
},
|
|
572
573
|
...(opts.onNotices ? { onNotices: opts.onNotices } : {}),
|
|
@@ -619,7 +620,7 @@ async function planConsolidation(opts, config, stashDir, memories, warnings, sta
|
|
|
619
620
|
const llmRunner = opts.llmRunner ?? undefined;
|
|
620
621
|
// 500 body chars per memory keep the judgement useful; chunk size varies instead.
|
|
621
622
|
const bodyTruncation = 500;
|
|
622
|
-
const chunkSize = computeSafeChunkSize(llmRunner?.
|
|
623
|
+
const chunkSize = computeSafeChunkSize((llmRunner && runnerLlmConnection(llmRunner)?.contextLength) ?? DEFAULT_CONTEXT_LENGTH_TOKENS, bodyTruncation, opts.maxChunkSize);
|
|
623
624
|
const sourceName = opts.target ?? stashDir;
|
|
624
625
|
let budgeted = memories;
|
|
625
626
|
const budgetMs = opts.signal?.remainingBudgetMs;
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
3
|
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
4
|
import { deepMergeConfig } from "../../core/config/deep-merge.js";
|
|
5
|
+
import { MODEL_WORK_TOOLS } from "../../execution/source.js";
|
|
5
6
|
import { buildExecution, resolveExecution } from "../../integrations/agent/execution.js";
|
|
6
7
|
function own(value, key) {
|
|
7
8
|
return value !== undefined && Object.hasOwn(value, key);
|
|
@@ -22,6 +23,8 @@ function mergeDefaults(farther, nearer) {
|
|
|
22
23
|
/**
|
|
23
24
|
* Resolve improve-owned model work through the canonical execution cascade:
|
|
24
25
|
* defaults.llmEngine -> strategy -> index.<pass> -> process -> current invocation.
|
|
26
|
+
* The engine may be of any kind that confines the model-work tool policy: one
|
|
27
|
+
* that cannot, chosen with `--engine` say, is refused here, before any work.
|
|
25
28
|
*/
|
|
26
29
|
export function resolveImproveExecution(options) {
|
|
27
30
|
const defaultEngine = options.config.defaults?.llmEngine;
|
|
@@ -33,7 +36,7 @@ export function resolveImproveExecution(options) {
|
|
|
33
36
|
if (selectedEngine === undefined || selectedEngine === null)
|
|
34
37
|
return null;
|
|
35
38
|
const invocationDefaults = mergeDefaults(mergeDefaults(defaultEngine ? { engine: defaultEngine } : {}, profileDefaults), indexDefaults);
|
|
36
|
-
const current = mergeDefaults(processDefaults, currentDefaults);
|
|
39
|
+
const current = { ...mergeDefaults(processDefaults, currentDefaults), tools: MODEL_WORK_TOOLS };
|
|
37
40
|
const prepared = resolveExecution({
|
|
38
41
|
content: `improve ${options.processName} execution selection`,
|
|
39
42
|
config: options.config,
|
|
@@ -43,12 +46,3 @@ export function resolveImproveExecution(options) {
|
|
|
43
46
|
const lowered = buildExecution(prepared.request, prepared.runner);
|
|
44
47
|
return Object.freeze({ runner: lowered.runner, notices: lowered.notices });
|
|
45
48
|
}
|
|
46
|
-
export function resolveImproveLlmExecution(options) {
|
|
47
|
-
const resolved = resolveImproveExecution(options);
|
|
48
|
-
if (!resolved)
|
|
49
|
-
return null;
|
|
50
|
-
if (resolved.runner.kind !== "llm") {
|
|
51
|
-
return null;
|
|
52
|
-
}
|
|
53
|
-
return { runner: resolved.runner, notices: resolved.notices };
|
|
54
|
-
}
|
|
@@ -9,9 +9,9 @@
|
|
|
9
9
|
* session data into the markdown template loaded from
|
|
10
10
|
* `src/assets/prompts/extract-session.md`.
|
|
11
11
|
*
|
|
12
|
-
* The schema is intentionally strict —
|
|
13
|
-
*
|
|
14
|
-
* happy path. `additionalProperties: false` means any hallucinated keys
|
|
12
|
+
* The schema is intentionally strict — a provider that honours
|
|
13
|
+
* `response_format` enforces shape upstream, so the parser only has to handle
|
|
14
|
+
* the happy path. `additionalProperties: false` means any hallucinated keys
|
|
15
15
|
* the model emits get dropped before we parse.
|
|
16
16
|
*/
|
|
17
17
|
import promptTemplate from "../../assets/prompts/extract-session.md" with { type: "text" };
|
|
@@ -20,7 +20,7 @@ const EXTRACT_CANDIDATE_NAME_PATTERN = "^[a-z0-9](?:[a-z0-9-]*[a-z0-9])?(?:/[a-z
|
|
|
20
20
|
const EXTRACT_CANDIDATE_NAME_RE = new RegExp(EXTRACT_CANDIDATE_NAME_PATTERN);
|
|
21
21
|
/**
|
|
22
22
|
* JSON Schema for the structured extract output. Passed to `chatCompletion`
|
|
23
|
-
*
|
|
23
|
+
* unless the configured LLM connection sets `supportsJsonSchema: false`.
|
|
24
24
|
*
|
|
25
25
|
* Shape:
|
|
26
26
|
* {
|
|
@@ -32,16 +32,15 @@ import { indexWrittenAssets } from "../../indexer/index-written-assets.js";
|
|
|
32
32
|
import { assertRunnerCredentials } from "../../integrations/agent/runner-dispatch.js";
|
|
33
33
|
import { getAvailableHarnesses } from "../../integrations/session-logs/index.js";
|
|
34
34
|
import { preFilterSession } from "../../integrations/session-logs/pre-filter.js";
|
|
35
|
-
import { isJsonSchemaKnownUnsupported } from "../../llm/client.js";
|
|
36
35
|
import { getExtractedSessionsMap, getLastExtractRunAt, shouldSkipAlreadyExtractedSession, upsertExtractedSession, } from "../../storage/repositories/extract-sessions-repository.js";
|
|
37
36
|
import { openSqliteReadSnapshot } from "../../storage/sqlite-read-snapshot.js";
|
|
38
37
|
import { contentHash } from "./content-hash.js";
|
|
39
|
-
import {
|
|
38
|
+
import { resolveImproveExecution } from "./execution.js";
|
|
40
39
|
import { buildExtractPrompt, EXTRACT_JSON_SCHEMA, parseExtractPayload, } from "./extract-prompt.js";
|
|
41
40
|
import { cloneAndFreeze, resolveImproveStrategy, resolveProcessEnabled } from "./improve-strategies.js";
|
|
42
41
|
import { isLedgerBlocked, ledgerKey, loadLedgerSnapshot } from "./ledger.js";
|
|
43
42
|
import { buildSessionSummaryPrompt, parseSessionSummary, SESSION_SUMMARY_JSON_SCHEMA, sessionMeetsDurationGate, writeSessionAsset, } from "./session-asset.js";
|
|
44
|
-
import { callStage, mintProposal, noticeSet } from "./stage.js";
|
|
43
|
+
import { callStage, callStageOnce, mintProposal, noticeSet } from "./stage.js";
|
|
45
44
|
/** Minimum session duration (minutes) for writing a session asset. */
|
|
46
45
|
const DEFAULT_MIN_SESSION_DURATION_MINUTES = 5;
|
|
47
46
|
/** Raw session size (chars) below which the LLM call is skipped; only truly empty sessions are safe to skip. */
|
|
@@ -119,7 +118,7 @@ export function resolveStandaloneExtractPlan(config, selection) {
|
|
|
119
118
|
}
|
|
120
119
|
const selected = resolveImproveStrategy(selection.strategy, config);
|
|
121
120
|
const process = cloneAndFreeze(getImproveProcessConfig("extract", selected.config) ?? {});
|
|
122
|
-
const resolved =
|
|
121
|
+
const resolved = resolveImproveExecution({
|
|
123
122
|
config,
|
|
124
123
|
profile: selected.config,
|
|
125
124
|
process,
|
|
@@ -130,7 +129,7 @@ export function resolveStandaloneExtractPlan(config, selection) {
|
|
|
130
129
|
processName: "extract",
|
|
131
130
|
});
|
|
132
131
|
if (!resolved) {
|
|
133
|
-
throw new ConfigError("No
|
|
132
|
+
throw new ConfigError("No engine configured for extract. Set defaults.llmEngine, pass --engine, or select an improve strategy with processes.extract.engine.", "LLM_NOT_CONFIGURED");
|
|
134
133
|
}
|
|
135
134
|
const runner = resolved.runner;
|
|
136
135
|
return Object.freeze({
|
|
@@ -358,15 +357,16 @@ function planExtractSessions(args) {
|
|
|
358
357
|
}
|
|
359
358
|
const EXTRACT_LLM_UNAVAILABLE = Symbol("extract-llm-unavailable");
|
|
360
359
|
/**
|
|
361
|
-
* One session's extraction call
|
|
362
|
-
*
|
|
360
|
+
* One session's extraction call, with one corrective retry; configuration
|
|
361
|
+
* errors escape before any state is written.
|
|
363
362
|
*/
|
|
364
363
|
async function extractFromSession(run, prompt) {
|
|
365
364
|
const { llmRunner } = run;
|
|
366
365
|
try {
|
|
367
366
|
const result = await runStructured({
|
|
368
367
|
dispatch: async (feedback) => {
|
|
369
|
-
|
|
368
|
+
// This loop parses and repairs the reply itself, so each attempt is one unvalidated dispatch.
|
|
369
|
+
const outcome = await callStageOnce({
|
|
370
370
|
feature: "session_extraction",
|
|
371
371
|
runner: llmRunner,
|
|
372
372
|
prompt: feedback ? `${prompt}\n\n## Corrective output instruction\n\n${feedback}` : prompt,
|
|
@@ -388,9 +388,6 @@ async function extractFromSession(run, prompt) {
|
|
|
388
388
|
return payload.parseFailure ? undefined : payload;
|
|
389
389
|
},
|
|
390
390
|
validate: (payload) => ({ ok: true, value: payload }),
|
|
391
|
-
maxAttempts: llmRunner.connection.supportsJsonSchema !== false && !isJsonSchemaKnownUnsupported(llmRunner.connection)
|
|
392
|
-
? 1
|
|
393
|
-
: 2,
|
|
394
391
|
buildFeedback: () => "Your previous response did not contain a valid extraction payload. Respond with ONLY a JSON object matching the requested schema, with a candidates array and no prose or code fences.",
|
|
395
392
|
});
|
|
396
393
|
if (result.ok)
|
|
@@ -714,13 +711,13 @@ function resolveExtractRun(options, config, process, activeProfile) {
|
|
|
714
711
|
llmRunner = options.llmRunner;
|
|
715
712
|
}
|
|
716
713
|
else {
|
|
717
|
-
const resolved =
|
|
714
|
+
const resolved = resolveImproveExecution({ config, profile: activeProfile, process, processName: "extract" });
|
|
718
715
|
llmRunner = resolved?.runner;
|
|
719
716
|
if (resolved)
|
|
720
717
|
notices.add(resolved.notices);
|
|
721
718
|
}
|
|
722
719
|
if (!llmRunner) {
|
|
723
|
-
throw new ConfigError("No
|
|
720
|
+
throw new ConfigError("No engine configured for extract. Set defaults.llmEngine or improve.strategies.<name>.processes.extract.engine.", "LLM_NOT_CONFIGURED");
|
|
724
721
|
}
|
|
725
722
|
const runner = llmRunner;
|
|
726
723
|
const timeoutMs = options.resolvedPlan
|