akm-cli 0.9.23 → 0.9.25-alpha.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/CHANGELOG.md +339 -0
  2. package/dist/commands/health/checks.js +10 -11
  3. package/dist/commands/improve/consolidate/pair-pass.js +3 -2
  4. package/dist/commands/improve/consolidate.js +3 -2
  5. package/dist/commands/improve/execution.js +4 -10
  6. package/dist/commands/improve/extract-prompt.js +4 -4
  7. package/dist/commands/improve/extract.js +10 -13
  8. package/dist/commands/improve/improve-cli.js +32 -33
  9. package/dist/commands/improve/improve-strategies.js +49 -43
  10. package/dist/commands/improve/improve-usage-report.js +8 -17
  11. package/dist/commands/improve/preparation.js +3 -1
  12. package/dist/commands/improve/reflect.js +132 -177
  13. package/dist/commands/improve/stage.js +69 -18
  14. package/dist/commands/proposal/drain.js +19 -7
  15. package/dist/commands/proposal/proposal-cli.js +1 -5
  16. package/dist/commands/proposal/propose-cli.js +2 -2
  17. package/dist/commands/proposal/propose.js +80 -83
  18. package/dist/commands/remember.js +3 -3
  19. package/dist/commands/sources/schema-repair.js +1 -1
  20. package/dist/core/config/config-schema.js +37 -60
  21. package/dist/core/config/engine-semantics.js +15 -11
  22. package/dist/core/config/schema/engines.js +33 -15
  23. package/dist/core/config/schema/improve-processes.js +2 -2
  24. package/dist/core/improve-result.js +3 -3
  25. package/dist/core/structured.js +10 -0
  26. package/dist/execution/source.js +14 -0
  27. package/dist/indexer/passes/memory-inference.js +2 -1
  28. package/dist/integrations/agent/builder-shared.js +15 -0
  29. package/dist/integrations/agent/config.js +2 -0
  30. package/dist/integrations/agent/engine-resolution.js +16 -31
  31. package/dist/integrations/agent/execution.js +46 -21
  32. package/dist/integrations/agent/index.js +1 -1
  33. package/dist/integrations/agent/model-map.js +16 -15
  34. package/dist/integrations/agent/prompts.js +53 -76
  35. package/dist/integrations/agent/request-lowering.js +20 -9
  36. package/dist/integrations/agent/runner-dispatch.js +103 -4
  37. package/dist/integrations/agent/runner.js +8 -2
  38. package/dist/integrations/harnesses/aider/agent-builder.js +11 -23
  39. package/dist/integrations/harnesses/aider/index.js +0 -5
  40. package/dist/integrations/harnesses/amazonq/agent-builder.js +10 -21
  41. package/dist/integrations/harnesses/amazonq/index.js +0 -5
  42. package/dist/integrations/harnesses/claude/agent-builder.js +61 -38
  43. package/dist/integrations/harnesses/claude/index.js +0 -14
  44. package/dist/integrations/harnesses/codex/agent-builder.js +4 -4
  45. package/dist/integrations/harnesses/codex/index.js +0 -4
  46. package/dist/integrations/harnesses/copilot/agent-builder.js +4 -13
  47. package/dist/integrations/harnesses/copilot/index.js +2 -7
  48. package/dist/integrations/harnesses/gemini/agent-builder.js +4 -13
  49. package/dist/integrations/harnesses/gemini/index.js +0 -5
  50. package/dist/integrations/harnesses/ids.js +18 -10
  51. package/dist/integrations/harnesses/opencode/agent-builder.js +52 -10
  52. package/dist/integrations/harnesses/opencode/index.js +0 -8
  53. package/dist/integrations/harnesses/opencode/model-config.js +80 -0
  54. package/dist/integrations/harnesses/opencode/model-work-agent.js +80 -0
  55. package/dist/integrations/harnesses/opencode-sdk/harness.js +6 -7
  56. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +168 -24
  57. package/dist/integrations/harnesses/openhands/agent-builder.js +8 -21
  58. package/dist/integrations/harnesses/openhands/index.js +0 -5
  59. package/dist/integrations/harnesses/pi/agent-builder.js +8 -21
  60. package/dist/integrations/harnesses/pi/index.js +0 -5
  61. package/dist/llm/client.js +5 -0
  62. package/dist/llm/feature-gate.js +10 -4
  63. package/dist/llm/index-passes.js +3 -6
  64. package/dist/llm/memory-infer.js +6 -5
  65. package/dist/llm/structured-call.js +33 -11
  66. package/dist/scripts/akm-migrate-node.js +298 -204
  67. package/dist/scripts/akm-migrate.js +298 -204
  68. package/dist/workflows/exec/step-work.js +6 -5
  69. package/dist/workflows/freeze/step-values.js +1 -1
  70. package/docs/reference/cli.md +34 -14
  71. package/docs/reference/configuration.md +171 -12
  72. package/docs/reference/workflow-schema.md +13 -9
  73. package/package.json +1 -1
  74. package/schemas/akm-config.json +36 -0
package/CHANGELOG.md CHANGED
@@ -4,6 +4,345 @@ All notable changes to this project will be documented in this file.
4
4
 
5
5
  The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
6
6
 
7
+ ## [Unreleased]
8
+
9
+ ## [0.9.25-alpha.1] - 2026-10-02
10
+
11
+ ### Changed
12
+
13
+ - **One schema instruction, appended by the shared agent request lowering.**
14
+ Seven harness builders and the workflow engine each kept a copy of
15
+ "Respond with ONLY a JSON value matching this JSON Schema (no prose, no code
16
+ fences)". The lowering now appends it to every agent engine's prompt when a
17
+ schema is requested, so `codex` gets it beside `--output-schema` outside
18
+ workflows too. A workflow unit sends the same prompt bytes as before, less
19
+ the duplicate fixed below; a direct-LLM unit still carries the instruction
20
+ in its own prompt.
21
+ - **Model work is bounded at 600 seconds on every engine kind.** An improve
22
+ process, a quality or triage judge, or an index pass whose engine sets no
23
+ `timeoutMs` now stops after 600 seconds. Before, an agent or `opencode-sdk`
24
+ engine ran until it finished, and so did memory inference and consolidation
25
+ on an LLM engine. An engine's own `timeoutMs` still applies.
26
+ - **An improve stage's reply that fails its JSON Schema gets one corrective
27
+ retry.** The stage retries once with the validation errors, then reads the
28
+ last reply with its own parser as before. `akm extract` now allows its
29
+ corrective retry on every engine, not only on one without JSON Schema
30
+ support. Reflect keeps its own repair turn.
31
+ - **Agent and `opencode-sdk` dispatches leave a usage record.** Each one now
32
+ writes one `llm_usage` record through the same sink and stage attribution as
33
+ the LLM path, with the request's model and the tokens the runner reports. An
34
+ LLM engine still records each HTTP attempt.
35
+ - **`akm proposal new` gets the proposal as JSON from every engine kind, with
36
+ no live session.** It told every engine to write the asset to a draft file
37
+ and print no JSON, then parsed stdout as JSON. An LLM engine that followed
38
+ the instruction could not succeed, and an agent engine succeeded only by
39
+ writing the file: an agent CLI ran in a live terminal session whose output
40
+ akm could not read, and otherwise failed with an "interactive mode" error.
41
+ Every engine now returns the proposal as one JSON object on stdout, and the
42
+ request carries its JSON Schema: as `response_format` for an LLM engine, as
43
+ the schema instruction for an agent engine. akm captures the reply, unwraps
44
+ a harness envelope such as claude's `--output-format json` result, and
45
+ validates it. A reply that is not a proposal gets one corrective retry, then
46
+ fails with an error that names the engine. An agent CLI now runs headless:
47
+ you see the queued proposal, not the agent at work, and no draft file is
48
+ written.
49
+ - **Unattended model work has one tool policy, which each engine confines or
50
+ refuses when the request is built.** The policy allows reading, editing only
51
+ inside a scratch working directory that akm creates for the dispatch and
52
+ removes after it, and running `akm search` and `akm show`. The stash stays
53
+ read-only. An LLM engine has no tools. `claude` runs it with `--restricted`
54
+ (its user, project and local settings, whose allow rules could pre-approve
55
+ any command or path, are ignored, and its file tools stay in the working
56
+ directory), `--strict-mcp-config`, `--tools Read,Edit,Bash`, `--allowedTools`
57
+ for Read, Edit and the two `akm` commands, and `--permission-mode dontAsk`.
58
+ `opencode` and `opencode-sdk` run an injected `akm-model-work` agent that
59
+ can read and edit only inside the working directory and has no bash:
60
+ opencode checks a bash rule against the command's words only, so
61
+ `akm show x > ~/stash/asset.md` would pass an `akm show *` rule. The agent
62
+ has its own short prompt in place of opencode's coding-assistant prompt, an
63
+ 8-step limit, and no automatic compaction. opencode only asks the model to
64
+ stop at that limit, so `opencode-sdk` aborts the session two steps past it,
65
+ and aborts a session the dispatch times out on; on `opencode` the dispatch
66
+ timeout is the bound. Every other harness refuses the policy. Such a
67
+ dispatch ignores the engine's `args` and `workspace`, refuses to start when
68
+ the temporary directory is inside a git repository, which opencode would
69
+ treat as its working directory, and fails with `parse_error` when the agent
70
+ ends with no answer, as opencode can at its step limit.
71
+ - **Unattended model work runs on any engine that confines the model-work
72
+ tool policy, and config checks that with one rule.**
73
+ - **Who sends the policy.** The improve processes, the quality, triage and
74
+ retrieval-gate judges, index passes and `akm remember --enrich` now send
75
+ it, so they may run on an LLM engine or on a `claude`, `opencode` or
76
+ `opencode-sdk` agent engine, where they could only use an LLM engine
77
+ before.
78
+ - **The one rule.** Every key model work reads its engine from must name
79
+ such an engine. The keys are `defaults.llmEngine`, `index.defaults.engine`,
80
+ `index.<pass>.engine`, a strategy's `engine`, a process's `engine`, an
81
+ enabled triage `judgment.engine`, and a `qualityGate.engine`. A config
82
+ that breaks the rule fails to load, naming the key, the engine and its
83
+ platform. The rule replaces six checks that each required an LLM engine
84
+ for some of those keys, or let the triage judgment use any agent.
85
+ - **What else is gone.** The plan, reflect, the quality gate and index
86
+ passes no longer turn an agent engine away. A triage judgment on an agent
87
+ engine takes a strategy's `llm` overrides as `untranslated-field`
88
+ notices, where it was refused before.
89
+ - **Reflect on an agent engine.** The agent returns its proposal as JSON on
90
+ stdout instead of writing a draft file, which the policy's scratch working
91
+ directory would not keep.
92
+ - **The answer is unwrapped.** A model-work reply from an agent engine goes
93
+ through its harness's result extractor, so a stage call on `claude` gets
94
+ the answer, not the `--output-format json` envelope around it.
95
+ - **`--require-engines` checks agent engines too.** It checks an agent
96
+ engine by its binary on PATH, and an `opencode-sdk` engine by its binary
97
+ and its LLM fallback's endpoint. It probed only LLM connections before.
98
+ The usage report and `akm health` count a process's calls whatever its
99
+ engine's kind.
100
+ - **An `opencode-sdk` engine gets its LLM fallback connection only from its own
101
+ `llmEngine`. `defaults.llmEngine` no longer supplies one.** An SDK engine
102
+ that set no `llmEngine` used to borrow `defaults.llmEngine`: its connection,
103
+ its model (unless the engine set its own) and its timeout. `defaults.llmEngine`
104
+ names the default engine for unattended model work, and since that work may
105
+ now run on an agent engine it can be one too, so it is no longer also a
106
+ connection that every SDK engine shares. **If you relied on the inheritance,
107
+ set `llmEngine` on the SDK engine**, for example
108
+ `"sdk": { "kind": "agent", "platform": "opencode-sdk", "llmEngine": "fast" }`.
109
+ Without one, the SDK engine runs on opencode's own provider and auth, and on
110
+ its own `model` if it sets one: akm sends it no connection. An SDK engine
111
+ that sets `llmEngine` is unchanged, and so is a config with no `opencode-sdk`
112
+ engine. The rule holds everywhere the fallback is read: dispatch, a
113
+ workflow's frozen concurrency cap, `akm health`, and
114
+ `akm improve --require-engines`, which now check an SDK engine's fallback
115
+ endpoint only when it sets `llmEngine`.
116
+ - **Inference reaches `opencode`, `opencode-sdk` and `claude` engines.** A
117
+ request's `temperature`, `maxTokens`, `contextLength`, `enableThinking` and
118
+ `reasoningEffort` came from the engine's own settings, an improve process's
119
+ `llm` overlay, a task, command or agent asset's `inference`, a workflow's
120
+ `llm:` and a `models.json` alias, and every agent engine dropped all of them
121
+ with an `untranslated-field` notice. Each platform now translates what it can
122
+ carry, and the nearest layer wins, field by field, as on an LLM engine. With
123
+ no setting anywhere akm sends nothing of its own, so the model's configured
124
+ default still applies, such as a `reasoningEffort: "none"` in your opencode
125
+ config.
126
+ - **`claude`** gets `reasoningEffort` as `--effort <level>`, passed as given
127
+ (Claude Code 2.1.283 takes `low`, `medium`, `high`, `xhigh` and `max`).
128
+ Its other four fields are still reported as untranslated.
129
+ - **`opencode` and `opencode-sdk`** get inference as opencode config, which
130
+ merges over your own opencode config for the same provider and model:
131
+ `temperature` and `reasoningEffort` become `options.temperature` and
132
+ `options.reasoningEffort` (opencode drops the snake_case spelling),
133
+ `enableThinking` becomes both wire forms an LLM engine sends, and
134
+ `maxTokens` with `contextLength` becomes `limit.output` and
135
+ `limit.context`. Set both limit fields: opencode refuses half a limit, and
136
+ a half would overwrite the other half of one you declared, so a lone one
137
+ is reported as untranslated. Without `limit.output` opencode asks for
138
+ `max_tokens: 32000`, which a small-context server rejects, and a wrong
139
+ `limit.context` lets it build requests past the server's window. The fields
140
+ need a `provider/model`: the request's model, or the `--model` an
141
+ `opencode` engine's `args` name. Model work needs none for its options, so
142
+ an `opencode-sdk` engine with no `model` and no `llmEngine` gets them too.
143
+ - **Where it goes.** Model work puts the options on the `akm-model-work`
144
+ agent that runs it, so opencode's own title call on the same model keeps
145
+ the model's defaults. Any other dispatch runs your own agent, whose name
146
+ akm cannot rely on, so its options go on the model and the title call sees
147
+ them too. `opencode-sdk` makes no title call. An `opencode-sdk` engine
148
+ declares its `llmEngine` fallback's model with the fallback's own
149
+ inference, under the engine's and the request's; each distinct set starts
150
+ its own `opencode serve`, as a different model does.
151
+ - **An agent engine may set the inference fields its platform translates.**
152
+ `engines.<name>` of `kind: "agent"` took none, so an engine could not carry a
153
+ `temperature` or a `reasoningEffort` of its own. `opencode` and
154
+ `opencode-sdk` may set `temperature`, `maxTokens`, `contextLength`,
155
+ `enableThinking` and `reasoningEffort`, `claude` may set `reasoningEffort`,
156
+ and every other platform none. A field its platform does not translate fails
157
+ to load, naming the platform and the fields it does translate. `provider`,
158
+ `endpoint`, `apiKey`, `apiKeyFile`, `concurrency` and `extraParams` stay
159
+ invalid on an agent engine.
160
+ - **Reasoning effort has one word in a request, `reasoningEffort`.** The
161
+ starter `reasoning` alias, an alias in your `models.json` and an asset's
162
+ `effort:` frontmatter said `effort`, and engines, opencode and the LLM
163
+ request said `reasoningEffort`. `effort` is now read as `reasoningEffort`
164
+ where layers are merged, so the nearest layer wins whichever word it used.
165
+ On an LLM engine an alias's or asset's `effort` is therefore sent as
166
+ `reasoning_effort`; it was reported as untranslated and dropped before. An
167
+ LLM engine's own request is unchanged.
168
+ - **Reflect asks every engine kind for the same JSON reply, checks it the same
169
+ way and repairs it once.** An LLM engine was sent the reply's JSON Schema,
170
+ and its reply was held to exact fields and repaired once. An agent or
171
+ `opencode-sdk` engine got a looser contract in its prompt (`ref`, `content`
172
+ and an optional `frontmatter`) with no schema and no repair, so one invalid
173
+ reply failed the run. Every engine kind now runs the same iteration:
174
+ - **The request.** The prompt carries the same output contract, and the
175
+ reply's JSON Schema is the request's output schema: `response_format` for
176
+ an LLM engine, as before, and the schema instruction at the end of the
177
+ prompt for an agent engine. `claude` also gets `--output-format json`,
178
+ which akm unwraps. An LLM endpoint that rejects JSON Schema still gets the
179
+ framed-markdown contract.
180
+ - **The reply.** An agent now returns `content`, `confidence` and a
181
+ `frontmatterPatch` of `description` and `when_to_use`, and `ref` too when
182
+ no asset was named, as an LLM does. akm derives a named asset's ref and
183
+ merges the patch with the source's frontmatter, so an agent can no longer
184
+ set other frontmatter keys or retarget the proposal.
185
+ - **The repair.** A reply that fails the contract gets one repair turn that
186
+ carries the first reply, shared across self-refine passes. A reply that is
187
+ still invalid fails with `parse_error` and queues nothing. On an agent or
188
+ `opencode-sdk` engine the error names the engine, as in
189
+ `Engine "oc" reply was not a valid reflect proposal after 2 attempts: …`.
190
+ An LLM engine's message is unchanged, because improve feeds it into later
191
+ prompts as a pattern to avoid. A failed agent dispatch is still reported
192
+ with its exit code and stderr.
193
+ - **What reflect sends and reports on an agent engine.** Reflect asks every
194
+ engine kind for no visible chain of thought (`enableThinking: false`).
195
+ `opencode` and `opencode-sdk` carry it, as the inference entry above
196
+ says; `claude` reports it as an `untranslated-field` notice, as it does
197
+ for every other stage. `reflect_completed` carries `outputMode` and
198
+ `repairAttempts` for every engine kind.
199
+ - **Unchanged.** An LLM engine's requests are byte-identical to before:
200
+ reflect's generation and repair, and its quality judge. So are the
201
+ refine passes, the content budget (an LLM engine's context length), the
202
+ protected frontmatter fields, the review routing and the judge selection.
203
+
204
+ ### Removed
205
+
206
+ - **Two harness metadata fields that nothing read.** Each of the ten harness
207
+ descriptors carried an execution `pattern` and a `structuredOutput` tier.
208
+ No code branched on either: the shared request lowering appends the schema
209
+ instruction to every agent prompt, and a harness's own argv builder adds its
210
+ native channel, as codex does with `--output-schema`. The fields, their two
211
+ types and the tests that pinned their values are gone. Nothing you run
212
+ changes.
213
+ - **`resolveLlmEngineUse`'s swap of an agent engine for an LLM engine.** Given
214
+ an agent engine, it used the engine's `llmEngine`, then `defaults.llmEngine`,
215
+ and warned. Only the implicit SDK fallback above could reach it, so nothing
216
+ you run changes. An agent engine passed to it is now an error.
217
+ - **An `effort` hint on the agent dispatch request that nothing read.** The
218
+ lowering set it from `inference.effort`, reserved for a workflow field, and
219
+ no builder consumed it. `reasoningEffort` in the request's inference is read
220
+ where it is translated, and the field is gone.
221
+ - **The agent file-write contract.** An agent was once told to write its
222
+ proposal to a draft file and print `DRAFT_WRITTEN confidence=<n>`. Reflect and
223
+ `akm proposal new` had already stopped sending that instruction, because the
224
+ model-work scratch directory does not outlive a dispatch. The instruction, the
225
+ function that read the `DRAFT_WRITTEN` line and the `draftFilePath` prompt
226
+ inputs that nothing passed any more are gone, with their tests. So is
227
+ reflect's `ref_mismatch` check: a reflect that names an asset derives the
228
+ proposal's ref, so an engine can no longer name another one. Nothing you run
229
+ changes.
230
+
231
+ ### Fixed
232
+
233
+ - **An `opencode-sdk` engine with an LLM fallback now reaches its endpoint,
234
+ and a failed dispatch is reported as a failure (#1015).** The `akm-custom`
235
+ provider that akm generates for the fallback listed no models, so opencode
236
+ could not find the model and failed every dispatch. akm now declares the
237
+ routed model under the provider's `models`. The runner also ignored the
238
+ error that the SDK client returns for an HTTP error, and the error opencode
239
+ puts on a reply when the provider rejects a request, and reported
240
+ `ok: true` with empty output. Both now give `ok: false`, with opencode's
241
+ error name and message in `error` and `stderr`. An aborted message is
242
+ `aborted`, a reply cut off at the output limit is `parse_error`, and any
243
+ other error is `non_zero_exit`. A reply with several text parts now returns
244
+ the last one, which is the answer, instead of the first.
245
+ - **An LLM engine reports a provider error sent with HTTP 200 as a failure.**
246
+ OpenRouter, for one, answers a request whose provider fails after the
247
+ response has started with HTTP 200 and a body that holds only an `error`
248
+ object and no `choices`. akm returned that as an empty reply with
249
+ `ok: true`. It is now a provider error with the body in its message, as an
250
+ error status is.
251
+ - **An `opencode` engine runs a persona instead of failing.** akm passed the
252
+ persona to `opencode run` as `--system-prompt`, which opencode 1.18 does
253
+ not accept, so opencode printed its usage and exited 1 on every dispatch
254
+ that carried a persona, including `akm agent <agent-ref> --engine opencode`.
255
+ akm now composes the persona into the prompt in an `<AKM_PERSONA>` block,
256
+ as it does for harnesses with no system-prompt option.
257
+ - **`opencode` and `opencode-sdk` engines receive a requested output schema.**
258
+ They dropped it with only an `untranslated-field` warning, so a schema from
259
+ `akm agent`, `akm command run`, a command's frontmatter or a task's
260
+ `output:` never reached the model. They now get it as the same instruction
261
+ every other agent engine gets.
262
+ - **An LLM engine sends a requested schema unless it opts out.** It sent
263
+ `response_format` only when the engine set `supportsJsonSchema: true`, so an
264
+ engine that left the flag unset never had its output constrained. It now
265
+ sends it unless the engine sets `supportsJsonSchema: false`. An endpoint
266
+ that rejects it with a 4xx is still retried once without it.
267
+ - **A workflow unit on `claude`, `copilot`, `gemini`, `pi`, `aider`,
268
+ `amazonq` or `openhands` gets the schema instruction once.** The unit prompt
269
+ carried it and the harness builder appended a second copy.
270
+ - **A feature gate's timeout now stops the call it bounds.** When an improve
271
+ stage's gate timed out (600 seconds unless the call sets its own), the model
272
+ call kept running in the background, on an engine with `timeoutMs: 900000`
273
+ for up to five more minutes. The gate now aborts it.
274
+ - **A stage call reports a timeout or abort by the dispatch's own reason.** A
275
+ timed-out or aborted agent or `opencode-sdk` dispatch, and an LLM timeout
276
+ inside a feature gate, came back as `error`. They now come back as `timeout`
277
+ or `aborted`.
278
+
279
+ - **`akm proposal new` keeps the reply's confidence.** The engine's
280
+ self-rated `confidence` was parsed and then dropped, so a proposal from
281
+ `proposal new` never carried the field the reference says it has.
282
+ - **Model work on an agent or `opencode-sdk` engine that sets no `timeoutMs` now
283
+ stops after 600 seconds, as documented.** Such an engine resolved to an
284
+ explicit "no timeout", so the 600-second bound for model work never applied
285
+ to it. An engine's own `timeoutMs`, `null` included, still applies, and
286
+ other work on an agent engine still runs until it finishes.
287
+ - **The announcement of the implicit `opencode-sdk` fallback is now true.** It
288
+ says provider, model and auth come from opencode's own configuration, but the
289
+ fallback engine borrowed `defaults.llmEngine`'s connection whenever one was
290
+ set, so opencode got an akm-generated provider and model instead. It now runs
291
+ on opencode's own configuration, as announced.
292
+ - **`model: reasoning` set no effort on `claude`, `opencode` or `opencode-sdk`.**
293
+ The starter alias supplies `effort: high` for all three, and each dropped it
294
+ with an `untranslated-field` notice, so the alias chose a stronger model and
295
+ nothing more. It now sets `--effort high` on `claude` and
296
+ `options.reasoningEffort` on opencode (see "Inference reaches `opencode`,
297
+ `opencode-sdk` and `claude` engines" above). An improve process's
298
+ `llm.reasoningEffort` and `llm.temperature` overlay now reaches those
299
+ engines the same way.
300
+
301
+ ## [0.9.24] - 2026-10-02
302
+
303
+ ### Changed
304
+
305
+ - **Reflect never auto-accepts a revision that changes the body; it waits for
306
+ review.** On 396 labelled reflect edits, the judge-passed edits that changed
307
+ the body were good 12 times in 37, and those that changed only the
308
+ frontmatter 13 times in 13. A revision whose body differs from the asset's,
309
+ ignoring whitespace, or whose source could not be read, is now created
310
+ pending and deferred for review with the reason `body-edit` (gate
311
+ `reflect`). When the quality judge passed it, the judge's scores and reason
312
+ stay on its gate decision. With `processes.reflect.qualityGate` off, no
313
+ judge ran, so the decision carries none. Before, the triage drain accepted a
314
+ judge-passed body edit, and its judgment tier could accept one made with the
315
+ gate off. These body edits now appear in the review queue
316
+ (`akm proposal list`), and `akm proposal show <id>` gives the reason and any
317
+ judge scores. A judge-passed revision that leaves the body unchanged is
318
+ still staged and accepted by the drain. A frontmatter-only revision made
319
+ with the gate off is still left to the drain, and a revision the judge fails
320
+ is still refused.
321
+
322
+ ### Fixed
323
+
324
+ - **Reflect sends a revision to review when its judge fails, instead of
325
+ rejecting it.** A quality judge that timed out, errored or returned a reply
326
+ that could not be parsed gave no verdict, but reflect treated that as a
327
+ rejection. It created no proposal, recorded a `quality_rejected` attempt that
328
+ kept the asset from being reflected again for 14 days, and reported
329
+ `quality gate rejected: score=-1` with a reason saying the revision had been
330
+ routed to review. Reflect now creates the proposal and leaves it pending for
331
+ a person, deferred by the quality gate with the reason `judge-error`, as
332
+ distill does when its judge fails. The triage drain leaves it alone. It has
333
+ not been through the retrieval regression check, which runs only on a
334
+ revision the judge passes. A revision the judge scores too low is still
335
+ refused.
336
+ - **The triage drain leaves every proposal a stage deferred for review to a
337
+ person.** It skipped a deferred proposal only when the quality gate had
338
+ deferred it. Reflect defers with its own gate, `reflect`: a revision whose
339
+ size the size guard flagged, one that echoed the truncation notice, one made
340
+ with no judge configured, and now a body edit. With
341
+ `processes.triage.judgment` enabled, the drain's judgment tier decided those
342
+ proposals, and under `applyMode: promote` it could accept them before anyone
343
+ saw them. The drain now skips every deferral it did not make itself, and
344
+ judges again, as before, the ones it did.
345
+
7
346
  ## [0.9.23] - 2026-10-01
8
347
 
9
348
  ### Added
@@ -3,7 +3,6 @@
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  import { spawnSync } from "node:child_process";
5
5
  import { loadConfig } from "../../core/config/config.js";
6
- import { IMPROVE_PROCESS_ENGINE_CAPABILITIES } from "../../core/config/engine-semantics.js";
7
6
  import { listEnvsRecursive } from "../../core/env-secret-ref.js";
8
7
  import { ConfigError } from "../../core/errors.js";
9
8
  import { EXTRACT_INFRASTRUCTURE_SKIP_REASONS } from "../../core/improve-types.js";
@@ -15,7 +14,7 @@ import { loadModelMap, mergeModelMapLayers, parseModelMapLayer, readInstalledMod
15
14
  import { probeEndpointOnce } from "../../llm/client.js";
16
15
  import { STATE_DB_FREELIST_WARN_RATIO, } from "../../storage/state-db-integrity.js";
17
16
  import { listKeys } from "../env/env.js";
18
- import { resolveImprovePlan } from "../improve/improve-strategies.js";
17
+ import { MODEL_CALLING_PROCESSES, resolveImprovePlan } from "../improve/improve-strategies.js";
19
18
  import { ENGINE_LAST_USED_LOOKBACK_DAYS } from "./engine-usage.js";
20
19
  import { ACTIVE_RUN_WARN_MS, TASK_FAIL_RATE_WARN, } from "./types.js";
21
20
  /** Probe one connection's reachability, once per endpoint; `undefined` when no probe seam is supplied. */
@@ -142,7 +141,7 @@ async function runConfiguredEngineProbe(checkName, engineName, config, deps, rea
142
141
  evidence: { engine: engineName, runtimeKind: "sdk", binaryAvailable: false },
143
142
  };
144
143
  }
145
- const fallbackEngine = configuredEngine.llmEngine ?? config.defaults?.llmEngine;
144
+ const fallbackEngine = configuredEngine.llmEngine;
146
145
  let fallback;
147
146
  let fallbackCredential;
148
147
  let fallbackApiKeyFile;
@@ -523,17 +522,17 @@ export function probeActiveImproveStrategy(deps = {}) {
523
522
  .sort(([a], [b]) => a.localeCompare(b))
524
523
  .map(([process, engine]) => `${process}: "${engine}"`)
525
524
  .join(", ");
526
- // #957: fail only when the strategy's LLM-backed work would be a total
527
- // no-op — every process the strategy actually enabled among the
528
- // `capability: "llm"` set (see IMPROVE_PROCESS_ENGINE_CAPABILITIES) ended
529
- // up unavailable. A partial failure (some processes still have a working
525
+ // #957: fail only when the strategy's model work would be a total no-op —
526
+ // every process the strategy actually enabled among the ones that call a
527
+ // model themselves (MODEL_CALLING_PROCESSES, any engine kind) ended up
528
+ // unavailable. A partial failure (some processes still have a working
530
529
  // engine) stays a `warn`, matching #914's "a credential warn stays a warn"
531
530
  // policy for the general per-engine probes; this is the strategy-scoped
532
531
  // "is the whole run a no-op" question instead.
533
- const llmProcessNames = Object.keys(IMPROVE_PROCESS_ENGINE_CAPABILITIES).filter((name) => IMPROVE_PROCESS_ENGINE_CAPABILITIES[name] === "llm");
534
- const requiredLlmProcessNames = llmProcessNames.filter((name) => plan.processes[name].enabled || plan.engineUnavailable.some((item) => item.process === name));
535
- const availableLlmProcessNames = llmProcessNames.filter((name) => plan.processes[name].enabled);
536
- const allRequiredUnavailable = requiredLlmProcessNames.length > 0 && availableLlmProcessNames.length === 0;
532
+ const modelProcessNames = [...MODEL_CALLING_PROCESSES];
533
+ const requiredModelProcessNames = modelProcessNames.filter((name) => plan.processes[name].enabled || plan.engineUnavailable.some((item) => item.process === name));
534
+ const availableModelProcessNames = modelProcessNames.filter((name) => plan.processes[name].enabled);
535
+ const allRequiredUnavailable = requiredModelProcessNames.length > 0 && availableModelProcessNames.length === 0;
537
536
  const status = unavailableProcesses.length === 0 ? "pass" : allRequiredUnavailable ? "fail" : "warn";
538
537
  return {
539
538
  check: {
@@ -36,6 +36,7 @@ import { concurrentMap } from "../../../core/concurrent.js";
36
36
  import { parseEmbeddedJsonResponse } from "../../../core/parse.js";
37
37
  import { DERIVED_SUFFIX } from "../../../core/recognition-util.js";
38
38
  import { warnOnce } from "../../../core/warn.js";
39
+ import { runnerLlmConnection } from "../../../integrations/agent/runner.js";
39
40
  import { assertRunnerCredentials } from "../../../integrations/agent/runner-dispatch.js";
40
41
  import { runGit } from "../../../sources/providers/git-install.js";
41
42
  import { closeDatabase, openExistingDatabase, openReadonlyExistingDatabase, } from "../../../storage/repositories/index-connection.js";
@@ -501,7 +502,7 @@ async function judgeOne(ctx, candidate) {
501
502
  request: {
502
503
  responseSchema: PAIR_JUDGE_JSON_SCHEMA,
503
504
  enableThinking: false,
504
- timeoutMs: ctx.llmRunner.timeoutMs,
505
+ ...(Object.hasOwn(ctx.llmRunner, "timeoutMs") ? { timeoutMs: ctx.llmRunner.timeoutMs } : {}),
505
506
  signal: ctx.opts.signal,
506
507
  ...(ctx.chat ? { chat: ctx.chat } : {}),
507
508
  },
@@ -747,7 +748,7 @@ seams = {}) {
747
748
  // entirely and needs no credential.
748
749
  if (!seams.chat)
749
750
  assertRunnerCredentials(llmRunner);
750
- const results = await concurrentMap(judgeable, (candidate) => judgeOne(ctx, candidate), llmRunner.connection.concurrency ?? 1, { signal: opts.signal });
751
+ const results = await concurrentMap(judgeable, (candidate) => judgeOne(ctx, candidate), runnerLlmConnection(llmRunner)?.concurrency ?? 1, { signal: opts.signal });
751
752
  results.forEach((r, idx) => {
752
753
  // Must-fix 3: `concurrentMap` leaves an entry `undefined` for a call an
753
754
  // aborted run never sent at all — `r?.failed === true` reads that as
@@ -36,6 +36,7 @@ import { resolveWriteTarget } from "../../core/write-source.js";
36
36
  import { deriveInstallations } from "../../indexer/installations.js";
37
37
  import { resolveSourceEntries } from "../../indexer/search/search-source.js";
38
38
  import { USAGE_EVENT_RETENTION_DAYS } from "../../indexer/usage/usage-events.js";
39
+ import { runnerLlmConnection } from "../../integrations/agent/runner.js";
39
40
  import { assertRunnerCredentials } from "../../integrations/agent/runner-dispatch.js";
40
41
  import { cosineSimilarity, embedBatch, resolveEmbeddingModelId } from "../../llm/embedder.js";
41
42
  import { getBodyEmbeddings, upsertBodyEmbeddings } from "../../storage/repositories/embeddings-repository.js";
@@ -566,7 +567,7 @@ async function judgeConsolidationChunks(args) {
566
567
  request: {
567
568
  responseSchema: CONSOLIDATE_PLAN_JSON_SCHEMA,
568
569
  enableThinking: false,
569
- timeoutMs: llmRunner.timeoutMs,
570
+ ...(Object.hasOwn(llmRunner, "timeoutMs") ? { timeoutMs: llmRunner.timeoutMs } : {}),
570
571
  signal: opts.signal,
571
572
  },
572
573
  ...(opts.onNotices ? { onNotices: opts.onNotices } : {}),
@@ -619,7 +620,7 @@ async function planConsolidation(opts, config, stashDir, memories, warnings, sta
619
620
  const llmRunner = opts.llmRunner ?? undefined;
620
621
  // 500 body chars per memory keep the judgement useful; chunk size varies instead.
621
622
  const bodyTruncation = 500;
622
- const chunkSize = computeSafeChunkSize(llmRunner?.connection.contextLength ?? DEFAULT_CONTEXT_LENGTH_TOKENS, bodyTruncation, opts.maxChunkSize);
623
+ const chunkSize = computeSafeChunkSize((llmRunner && runnerLlmConnection(llmRunner)?.contextLength) ?? DEFAULT_CONTEXT_LENGTH_TOKENS, bodyTruncation, opts.maxChunkSize);
623
624
  const sourceName = opts.target ?? stashDir;
624
625
  let budgeted = memories;
625
626
  const budgetMs = opts.signal?.remainingBudgetMs;
@@ -2,6 +2,7 @@
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  import { deepMergeConfig } from "../../core/config/deep-merge.js";
5
+ import { MODEL_WORK_TOOLS } from "../../execution/source.js";
5
6
  import { buildExecution, resolveExecution } from "../../integrations/agent/execution.js";
6
7
  function own(value, key) {
7
8
  return value !== undefined && Object.hasOwn(value, key);
@@ -22,6 +23,8 @@ function mergeDefaults(farther, nearer) {
22
23
  /**
23
24
  * Resolve improve-owned model work through the canonical execution cascade:
24
25
  * defaults.llmEngine -> strategy -> index.<pass> -> process -> current invocation.
26
+ * The engine may be of any kind that confines the model-work tool policy: one
27
+ * that cannot, chosen with `--engine` say, is refused here, before any work.
25
28
  */
26
29
  export function resolveImproveExecution(options) {
27
30
  const defaultEngine = options.config.defaults?.llmEngine;
@@ -33,7 +36,7 @@ export function resolveImproveExecution(options) {
33
36
  if (selectedEngine === undefined || selectedEngine === null)
34
37
  return null;
35
38
  const invocationDefaults = mergeDefaults(mergeDefaults(defaultEngine ? { engine: defaultEngine } : {}, profileDefaults), indexDefaults);
36
- const current = mergeDefaults(processDefaults, currentDefaults);
39
+ const current = { ...mergeDefaults(processDefaults, currentDefaults), tools: MODEL_WORK_TOOLS };
37
40
  const prepared = resolveExecution({
38
41
  content: `improve ${options.processName} execution selection`,
39
42
  config: options.config,
@@ -43,12 +46,3 @@ export function resolveImproveExecution(options) {
43
46
  const lowered = buildExecution(prepared.request, prepared.runner);
44
47
  return Object.freeze({ runner: lowered.runner, notices: lowered.notices });
45
48
  }
46
- export function resolveImproveLlmExecution(options) {
47
- const resolved = resolveImproveExecution(options);
48
- if (!resolved)
49
- return null;
50
- if (resolved.runner.kind !== "llm") {
51
- return null;
52
- }
53
- return { runner: resolved.runner, notices: resolved.notices };
54
- }
@@ -9,9 +9,9 @@
9
9
  * session data into the markdown template loaded from
10
10
  * `src/assets/prompts/extract-session.md`.
11
11
  *
12
- * The schema is intentionally strict — providers with `supportsJsonSchema:
13
- * true` enforce shape upstream, so the parser only has to handle the
14
- * happy path. `additionalProperties: false` means any hallucinated keys
12
+ * The schema is intentionally strict — a provider that honours
13
+ * `response_format` enforces shape upstream, so the parser only has to handle
14
+ * the happy path. `additionalProperties: false` means any hallucinated keys
15
15
  * the model emits get dropped before we parse.
16
16
  */
17
17
  import promptTemplate from "../../assets/prompts/extract-session.md" with { type: "text" };
@@ -20,7 +20,7 @@ const EXTRACT_CANDIDATE_NAME_PATTERN = "^[a-z0-9](?:[a-z0-9-]*[a-z0-9])?(?:/[a-z
20
20
  const EXTRACT_CANDIDATE_NAME_RE = new RegExp(EXTRACT_CANDIDATE_NAME_PATTERN);
21
21
  /**
22
22
  * JSON Schema for the structured extract output. Passed to `chatCompletion`
23
- * when the configured LLM connection has `supportsJsonSchema: true`.
23
+ * unless the configured LLM connection sets `supportsJsonSchema: false`.
24
24
  *
25
25
  * Shape:
26
26
  * {
@@ -32,16 +32,15 @@ import { indexWrittenAssets } from "../../indexer/index-written-assets.js";
32
32
  import { assertRunnerCredentials } from "../../integrations/agent/runner-dispatch.js";
33
33
  import { getAvailableHarnesses } from "../../integrations/session-logs/index.js";
34
34
  import { preFilterSession } from "../../integrations/session-logs/pre-filter.js";
35
- import { isJsonSchemaKnownUnsupported } from "../../llm/client.js";
36
35
  import { getExtractedSessionsMap, getLastExtractRunAt, shouldSkipAlreadyExtractedSession, upsertExtractedSession, } from "../../storage/repositories/extract-sessions-repository.js";
37
36
  import { openSqliteReadSnapshot } from "../../storage/sqlite-read-snapshot.js";
38
37
  import { contentHash } from "./content-hash.js";
39
- import { resolveImproveLlmExecution } from "./execution.js";
38
+ import { resolveImproveExecution } from "./execution.js";
40
39
  import { buildExtractPrompt, EXTRACT_JSON_SCHEMA, parseExtractPayload, } from "./extract-prompt.js";
41
40
  import { cloneAndFreeze, resolveImproveStrategy, resolveProcessEnabled } from "./improve-strategies.js";
42
41
  import { isLedgerBlocked, ledgerKey, loadLedgerSnapshot } from "./ledger.js";
43
42
  import { buildSessionSummaryPrompt, parseSessionSummary, SESSION_SUMMARY_JSON_SCHEMA, sessionMeetsDurationGate, writeSessionAsset, } from "./session-asset.js";
44
- import { callStage, mintProposal, noticeSet } from "./stage.js";
43
+ import { callStage, callStageOnce, mintProposal, noticeSet } from "./stage.js";
45
44
  /** Minimum session duration (minutes) for writing a session asset. */
46
45
  const DEFAULT_MIN_SESSION_DURATION_MINUTES = 5;
47
46
  /** Raw session size (chars) below which the LLM call is skipped; only truly empty sessions are safe to skip. */
@@ -119,7 +118,7 @@ export function resolveStandaloneExtractPlan(config, selection) {
119
118
  }
120
119
  const selected = resolveImproveStrategy(selection.strategy, config);
121
120
  const process = cloneAndFreeze(getImproveProcessConfig("extract", selected.config) ?? {});
122
- const resolved = resolveImproveLlmExecution({
121
+ const resolved = resolveImproveExecution({
123
122
  config,
124
123
  profile: selected.config,
125
124
  process,
@@ -130,7 +129,7 @@ export function resolveStandaloneExtractPlan(config, selection) {
130
129
  processName: "extract",
131
130
  });
132
131
  if (!resolved) {
133
- throw new ConfigError("No LLM engine configured for extract. Set defaults.llmEngine, pass --engine, or select an improve strategy with processes.extract.engine.", "LLM_NOT_CONFIGURED");
132
+ throw new ConfigError("No engine configured for extract. Set defaults.llmEngine, pass --engine, or select an improve strategy with processes.extract.engine.", "LLM_NOT_CONFIGURED");
134
133
  }
135
134
  const runner = resolved.runner;
136
135
  return Object.freeze({
@@ -358,15 +357,16 @@ function planExtractSessions(args) {
358
357
  }
359
358
  const EXTRACT_LLM_UNAVAILABLE = Symbol("extract-llm-unavailable");
360
359
  /**
361
- * One session's extraction call. A connection without structured output gets
362
- * one corrective retry; configuration errors escape before any state is written.
360
+ * One session's extraction call, with one corrective retry; configuration
361
+ * errors escape before any state is written.
363
362
  */
364
363
  async function extractFromSession(run, prompt) {
365
364
  const { llmRunner } = run;
366
365
  try {
367
366
  const result = await runStructured({
368
367
  dispatch: async (feedback) => {
369
- const outcome = await callStage({
368
+ // This loop parses and repairs the reply itself, so each attempt is one unvalidated dispatch.
369
+ const outcome = await callStageOnce({
370
370
  feature: "session_extraction",
371
371
  runner: llmRunner,
372
372
  prompt: feedback ? `${prompt}\n\n## Corrective output instruction\n\n${feedback}` : prompt,
@@ -388,9 +388,6 @@ async function extractFromSession(run, prompt) {
388
388
  return payload.parseFailure ? undefined : payload;
389
389
  },
390
390
  validate: (payload) => ({ ok: true, value: payload }),
391
- maxAttempts: llmRunner.connection.supportsJsonSchema !== false && !isJsonSchemaKnownUnsupported(llmRunner.connection)
392
- ? 1
393
- : 2,
394
391
  buildFeedback: () => "Your previous response did not contain a valid extraction payload. Respond with ONLY a JSON object matching the requested schema, with a candidates array and no prose or code fences.",
395
392
  });
396
393
  if (result.ok)
@@ -714,13 +711,13 @@ function resolveExtractRun(options, config, process, activeProfile) {
714
711
  llmRunner = options.llmRunner;
715
712
  }
716
713
  else {
717
- const resolved = resolveImproveLlmExecution({ config, profile: activeProfile, process, processName: "extract" });
714
+ const resolved = resolveImproveExecution({ config, profile: activeProfile, process, processName: "extract" });
718
715
  llmRunner = resolved?.runner;
719
716
  if (resolved)
720
717
  notices.add(resolved.notices);
721
718
  }
722
719
  if (!llmRunner) {
723
- throw new ConfigError("No LLM engine configured for extract. Set defaults.llmEngine or improve.strategies.<name>.processes.extract.engine.", "LLM_NOT_CONFIGURED");
720
+ throw new ConfigError("No engine configured for extract. Set defaults.llmEngine or improve.strategies.<name>.processes.extract.engine.", "LLM_NOT_CONFIGURED");
724
721
  }
725
722
  const runner = llmRunner;
726
723
  const timeoutMs = options.resolvedPlan