akm-cli 0.9.25-alpha.1 → 0.9.25-alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/CHANGELOG.md +159 -280
  2. package/dist/cli.js +1 -1
  3. package/dist/commands/improve/consolidate/pair-pass.js +1 -0
  4. package/dist/commands/improve/consolidate.js +7 -2
  5. package/dist/commands/improve/execution.js +2 -3
  6. package/dist/commands/improve/extract.js +1 -0
  7. package/dist/commands/improve/improve-cli.js +33 -1
  8. package/dist/commands/improve/loop-stages.js +3 -0
  9. package/dist/commands/improve/reflect-noise.js +125 -0
  10. package/dist/commands/improve/reflect.js +13 -16
  11. package/dist/commands/improve/retrieval-gate.js +7 -2
  12. package/dist/commands/improve/stage.js +31 -39
  13. package/dist/commands/proposal/drain.js +4 -7
  14. package/dist/commands/proposal/propose.js +2 -11
  15. package/dist/commands/proposal/validators/proposal-quality-validators.js +4 -2
  16. package/dist/commands/read/search-cli.js +0 -38
  17. package/dist/core/config/schema/engines.js +15 -33
  18. package/dist/core/config/schema/improve-processes.js +16 -0
  19. package/dist/core/redaction.js +4 -0
  20. package/dist/core/spawn-env.js +25 -0
  21. package/dist/core/structured.js +1 -1
  22. package/dist/execution/source.js +8 -12
  23. package/dist/integrations/agent/config.js +1 -3
  24. package/dist/integrations/agent/engine-resolution.js +0 -3
  25. package/dist/integrations/agent/execution.js +14 -13
  26. package/dist/integrations/agent/index.js +1 -1
  27. package/dist/integrations/agent/model-map.js +15 -16
  28. package/dist/integrations/agent/profiles.js +2 -2
  29. package/dist/integrations/agent/prompts.js +3 -39
  30. package/dist/integrations/agent/request-lowering.js +9 -7
  31. package/dist/integrations/agent/runner-dispatch.js +25 -31
  32. package/dist/integrations/harnesses/aider/agent-builder.js +1 -2
  33. package/dist/integrations/harnesses/amazonq/agent-builder.js +1 -2
  34. package/dist/integrations/harnesses/claude/agent-builder.js +4 -16
  35. package/dist/integrations/harnesses/codex/agent-builder.js +1 -2
  36. package/dist/integrations/harnesses/ids.js +10 -16
  37. package/dist/integrations/harnesses/opencode/agent-builder.js +14 -24
  38. package/dist/integrations/harnesses/opencode/model-config.js +15 -62
  39. package/dist/integrations/harnesses/opencode/model-work-agent.js +71 -36
  40. package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -6
  41. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +32 -90
  42. package/dist/integrations/harnesses/openhands/agent-builder.js +1 -2
  43. package/dist/integrations/harnesses/pi/agent-builder.js +1 -2
  44. package/dist/llm/feature-gate.js +2 -5
  45. package/dist/llm/index-passes.js +2 -2
  46. package/dist/llm/structured-call.js +5 -5
  47. package/dist/output/shapes/passthrough.js +1 -0
  48. package/dist/scripts/akm-migrate-node.js +170 -194
  49. package/dist/scripts/akm-migrate.js +170 -194
  50. package/dist/workflows/exec/unit-dispatch.js +4 -13
  51. package/docs/reference/cli.md +12 -7
  52. package/docs/reference/configuration.md +78 -82
  53. package/docs/reference/data-and-telemetry.md +2 -3
  54. package/docs/reference/workflow-schema.md +6 -9
  55. package/package.json +1 -1
  56. package/schemas/akm-config.json +108 -36
package/CHANGELOG.md CHANGED
@@ -6,297 +6,176 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.9.25-alpha.2] - 2026-10-03
10
+
11
+ ### Added
12
+
13
+ - **`akm improve judge`** runs reflect's quality judge on one revision, read as
14
+ `{"source", "candidate", "feedback", "ref"}` JSON from stdin, with the engine
15
+ `processes.reflect.qualityGate.engine` names, and prints the verdict. It
16
+ writes nothing: it tests a judge engine on revisions whose right answer you
17
+ know.
18
+
19
+ ### Changed
20
+
21
+ - **Reflect refuses three kinds of defective revision before the judge runs,**
22
+ with the quality gate on or off: one that adds placeholder text ("please
23
+ confirm", "to be confirmed"; not `TODO`, `TBD` or `FIXME`), one that talks about
24
+ its own edit ("the feedback says", "this revision", "the source asset", a
25
+ quoted gate rejection), and one that copies frontmatter into its body (key
26
+ lines such as `sources:` or `updated:` outside code, or a `sources`,
27
+ `xrefs` or `contradictedBy` value). Each rule counts only what the revision
28
+ adds to its source. A hit is a `quality_rejected` refusal with no proposal
29
+ and no judge call, and the event names the rule (`reflectDefect`). Each
30
+ rule's wording is a list that `processes.reflect.defectFilter` can replace:
31
+ `placeholders` and `metaCommentary` are plain phrases, matched as whole words
32
+ in any case, and `frontmatterKeys` are exact key names. A list left out keeps
33
+ its default, and `[]` turns its rule off. On 396 labelled reflect edits the
34
+ default lists hit 22 of 313 bad edits and none of 83 good ones.
35
+ - **The reflect quality judge's rubric names what it kept missing.** Tuned on
36
+ the labelled judge-gate set (plain judge, qwen3.8-27b): a description with a
37
+ sentence split by a stray period or an unbalanced quote is broken text; a
38
+ missing title, description or `when_to_use` is a concrete problem whether or
39
+ not the feedback mentions it, while a bare `type:` or a provenance stamp is
40
+ not; PRESERVATION and QUALITY are judged line by line in the changed region,
41
+ so a fix elsewhere no longer excuses a dropped fact, an invented or
42
+ strengthened claim, a hedge or a placeholder. On a stratified 100-case sample
43
+ the gate passed 31 of 40 good edits instead of 18 and 3 of 60 bad instead of 2.
44
+ - **A reflect quality judge on an agent engine reads only to verify what a
45
+ revision adds.** Its prompt names the revised asset's ref and adds one
46
+ paragraph: read that asset, or one the changed region names, with `akm_show`
47
+ only to check a fact the revision adds or alters, at most twice, and never
48
+ search; before scoring, find each added statement in the asset or the
49
+ feedback (a step or cause that merely seems to follow does not count), each
50
+ source fact in the revision, and each feedback point in a change to the text
51
+ it is about. The plain judge's prompt is unchanged. On the same 100 cases
52
+ (qwen3.8-27b, thinking on) the agent judge passed 38 of 40 good edits and 14
53
+ of 60 bad, against 34 and 12 for the plain judge on that model. Give the
54
+ engine's `llmEngine` `enableThinking: true`: without thinking the model
55
+ looped on tool calls.
56
+ - **A failed reflect reply reads the same on every engine kind:** the parser's
57
+ own message. 0.9.25-alpha.1 named the engine on an agent engine.
58
+ - **Inference reaches opencode only where akm writes opencode's config:** an
59
+ improve process's `llm` overlay on model work's agent, and an `opencode-sdk`
60
+ engine's `llmEngine` fallback model. Set the rest in your opencode config; a
61
+ task's or workflow's `inference` is an `untranslated-field` notice, as before
62
+ 0.9.25-alpha.1. `claude` takes no `--effort`.
63
+ - **An `opencode-sdk` session the dispatch times out on or aborts is aborted on
64
+ the server for every dispatch,** not only model work.
65
+ - **An improve stage retries a reply only when the stage cannot read it,** not
66
+ when it misses the JSON Schema. A reply it can read costs no second call.
67
+ - **opencode model work can read the stash and search it.** The `akm-model-work`
68
+ agent reads, greps and globs in the stash and its working directory, edits
69
+ only in the working directory, and has `akm_search` and `akm_show` from the
70
+ akm-opencode plugin (0.9.21 or later, in your own opencode config). The
71
+ plugin's curation, learning and write gate are off for these dispatches, and
72
+ its state goes to akm's state directory, not `~/.local/state/akm-opencode`.
73
+
74
+ ### Removed
75
+
76
+ - **The agent-engine inference fields,** which 0.9.25-alpha.1 accepted: they
77
+ fail to load again.
78
+ - **The `opencode-sdk` step watcher and the git-repository refusal for model
79
+ work's scratch directory.**
80
+ - **opencode model work's step limit.** At the limit opencode sends a "maximum
81
+ steps" message as a trailing assistant message, which a qwen chat template
82
+ (LM Studio, llama-server) renders as the start of the model's reply: LM
83
+ Studio returned nothing, llama-server returned that message as the answer,
84
+ and the dispatch failed. The dispatch timeout bounds a run.
85
+ - **A prompt builder that nothing called** (`buildSchemaRepairPrompt`).
86
+ - **The `--track-usage` / `--no-track-usage` flag on `akm search`, `akm curate`
87
+ and `akm show`.** A successful read always records its usage event, stamped
88
+ with its source (`user`, `improve`, `task` or `audit`); only `user` events
89
+ feed ranking and eval, so machine reads never skew them. Either spelling now
90
+ fails as an unknown flag.
91
+
92
+ ### Fixed
93
+
94
+ - **The reflect size guard no longer flags a body that does not grow.** The
95
+ expansion ceiling is capped at 25,000 characters, so a source body longer than
96
+ that was flagged `EXCESSIVE_EXPANSION` even when the proposed body was its own
97
+ length (ratio 1.00), and went to review instead of the judge. A body no longer
98
+ than its source is never expansion; one that grows past the cap still is, by
99
+ any amount. The reflect prompt agrees: it told the model its body could be at
100
+ most 25,000 characters even when the source was longer, and now gives such a
101
+ source's own length.
102
+ - **An asset's or a task's own `tools:` can no longer name the model-work
103
+ policy.** In 0.9.25-alpha.1 exactly `read`, `edit`, `akm search`, `akm show`,
104
+ in that order, skipped `execution.allowedTools`. It is ordinary tools now:
105
+ only akm's own model-work callers ask for the policy.
106
+ - **An agent engine that names its model only in `args` is named in the usage
107
+ report,** where it showed `unattributed`.
108
+ - **`opencode` and `opencode-sdk` engines receive the XDG base-directory
109
+ variables.** Under a custom `XDG_CONFIG_HOME` the spawned opencode missed its
110
+ provider config and failed every dispatch with `Unexpected server error`.
111
+
9
112
  ## [0.9.25-alpha.1] - 2026-10-02
10
113
 
11
114
  ### Changed
12
115
 
13
- - **One schema instruction, appended by the shared agent request lowering.**
14
- Seven harness builders and the workflow engine each kept a copy of
15
- "Respond with ONLY a JSON value matching this JSON Schema (no prose, no code
16
- fences)". The lowering now appends it to every agent engine's prompt when a
17
- schema is requested, so `codex` gets it beside `--output-schema` outside
18
- workflows too. A workflow unit sends the same prompt bytes as before, less
19
- the duplicate fixed below; a direct-LLM unit still carries the instruction
20
- in its own prompt.
21
- - **Model work is bounded at 600 seconds on every engine kind.** An improve
22
- process, a quality or triage judge, or an index pass whose engine sets no
23
- `timeoutMs` now stops after 600 seconds. Before, an agent or `opencode-sdk`
24
- engine ran until it finished, and so did memory inference and consolidation
25
- on an LLM engine. An engine's own `timeoutMs` still applies.
116
+ - **Unattended model work runs under one tool policy, on any engine that
117
+ confines it.** The model may read, edit inside a scratch directory akm removes
118
+ after the dispatch, and run `akm search` and `akm show`; the stash stays
119
+ read-only. An LLM engine has no tools, `claude` confines the policy, and
120
+ `opencode` and `opencode-sdk` run an injected `akm-model-work` agent with
121
+ read and edit only. Every other harness refuses it before the request starts.
122
+ - **One config rule names the engines model work may use.** Every key it reads
123
+ its engine from (`defaults.llmEngine`, `index.*.engine`, a strategy's or
124
+ process's `engine`, an enabled triage `judgment.engine`, a
125
+ `qualityGate.engine`) must name an LLM engine or a `claude`, `opencode` or
126
+ `opencode-sdk` agent engine, or the config fails to load. `--require-engines`
127
+ and `akm health` check agent engines too.
128
+ - **Model work is bounded at 600 seconds on every engine kind** whose engine
129
+ sets no `timeoutMs`; agent and `opencode-sdk` engines ran until they finished.
130
+ - **`akm proposal new` returns the proposal as JSON on every engine kind,**
131
+ with no draft file and no live session. A reply that is not a proposal gets
132
+ one retry.
133
+ - **Reflect asks every engine kind for the same JSON reply and repairs it
134
+ once.** An agent or `opencode-sdk` engine gets reflect's JSON Schema and a
135
+ repair turn, as an LLM engine did. An LLM engine's requests are unchanged.
26
136
  - **An improve stage's reply that fails its JSON Schema gets one corrective
27
- retry.** The stage retries once with the validation errors, then reads the
28
- last reply with its own parser as before. `akm extract` now allows its
29
- corrective retry on every engine, not only on one without JSON Schema
30
- support. Reflect keeps its own repair turn.
31
- - **Agent and `opencode-sdk` dispatches leave a usage record.** Each one now
32
- writes one `llm_usage` record through the same sink and stage attribution as
33
- the LLM path, with the request's model and the tokens the runner reports. An
34
- LLM engine still records each HTTP attempt.
35
- - **`akm proposal new` gets the proposal as JSON from every engine kind, with
36
- no live session.** It told every engine to write the asset to a draft file
37
- and print no JSON, then parsed stdout as JSON. An LLM engine that followed
38
- the instruction could not succeed, and an agent engine succeeded only by
39
- writing the file: an agent CLI ran in a live terminal session whose output
40
- akm could not read, and otherwise failed with an "interactive mode" error.
41
- Every engine now returns the proposal as one JSON object on stdout, and the
42
- request carries its JSON Schema: as `response_format` for an LLM engine, as
43
- the schema instruction for an agent engine. akm captures the reply, unwraps
44
- a harness envelope such as claude's `--output-format json` result, and
45
- validates it. A reply that is not a proposal gets one corrective retry, then
46
- fails with an error that names the engine. An agent CLI now runs headless:
47
- you see the queued proposal, not the agent at work, and no draft file is
48
- written.
49
- - **Unattended model work has one tool policy, which each engine confines or
50
- refuses when the request is built.** The policy allows reading, editing only
51
- inside a scratch working directory that akm creates for the dispatch and
52
- removes after it, and running `akm search` and `akm show`. The stash stays
53
- read-only. An LLM engine has no tools. `claude` runs it with `--restricted`
54
- (its user, project and local settings, whose allow rules could pre-approve
55
- any command or path, are ignored, and its file tools stay in the working
56
- directory), `--strict-mcp-config`, `--tools Read,Edit,Bash`, `--allowedTools`
57
- for Read, Edit and the two `akm` commands, and `--permission-mode dontAsk`.
58
- `opencode` and `opencode-sdk` run an injected `akm-model-work` agent that
59
- can read and edit only inside the working directory and has no bash:
60
- opencode checks a bash rule against the command's words only, so
61
- `akm show x > ~/stash/asset.md` would pass an `akm show *` rule. The agent
62
- has its own short prompt in place of opencode's coding-assistant prompt, an
63
- 8-step limit, and no automatic compaction. opencode only asks the model to
64
- stop at that limit, so `opencode-sdk` aborts the session two steps past it,
65
- and aborts a session the dispatch times out on; on `opencode` the dispatch
66
- timeout is the bound. Every other harness refuses the policy. Such a
67
- dispatch ignores the engine's `args` and `workspace`, refuses to start when
68
- the temporary directory is inside a git repository, which opencode would
69
- treat as its working directory, and fails with `parse_error` when the agent
70
- ends with no answer, as opencode can at its step limit.
71
- - **Unattended model work runs on any engine that confines the model-work
72
- tool policy, and config checks that with one rule.**
73
- - **Who sends the policy.** The improve processes, the quality, triage and
74
- retrieval-gate judges, index passes and `akm remember --enrich` now send
75
- it, so they may run on an LLM engine or on a `claude`, `opencode` or
76
- `opencode-sdk` agent engine, where they could only use an LLM engine
77
- before.
78
- - **The one rule.** Every key model work reads its engine from must name
79
- such an engine. The keys are `defaults.llmEngine`, `index.defaults.engine`,
80
- `index.<pass>.engine`, a strategy's `engine`, a process's `engine`, an
81
- enabled triage `judgment.engine`, and a `qualityGate.engine`. A config
82
- that breaks the rule fails to load, naming the key, the engine and its
83
- platform. The rule replaces six checks that each required an LLM engine
84
- for some of those keys, or let the triage judgment use any agent.
85
- - **What else is gone.** The plan, reflect, the quality gate and index
86
- passes no longer turn an agent engine away. A triage judgment on an agent
87
- engine takes a strategy's `llm` overrides as `untranslated-field`
88
- notices, where it was refused before.
89
- - **Reflect on an agent engine.** The agent returns its proposal as JSON on
90
- stdout instead of writing a draft file, which the policy's scratch working
91
- directory would not keep.
92
- - **The answer is unwrapped.** A model-work reply from an agent engine goes
93
- through its harness's result extractor, so a stage call on `claude` gets
94
- the answer, not the `--output-format json` envelope around it.
95
- - **`--require-engines` checks agent engines too.** It checks an agent
96
- engine by its binary on PATH, and an `opencode-sdk` engine by its binary
97
- and its LLM fallback's endpoint. It probed only LLM connections before.
98
- The usage report and `akm health` count a process's calls whatever its
99
- engine's kind.
100
- - **An `opencode-sdk` engine gets its LLM fallback connection only from its own
101
- `llmEngine`. `defaults.llmEngine` no longer supplies one.** An SDK engine
102
- that set no `llmEngine` used to borrow `defaults.llmEngine`: its connection,
103
- its model (unless the engine set its own) and its timeout. `defaults.llmEngine`
104
- names the default engine for unattended model work, and since that work may
105
- now run on an agent engine it can be one too, so it is no longer also a
106
- connection that every SDK engine shares. **If you relied on the inheritance,
107
- set `llmEngine` on the SDK engine**, for example
108
- `"sdk": { "kind": "agent", "platform": "opencode-sdk", "llmEngine": "fast" }`.
109
- Without one, the SDK engine runs on opencode's own provider and auth, and on
110
- its own `model` if it sets one: akm sends it no connection. An SDK engine
111
- that sets `llmEngine` is unchanged, and so is a config with no `opencode-sdk`
112
- engine. The rule holds everywhere the fallback is read: dispatch, a
113
- workflow's frozen concurrency cap, `akm health`, and
114
- `akm improve --require-engines`, which now check an SDK engine's fallback
115
- endpoint only when it sets `llmEngine`.
116
- - **Inference reaches `opencode`, `opencode-sdk` and `claude` engines.** A
117
- request's `temperature`, `maxTokens`, `contextLength`, `enableThinking` and
118
- `reasoningEffort` came from the engine's own settings, an improve process's
119
- `llm` overlay, a task, command or agent asset's `inference`, a workflow's
120
- `llm:` and a `models.json` alias, and every agent engine dropped all of them
121
- with an `untranslated-field` notice. Each platform now translates what it can
122
- carry, and the nearest layer wins, field by field, as on an LLM engine. With
123
- no setting anywhere akm sends nothing of its own, so the model's configured
124
- default still applies, such as a `reasoningEffort: "none"` in your opencode
125
- config.
126
- - **`claude`** gets `reasoningEffort` as `--effort <level>`, passed as given
127
- (Claude Code 2.1.283 takes `low`, `medium`, `high`, `xhigh` and `max`).
128
- Its other four fields are still reported as untranslated.
129
- - **`opencode` and `opencode-sdk`** get inference as opencode config, which
130
- merges over your own opencode config for the same provider and model:
131
- `temperature` and `reasoningEffort` become `options.temperature` and
132
- `options.reasoningEffort` (opencode drops the snake_case spelling),
133
- `enableThinking` becomes both wire forms an LLM engine sends, and
134
- `maxTokens` with `contextLength` becomes `limit.output` and
135
- `limit.context`. Set both limit fields: opencode refuses half a limit, and
136
- a half would overwrite the other half of one you declared, so a lone one
137
- is reported as untranslated. Without `limit.output` opencode asks for
138
- `max_tokens: 32000`, which a small-context server rejects, and a wrong
139
- `limit.context` lets it build requests past the server's window. The fields
140
- need a `provider/model`: the request's model, or the `--model` an
141
- `opencode` engine's `args` name. Model work needs none for its options, so
142
- an `opencode-sdk` engine with no `model` and no `llmEngine` gets them too.
143
- - **Where it goes.** Model work puts the options on the `akm-model-work`
144
- agent that runs it, so opencode's own title call on the same model keeps
145
- the model's defaults. Any other dispatch runs your own agent, whose name
146
- akm cannot rely on, so its options go on the model and the title call sees
147
- them too. `opencode-sdk` makes no title call. An `opencode-sdk` engine
148
- declares its `llmEngine` fallback's model with the fallback's own
149
- inference, under the engine's and the request's; each distinct set starts
150
- its own `opencode serve`, as a different model does.
151
- - **An agent engine may set the inference fields its platform translates.**
152
- `engines.<name>` of `kind: "agent"` took none, so an engine could not carry a
153
- `temperature` or a `reasoningEffort` of its own. `opencode` and
154
- `opencode-sdk` may set `temperature`, `maxTokens`, `contextLength`,
155
- `enableThinking` and `reasoningEffort`, `claude` may set `reasoningEffort`,
156
- and every other platform none. A field its platform does not translate fails
157
- to load, naming the platform and the fields it does translate. `provider`,
158
- `endpoint`, `apiKey`, `apiKeyFile`, `concurrency` and `extraParams` stay
159
- invalid on an agent engine.
160
- - **Reasoning effort has one word in a request, `reasoningEffort`.** The
161
- starter `reasoning` alias, an alias in your `models.json` and an asset's
162
- `effort:` frontmatter said `effort`, and engines, opencode and the LLM
163
- request said `reasoningEffort`. `effort` is now read as `reasoningEffort`
164
- where layers are merged, so the nearest layer wins whichever word it used.
165
- On an LLM engine an alias's or asset's `effort` is therefore sent as
166
- `reasoning_effort`; it was reported as untranslated and dropped before. An
167
- LLM engine's own request is unchanged.
168
- - **Reflect asks every engine kind for the same JSON reply, checks it the same
169
- way and repairs it once.** An LLM engine was sent the reply's JSON Schema,
170
- and its reply was held to exact fields and repaired once. An agent or
171
- `opencode-sdk` engine got a looser contract in its prompt (`ref`, `content`
172
- and an optional `frontmatter`) with no schema and no repair, so one invalid
173
- reply failed the run. Every engine kind now runs the same iteration:
174
- - **The request.** The prompt carries the same output contract, and the
175
- reply's JSON Schema is the request's output schema: `response_format` for
176
- an LLM engine, as before, and the schema instruction at the end of the
177
- prompt for an agent engine. `claude` also gets `--output-format json`,
178
- which akm unwraps. An LLM endpoint that rejects JSON Schema still gets the
179
- framed-markdown contract.
180
- - **The reply.** An agent now returns `content`, `confidence` and a
181
- `frontmatterPatch` of `description` and `when_to_use`, and `ref` too when
182
- no asset was named, as an LLM does. akm derives a named asset's ref and
183
- merges the patch with the source's frontmatter, so an agent can no longer
184
- set other frontmatter keys or retarget the proposal.
185
- - **The repair.** A reply that fails the contract gets one repair turn that
186
- carries the first reply, shared across self-refine passes. A reply that is
187
- still invalid fails with `parse_error` and queues nothing. On an agent or
188
- `opencode-sdk` engine the error names the engine, as in
189
- `Engine "oc" reply was not a valid reflect proposal after 2 attempts: …`.
190
- An LLM engine's message is unchanged, because improve feeds it into later
191
- prompts as a pattern to avoid. A failed agent dispatch is still reported
192
- with its exit code and stderr.
193
- - **What reflect sends and reports on an agent engine.** Reflect asks every
194
- engine kind for no visible chain of thought (`enableThinking: false`).
195
- `opencode` and `opencode-sdk` carry it, as the inference entry above
196
- says; `claude` reports it as an `untranslated-field` notice, as it does
197
- for every other stage. `reflect_completed` carries `outputMode` and
198
- `repairAttempts` for every engine kind.
199
- - **Unchanged.** An LLM engine's requests are byte-identical to before:
200
- reflect's generation and repair, and its quality judge. So are the
201
- refine passes, the content budget (an LLM engine's context length), the
202
- protected frontmatter fields, the review routing and the judge selection.
137
+ retry,** then the stage reads the last reply with its own parser.
138
+ - **One schema instruction for every agent engine,** appended by the shared
139
+ request lowering: `opencode` and `opencode-sdk` now receive a requested
140
+ schema, and a workflow unit on seven harnesses no longer carries it twice.
141
+ - **Agent and `opencode-sdk` dispatches leave a usage record.**
142
+ - **An `opencode-sdk` engine's LLM fallback comes only from its own
143
+ `llmEngine`; `defaults.llmEngine` no longer supplies one.** If you relied on
144
+ that, set `llmEngine` on the SDK engine. Without one, opencode uses its own
145
+ provider, model and auth.
146
+ - **Inference reaches `opencode`, `opencode-sdk` and `claude` engines.**
147
+ `temperature`, `maxTokens`, `contextLength`, `enableThinking` and
148
+ `reasoningEffort` from an engine, an improve process's `llm` overlay, an
149
+ asset, a workflow or a `models.json` alias were dropped on every agent
150
+ engine. An agent engine may set them, and `claude` takes `reasoningEffort` as
151
+ `--effort`.
152
+ - **Reasoning effort has one word, `reasoningEffort`.** `effort` in a
153
+ `models.json` alias or an asset is read as it, so an LLM engine now sends it
154
+ as `reasoning_effort`.
203
155
 
204
156
  ### Removed
205
157
 
206
- - **Two harness metadata fields that nothing read.** Each of the ten harness
207
- descriptors carried an execution `pattern` and a `structuredOutput` tier.
208
- No code branched on either: the shared request lowering appends the schema
209
- instruction to every agent prompt, and a harness's own argv builder adds its
210
- native channel, as codex does with `--output-schema`. The fields, their two
211
- types and the tests that pinned their values are gone. Nothing you run
212
- changes.
213
- - **`resolveLlmEngineUse`'s swap of an agent engine for an LLM engine.** Given
214
- an agent engine, it used the engine's `llmEngine`, then `defaults.llmEngine`,
215
- and warned. Only the implicit SDK fallback above could reach it, so nothing
216
- you run changes. An agent engine passed to it is now an error.
217
- - **An `effort` hint on the agent dispatch request that nothing read.** The
218
- lowering set it from `inference.effort`, reserved for a workflow field, and
219
- no builder consumed it. `reasoningEffort` in the request's inference is read
220
- where it is translated, and the field is gone.
221
- - **The agent file-write contract.** An agent was once told to write its
222
- proposal to a draft file and print `DRAFT_WRITTEN confidence=<n>`. Reflect and
223
- `akm proposal new` had already stopped sending that instruction, because the
224
- model-work scratch directory does not outlive a dispatch. The instruction, the
225
- function that read the `DRAFT_WRITTEN` line and the `draftFilePath` prompt
226
- inputs that nothing passed any more are gone, with their tests. So is
227
- reflect's `ref_mismatch` check: a reflect that names an asset derives the
228
- proposal's ref, so an engine can no longer name another one. Nothing you run
229
- changes.
158
+ - **Dead code; nothing you run changes:** each harness's unused `pattern` and
159
+ `structuredOutput` fields, `resolveLlmEngineUse`'s swap of an agent engine
160
+ for an LLM engine, the agent request's unread `effort` hint, and the agent
161
+ file-write contract (`DRAFT_WRITTEN`, reflect's `ref_mismatch` check).
230
162
 
231
163
  ### Fixed
232
164
 
233
- - **An `opencode-sdk` engine with an LLM fallback now reaches its endpoint,
234
- and a failed dispatch is reported as a failure (#1015).** The `akm-custom`
235
- provider that akm generates for the fallback listed no models, so opencode
236
- could not find the model and failed every dispatch. akm now declares the
237
- routed model under the provider's `models`. The runner also ignored the
238
- error that the SDK client returns for an HTTP error, and the error opencode
239
- puts on a reply when the provider rejects a request, and reported
240
- `ok: true` with empty output. Both now give `ok: false`, with opencode's
241
- error name and message in `error` and `stderr`. An aborted message is
242
- `aborted`, a reply cut off at the output limit is `parse_error`, and any
243
- other error is `non_zero_exit`. A reply with several text parts now returns
244
- the last one, which is the answer, instead of the first.
245
- - **An LLM engine reports a provider error sent with HTTP 200 as a failure.**
246
- OpenRouter, for one, answers a request whose provider fails after the
247
- response has started with HTTP 200 and a body that holds only an `error`
248
- object and no `choices`. akm returned that as an empty reply with
249
- `ok: true`. It is now a provider error with the body in its message, as an
250
- error status is.
251
- - **An `opencode` engine runs a persona instead of failing.** akm passed the
252
- persona to `opencode run` as `--system-prompt`, which opencode 1.18 does
253
- not accept, so opencode printed its usage and exited 1 on every dispatch
254
- that carried a persona, including `akm agent <agent-ref> --engine opencode`.
255
- akm now composes the persona into the prompt in an `<AKM_PERSONA>` block,
256
- as it does for harnesses with no system-prompt option.
257
- - **`opencode` and `opencode-sdk` engines receive a requested output schema.**
258
- They dropped it with only an `untranslated-field` warning, so a schema from
259
- `akm agent`, `akm command run`, a command's frontmatter or a task's
260
- `output:` never reached the model. They now get it as the same instruction
261
- every other agent engine gets.
262
- - **An LLM engine sends a requested schema unless it opts out.** It sent
263
- `response_format` only when the engine set `supportsJsonSchema: true`, so an
264
- engine that left the flag unset never had its output constrained. It now
265
- sends it unless the engine sets `supportsJsonSchema: false`. An endpoint
266
- that rejects it with a 4xx is still retried once without it.
267
- - **A workflow unit on `claude`, `copilot`, `gemini`, `pi`, `aider`,
268
- `amazonq` or `openhands` gets the schema instruction once.** The unit prompt
269
- carried it and the harness builder appended a second copy.
270
- - **A feature gate's timeout now stops the call it bounds.** When an improve
271
- stage's gate timed out (600 seconds unless the call sets its own), the model
272
- call kept running in the background, on an engine with `timeoutMs: 900000`
273
- for up to five more minutes. The gate now aborts it.
274
- - **A stage call reports a timeout or abort by the dispatch's own reason.** A
275
- timed-out or aborted agent or `opencode-sdk` dispatch, and an LLM timeout
276
- inside a feature gate, came back as `error`. They now come back as `timeout`
277
- or `aborted`.
278
-
279
- - **`akm proposal new` keeps the reply's confidence.** The engine's
280
- self-rated `confidence` was parsed and then dropped, so a proposal from
281
- `proposal new` never carried the field the reference says it has.
282
- - **Model work on an agent or `opencode-sdk` engine that sets no `timeoutMs` now
283
- stops after 600 seconds, as documented.** Such an engine resolved to an
284
- explicit "no timeout", so the 600-second bound for model work never applied
285
- to it. An engine's own `timeoutMs`, `null` included, still applies, and
286
- other work on an agent engine still runs until it finishes.
287
- - **The announcement of the implicit `opencode-sdk` fallback is now true.** It
288
- says provider, model and auth come from opencode's own configuration, but the
289
- fallback engine borrowed `defaults.llmEngine`'s connection whenever one was
290
- set, so opencode got an akm-generated provider and model instead. It now runs
291
- on opencode's own configuration, as announced.
292
- - **`model: reasoning` set no effort on `claude`, `opencode` or `opencode-sdk`.**
293
- The starter alias supplies `effort: high` for all three, and each dropped it
294
- with an `untranslated-field` notice, so the alias chose a stronger model and
295
- nothing more. It now sets `--effort high` on `claude` and
296
- `options.reasoningEffort` on opencode (see "Inference reaches `opencode`,
297
- `opencode-sdk` and `claude` engines" above). An improve process's
298
- `llm.reasoningEffort` and `llm.temperature` overlay now reaches those
299
- engines the same way.
165
+ - **An `opencode-sdk` engine with an LLM fallback reaches its endpoint, and a
166
+ failed dispatch is a failure (#1015).** An SDK error or provider rejection is
167
+ `ok: false` with opencode's message, and a reply's last text part is its answer.
168
+ - **An LLM engine reports a provider error sent with HTTP 200 as a failure**
169
+ (OpenRouter does this), not as an empty reply.
170
+ - **An `opencode` engine runs a persona** instead of failing on
171
+ `--system-prompt`, which opencode 1.18 rejects.
172
+ - **An LLM engine sends a requested schema** unless it sets
173
+ `supportsJsonSchema: false`; it sent one only when it set `true`.
174
+ - **A feature gate's timeout stops the call it bounds,** and a stage call
175
+ reports a timeout or abort as `timeout` or `aborted`, not `error`.
176
+ - **`akm proposal new` keeps the reply's `confidence`.**
177
+ - **Model work on an agent or `opencode-sdk` engine with no `timeoutMs` stops
178
+ after 600 seconds** as documented; it resolved to no timeout.
300
179
 
301
180
  ## [0.9.24] - 2026-10-02
302
181
 
package/dist/cli.js CHANGED
@@ -177,7 +177,7 @@ const setupCommand = defineCommand({
177
177
  // the work, matching the `sync --push/--no-push` pattern. A flag
178
178
  // DECLARED as `no-init` can never be negated: `--no-init` parses as
179
179
  // "negate `init`", a name nothing declared, leaving the real key at its
180
- // default forever — see `search --no-track-usage`'s identical fix.
180
+ // default forever.
181
181
  init: {
182
182
  type: "boolean",
183
183
  default: true,
@@ -506,6 +506,7 @@ async function judgeOne(ctx, candidate) {
506
506
  signal: ctx.opts.signal,
507
507
  ...(ctx.chat ? { chat: ctx.chat } : {}),
508
508
  },
509
+ parse: parsePairJudgeResponse,
509
510
  ...(ctx.opts.onNotices ? { onNotices: ctx.opts.onNotices } : {}),
510
511
  });
511
512
  if (!outcome.ok)
@@ -53,6 +53,10 @@ import { resolveImproveStrategy, resolveProcessEnabled } from "./improve-strateg
53
53
  import { isContentDrivenRow, isLedgerBlocked, ledgerKey, loadLedgerSnapshot, recordLedgerAttempt } from "./ledger.js";
54
54
  import { isInRetrievalScope, loadRetrievalScope } from "./retrieval-scope.js";
55
55
  import { callStage, mintProposal, noticeSet, stageRunner } from "./stage.js";
56
+ function parsePlan(raw) {
57
+ const plan = parseEmbeddedJsonResponse(raw);
58
+ return plan && Array.isArray(plan.operations) ? { ...plan, operations: plan.operations } : undefined;
59
+ }
56
60
  /** A plan op worth acting on. Retired advisory ops (merge/delete/contradict) are dropped, never thrown on. */
57
61
  export function isValidOp(op) {
58
62
  if (typeof op !== "object" || op === null)
@@ -570,6 +574,7 @@ async function judgeConsolidationChunks(args) {
570
574
  ...(Object.hasOwn(llmRunner, "timeoutMs") ? { timeoutMs: llmRunner.timeoutMs } : {}),
571
575
  signal: opts.signal,
572
576
  },
577
+ parse: parsePlan,
573
578
  ...(opts.onNotices ? { onNotices: opts.onNotices } : {}),
574
579
  });
575
580
  if (!outcome.ok) {
@@ -577,8 +582,8 @@ async function judgeConsolidationChunks(args) {
577
582
  continue;
578
583
  }
579
584
  warnVerbose(`[akm:consolidate] ${label} raw response (first 500 chars): ${outcome.raw.slice(0, 500)}`);
580
- const parsed = parseEmbeddedJsonResponse(outcome.raw);
581
- if (!parsed || !Array.isArray(parsed.operations)) {
585
+ const parsed = parsePlan(outcome.raw);
586
+ if (!parsed) {
582
587
  const hint = outcome.raw.trim() === "" ? " (empty response — if using a thinking model, disable thinking mode)" : "";
583
588
  const msg = `Chunk ${chunkIdx + 1}: invalid plan from AI — skipping.${hint}`;
584
589
  warn(msg);
@@ -2,7 +2,6 @@
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  import { deepMergeConfig } from "../../core/config/deep-merge.js";
5
- import { MODEL_WORK_TOOLS } from "../../execution/source.js";
6
5
  import { buildExecution, resolveExecution } from "../../integrations/agent/execution.js";
7
6
  function own(value, key) {
8
7
  return value !== undefined && Object.hasOwn(value, key);
@@ -36,12 +35,12 @@ export function resolveImproveExecution(options) {
36
35
  if (selectedEngine === undefined || selectedEngine === null)
37
36
  return null;
38
37
  const invocationDefaults = mergeDefaults(mergeDefaults(defaultEngine ? { engine: defaultEngine } : {}, profileDefaults), indexDefaults);
39
- const current = { ...mergeDefaults(processDefaults, currentDefaults), tools: MODEL_WORK_TOOLS };
40
38
  const prepared = resolveExecution({
41
39
  content: `improve ${options.processName} execution selection`,
42
40
  config: options.config,
43
41
  invocationDefaults,
44
- current,
42
+ current: mergeDefaults(processDefaults, currentDefaults),
43
+ modelWork: true,
45
44
  });
46
45
  const lowered = buildExecution(prepared.request, prepared.runner);
47
46
  return Object.freeze({ runner: lowered.runner, notices: lowered.notices });
@@ -741,6 +741,7 @@ function resolveExtractRun(options, config, process, activeProfile) {
741
741
  ...(options.signal ? { signal: options.signal } : {}),
742
742
  ...(options.chat ? { chat: options.chat } : {}),
743
743
  },
744
+ parse: parseSessionSummary,
744
745
  onNotices: notices.add,
745
746
  });
746
747
  return parseSessionSummary(outcome.ok ? outcome.raw : "");
@@ -20,13 +20,15 @@ import { collectEngineCredentialValues } from "../../integrations/agent/engine-r
20
20
  import { probeLlmReachable } from "../../llm/client.js";
21
21
  import { getOutputMode } from "../../output/context.js";
22
22
  import { deliverRendered } from "../../output/html-render.js";
23
+ import { readStdin } from "../../runtime.js";
23
24
  import { akmImprove, IMPROVE_TARGET_FLAG, resolveImproveReadSource } from "./improve.js";
24
25
  import { runImproveReportQuery } from "./improve-report.js";
25
26
  import { buildImproveRunId, recordImproveRunResult, recordTerminatedImproveRun, } from "./improve-result-file.js";
26
27
  import { runImproveSession } from "./improve-session.js";
27
- import { resolveImprovePlan, } from "./improve-strategies.js";
28
+ import { resolveImprovePlan, resolveImproveStrategy, } from "./improve-strategies.js";
28
29
  import { formatUsageReportTable } from "./improve-usage-report.js";
29
30
  import { renderReflectPromptPreview } from "./reflect.js";
31
+ import { resolveQualityGateJudge, runReflectQualityJudge } from "./stage.js";
30
32
  let akmImproveForRun = akmImprove;
31
33
  /** Swap the CLI's improve work implementation in deterministic subprocess tests. */
32
34
  export function _setAkmImproveForTests(fake) {
@@ -188,6 +190,32 @@ function rejectReportOnlyFlags(args) {
188
190
  return;
189
191
  throw new UsageError(`\`${flag}\` only applies to \`akm improve report\`. Use \`akm improve report ${flag} <value>\` instead.`, "INVALID_FLAG_VALUE");
190
192
  }
193
+ /**
194
+ * `akm improve judge`: reflect's quality judge on one revision, read as
195
+ * `{"source", "candidate", "feedback"}` JSON from stdin, with the engine the
196
+ * strategy's reflect quality gate names. It writes nothing.
197
+ */
198
+ async function runImproveJudgeCli(strategyName) {
199
+ const input = process.stdin.isTTY
200
+ ? {}
201
+ : JSON.parse((await readStdin()).toString("utf8"));
202
+ const { source, candidate, feedback, ref } = input;
203
+ if (typeof source !== "string" || typeof candidate !== "string") {
204
+ throw new UsageError('`akm improve judge` reads {"source": "...", "candidate": "...", "feedback": "...", "ref": "..."} JSON from stdin.', "MISSING_REQUIRED_ARGUMENT");
205
+ }
206
+ const config = loadConfig();
207
+ const judge = resolveQualityGateJudge(config, resolveImproveStrategy(strategyName, config).config, "reflect");
208
+ if (!judge) {
209
+ throw new ConfigError("`akm improve judge` judges with the reflect quality gate's engine. Set processes.reflect.qualityGate.engine.", "INVALID_CONFIG_FILE");
210
+ }
211
+ const notes = typeof feedback === "string" && feedback.trim() !== "" ? [feedback.trim()] : [];
212
+ const verdict = await runReflectQualityJudge(config, candidate, source, notes, undefined, {
213
+ runnerSelectionFrozen: true,
214
+ llmRunner: judge,
215
+ ...(typeof ref === "string" && ref ? { ref } : {}),
216
+ });
217
+ output("improve-judge", { engine: judge.engine, ...verdict });
218
+ }
191
219
  export const improveCommand = defineCommand({
192
220
  meta: {
193
221
  name: "improve",
@@ -270,6 +298,10 @@ export const improveCommand = defineCommand({
270
298
  return;
271
299
  }
272
300
  rejectReportOnlyFlags(args);
301
+ if (getStringArg(args, "scope") === "judge") {
302
+ await runImproveJudgeCli(getStringArg(args, "strategy"));
303
+ return;
304
+ }
273
305
  rejectRetiredImproveTargetFlag();
274
306
  const jsonToStdout = args["json-to-stdout"];
275
307
  const targetArg = getStringArg(args, "bundle");
@@ -149,6 +149,9 @@ async function runLoopReflectPass(planned, env, tally) {
149
149
  ...(reflectErrors.length > 0 ? { avoidPatterns: [...reflectErrors] } : {}),
150
150
  eventSource: "improve",
151
151
  lowValueFilter: improveProfile.processes?.reflect?.lowValueFilter?.enabled === true,
152
+ ...(improveProfile.processes?.reflect?.defectFilter
153
+ ? { defectFilter: improveProfile.processes.reflect.defectFilter }
154
+ : {}),
152
155
  ...(budgetMs > 0 ? { timeoutMs: budgetMs } : {}),
153
156
  signal: env.budgetSignal,
154
157
  eventsCtx: env.eventsCtx,