akm-cli 0.9.25-alpha.1 → 0.9.25-alpha.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/CHANGELOG.md +243 -280
  2. package/dist/assets/prompts/reflect-feedback-framing.md +1 -1
  3. package/dist/assets/prompts/reflect-llm-framed-contract.md +2 -9
  4. package/dist/assets/prompts/reflect-llm-schema-contract.md +1 -3
  5. package/dist/assets/prompts/reflect-output-repair.md +1 -1
  6. package/dist/cli.js +1 -1
  7. package/dist/commands/improve/consolidate/pair-pass.js +1 -0
  8. package/dist/commands/improve/consolidate.js +7 -2
  9. package/dist/commands/improve/execution.js +2 -3
  10. package/dist/commands/improve/extract-cli.js +3 -2
  11. package/dist/commands/improve/extract.js +2 -1
  12. package/dist/commands/improve/improve-cli.js +33 -1
  13. package/dist/commands/improve/loop-stages.js +3 -0
  14. package/dist/commands/improve/reflect-noise.js +125 -0
  15. package/dist/commands/improve/reflect.js +150 -333
  16. package/dist/commands/improve/retrieval-gate.js +7 -2
  17. package/dist/commands/improve/session-asset.js +6 -0
  18. package/dist/commands/improve/stage.js +31 -39
  19. package/dist/commands/proposal/drain.js +4 -7
  20. package/dist/commands/proposal/propose.js +2 -11
  21. package/dist/commands/proposal/validators/proposal-quality-validators.js +11 -5
  22. package/dist/commands/proposal/validators/proposal-validators.js +4 -5
  23. package/dist/commands/read/search-cli.js +0 -38
  24. package/dist/core/asset/asset-serialize.js +1 -1
  25. package/dist/core/config/schema/engines.js +15 -33
  26. package/dist/core/config/schema/improve-processes.js +16 -0
  27. package/dist/core/content-safety.js +0 -24
  28. package/dist/core/redaction.js +4 -0
  29. package/dist/core/spawn-env.js +25 -0
  30. package/dist/core/structured.js +1 -1
  31. package/dist/execution/source.js +8 -12
  32. package/dist/integrations/agent/config.js +1 -3
  33. package/dist/integrations/agent/engine-resolution.js +0 -3
  34. package/dist/integrations/agent/execution.js +14 -13
  35. package/dist/integrations/agent/index.js +1 -1
  36. package/dist/integrations/agent/model-map.js +15 -16
  37. package/dist/integrations/agent/profiles.js +2 -2
  38. package/dist/integrations/agent/prompts.js +51 -127
  39. package/dist/integrations/agent/request-lowering.js +9 -7
  40. package/dist/integrations/agent/runner-dispatch.js +25 -31
  41. package/dist/integrations/harnesses/aider/agent-builder.js +1 -2
  42. package/dist/integrations/harnesses/amazonq/agent-builder.js +1 -2
  43. package/dist/integrations/harnesses/claude/agent-builder.js +4 -16
  44. package/dist/integrations/harnesses/codex/agent-builder.js +1 -2
  45. package/dist/integrations/harnesses/codex/index.js +6 -11
  46. package/dist/integrations/harnesses/codex/session-log.js +211 -0
  47. package/dist/integrations/harnesses/ids.js +10 -16
  48. package/dist/integrations/harnesses/opencode/agent-builder.js +14 -24
  49. package/dist/integrations/harnesses/opencode/model-config.js +15 -62
  50. package/dist/integrations/harnesses/opencode/model-work-agent.js +71 -36
  51. package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -6
  52. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +32 -90
  53. package/dist/integrations/harnesses/openhands/agent-builder.js +1 -2
  54. package/dist/integrations/harnesses/pi/agent-builder.js +1 -2
  55. package/dist/integrations/harnesses/types.js +3 -3
  56. package/dist/llm/feature-gate.js +2 -5
  57. package/dist/llm/index-passes.js +2 -2
  58. package/dist/llm/structured-call.js +5 -5
  59. package/dist/output/shapes/passthrough.js +1 -0
  60. package/dist/scripts/akm-migrate-node.js +381 -239
  61. package/dist/scripts/akm-migrate.js +381 -239
  62. package/dist/workflows/exec/unit-dispatch.js +4 -13
  63. package/docs/reference/cli.md +29 -16
  64. package/docs/reference/configuration.md +79 -85
  65. package/docs/reference/data-and-telemetry.md +2 -3
  66. package/docs/reference/workflow-schema.md +6 -9
  67. package/package.json +1 -1
  68. package/schemas/akm-config.json +108 -36
package/CHANGELOG.md CHANGED
@@ -6,297 +6,260 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.9.25-alpha.3] - 2026-10-04
10
+
11
+ ### Added
12
+
13
+ - **akm learns from Codex sessions.** `akm proposal extract --type codex` reads
14
+ the rollout files Codex writes under `$CODEX_HOME/sessions` (`~/.codex/sessions`
15
+ by default), as it reads Claude Code's and opencode's session files. `--auto`
16
+ and `akm improve`'s session extraction include Codex on a machine that has
17
+ them, and a session is extracted once, as for the other harnesses. The model
18
+ sees what the person and Codex said and the tool calls and results between
19
+ them, without Codex's own instructions or injected context (AGENTS.md, the
20
+ environment, an invoked skill). The reader lists a person's sessions,
21
+ `codex exec` runs included. It leaves out the rollouts Codex writes for
22
+ subagents and its other internal agents, which are not sessions of their own, and
23
+ unlike a Claude Code subagent transcript it does not fold them into their
24
+ parent.
25
+
26
+ ### Changed
27
+
28
+ - **Reflect changes only an asset's `description`, `when_to_use` and title;
29
+ akm keeps the body byte for byte.** On 396 labelled reflect edits, those that
30
+ fixed a frontmatter defect and left the body alone were good 24 times in 26;
31
+ those that also rewrote the body were bad 175 times in 224. The reply is now
32
+ `confidence` and a `frontmatterPatch` of `description`, `when_to_use` and
33
+ `title` (each a non-empty single-line string, or `null` for no change), plus
34
+ `ref` when no target was given, and no body: the JSON Schema, both output
35
+ contracts and the repair prompt say so, and the framed reply for an endpoint
36
+ that rejects JSON Schema has no content markers. akm applies the patch to the
37
+ asset it read by rewriting only the changed keys' frontmatter lines, so every
38
+ other line, including a list beside a description the YAML parser cannot read,
39
+ stays as it was; a source whose closing `---` is fused onto a value gets no
40
+ proposal. A non-null `title` becomes a
41
+ `# <title>` heading and one blank line at the top of the body, only when the
42
+ body has no level-1 heading; otherwise it is ignored. A patch that changes
43
+ nothing, whether every field is `null` or equal to the source's, creates no
44
+ proposal (`no_change`); an asset that requires a `description` and has none
45
+ still gets one derived from its own text, as before (#636).
46
+ - **Reflect may answer "nothing to change."** Its prompt forbade it: "your
47
+ proposal must correct or add something the source lacks", "must meaningfully
48
+ differ" from a rejected proposal, "do not return the same content
49
+ unchanged", and "you MUST generate" a `when_to_use`. On the same edits, those
50
+ that fixed no defect were bad 86 times in 93, and those driven by negative
51
+ feedback 50 times in 53; most of that feedback says the asset did not help
52
+ with an unrelated task. One goal sentence now covers every asset type: check
53
+ the three fields against the body and the feedback, and return `null` for
54
+ each that needs no fix. The feedback caveat says that feedback about a task
55
+ the asset never claims to cover needs no change, and with no feedback the
56
+ prompt says to fix only a missing or broken field. A rejected proposal is not
57
+ to be proposed again, and the engine returns `null` when no other change is
58
+ justified.
59
+ - **Reflect's prompt names the frontmatter problems akm can see** (a
60
+ description split by a stray period or carrying an escaped quote, no
61
+ `when_to_use`, no title) and asks for a repair that keeps the description's
62
+ wording, names, numbers and paths rather than a rewrite. Without the list the
63
+ model fixed 51 of 122 broken descriptions and 93 of 193 missing titles; with
64
+ it, 120 and 189. When the feedback calls a note stale or historical, a new
65
+ `when_to_use` names the version or date the body records, or stays as it is.
66
+ On an 80-case sample of the labelled set, Claude Opus reviewers judged 64 of
67
+ the new reflect's 78 proposals good; the old reflect's edits in the set are
68
+ good 81 times in 396.
69
+ - **`akm proposal accept` warns about an echoed "Avoid These Patterns"
70
+ section instead of refusing.** Reflect keeps a body as it is, so a proposal
71
+ for an asset that already carries the section (a leftover of the run-only
72
+ prompt text) would never be accepted, and the person accepting cannot edit
73
+ the proposal.
74
+
75
+ ### Removed
76
+
77
+ - **Everything reflect needed to rewrite a body.** The prompt's "Content
78
+ preservation rules" and the size bounds computed for them; the related
79
+ distilled lessons section and the companion-doc
80
+ (`knowledge/skills/<skill>/references/<topic>`) option, with the gathering
81
+ behind them and the `derived_from_reflect` marker it read; the size guard and
82
+ the truncation-marker check on reflect's output, and their review reasons
83
+ `reflect-size-ratio` and `reflect-truncation-leak`; the stripping of an
84
+ appended frontmatter block and of an echoed "Avoid These Patterns" section
85
+ from a body; and the restore of identity fields, which a patch cannot name.
86
+ The accept-time size advisory and truncation-marker block stay.
87
+ - **The `body-edit` review reason.** A reflect revision no longer changes the
88
+ body, so one the judge passed is stamped `staged` and the triage drain
89
+ accepts it, as before 0.9.24; with `processes.reflect.qualityGate` off it is
90
+ minted unstamped for the drain to decide. A proposal already deferred as
91
+ `body-edit` stays deferred.
92
+
93
+ ## [0.9.25-alpha.2] - 2026-10-03
94
+
95
+ ### Added
96
+
97
+ - **`akm improve judge`** runs reflect's quality judge on one revision, read as
98
+ `{"source", "candidate", "feedback", "ref"}` JSON from stdin, with the engine
99
+ `processes.reflect.qualityGate.engine` names, and prints the verdict. It
100
+ writes nothing: it tests a judge engine on revisions whose right answer you
101
+ know.
102
+
103
+ ### Changed
104
+
105
+ - **Reflect refuses three kinds of defective revision before the judge runs,**
106
+ with the quality gate on or off: one that adds placeholder text ("please
107
+ confirm", "to be confirmed"; not `TODO`, `TBD` or `FIXME`), one that talks about
108
+ its own edit ("the feedback says", "this revision", "the source asset", a
109
+ quoted gate rejection), and one that copies frontmatter into its body (key
110
+ lines such as `sources:` or `updated:` outside code, or a `sources`,
111
+ `xrefs` or `contradictedBy` value). Each rule counts only what the revision
112
+ adds to its source. A hit is a `quality_rejected` refusal with no proposal
113
+ and no judge call, and the event names the rule (`reflectDefect`). Each
114
+ rule's wording is a list that `processes.reflect.defectFilter` can replace:
115
+ `placeholders` and `metaCommentary` are plain phrases, matched as whole words
116
+ in any case, and `frontmatterKeys` are exact key names. A list left out keeps
117
+ its default, and `[]` turns its rule off. On 396 labelled reflect edits the
118
+ default lists hit 22 of 313 bad edits and none of 83 good ones.
119
+ - **The reflect quality judge's rubric names what it kept missing.** Tuned on
120
+ the labelled judge-gate set (plain judge, qwen3.8-27b): a description with a
121
+ sentence split by a stray period or an unbalanced quote is broken text; a
122
+ missing title, description or `when_to_use` is a concrete problem whether or
123
+ not the feedback mentions it, while a bare `type:` or a provenance stamp is
124
+ not; PRESERVATION and QUALITY are judged line by line in the changed region,
125
+ so a fix elsewhere no longer excuses a dropped fact, an invented or
126
+ strengthened claim, a hedge or a placeholder. On a stratified 100-case sample
127
+ the gate passed 31 of 40 good edits instead of 18 and 3 of 60 bad instead of 2.
128
+ - **A reflect quality judge on an agent engine reads only to verify what a
129
+ revision adds.** Its prompt names the revised asset's ref and adds one
130
+ paragraph: read that asset, or one the changed region names, with `akm_show`
131
+ only to check a fact the revision adds or alters, at most twice, and never
132
+ search; before scoring, find each added statement in the asset or the
133
+ feedback (a step or cause that merely seems to follow does not count), each
134
+ source fact in the revision, and each feedback point in a change to the text
135
+ it is about. The plain judge's prompt is unchanged. On the same 100 cases
136
+ (qwen3.8-27b, thinking on) the agent judge passed 38 of 40 good edits and 14
137
+ of 60 bad, against 34 and 12 for the plain judge on that model. Give the
138
+ engine's `llmEngine` `enableThinking: true`: without thinking the model
139
+ looped on tool calls.
140
+ - **A failed reflect reply reads the same on every engine kind:** the parser's
141
+ own message. 0.9.25-alpha.1 named the engine on an agent engine.
142
+ - **Inference reaches opencode only where akm writes opencode's config:** an
143
+ improve process's `llm` overlay on model work's agent, and an `opencode-sdk`
144
+ engine's `llmEngine` fallback model. Set the rest in your opencode config; a
145
+ task's or workflow's `inference` is an `untranslated-field` notice, as before
146
+ 0.9.25-alpha.1. `claude` takes no `--effort`.
147
+ - **An `opencode-sdk` session the dispatch times out on or aborts is aborted on
148
+ the server for every dispatch,** not only model work.
149
+ - **An improve stage retries a reply only when the stage cannot read it,** not
150
+ when it misses the JSON Schema. A reply it can read costs no second call.
151
+ - **opencode model work can read the stash and search it.** The `akm-model-work`
152
+ agent reads, greps and globs in the stash and its working directory, edits
153
+ only in the working directory, and has `akm_search` and `akm_show` from the
154
+ akm-opencode plugin (0.9.21 or later, in your own opencode config). The
155
+ plugin's curation, learning and write gate are off for these dispatches, and
156
+ its state goes to akm's state directory, not `~/.local/state/akm-opencode`.
157
+
158
+ ### Removed
159
+
160
+ - **The agent-engine inference fields,** which 0.9.25-alpha.1 accepted: they
161
+ fail to load again.
162
+ - **The `opencode-sdk` step watcher and the git-repository refusal for model
163
+ work's scratch directory.**
164
+ - **opencode model work's step limit.** At the limit opencode sends a "maximum
165
+ steps" message as a trailing assistant message, which a qwen chat template
166
+ (LM Studio, llama-server) renders as the start of the model's reply: LM
167
+ Studio returned nothing, llama-server returned that message as the answer,
168
+ and the dispatch failed. The dispatch timeout bounds a run.
169
+ - **A prompt builder that nothing called** (`buildSchemaRepairPrompt`).
170
+ - **The `--track-usage` / `--no-track-usage` flag on `akm search`, `akm curate`
171
+ and `akm show`.** A successful read always records its usage event, stamped
172
+ with its source (`user`, `improve`, `task` or `audit`); only `user` events
173
+ feed ranking and eval, so machine reads never skew them. Either spelling now
174
+ fails as an unknown flag.
175
+
176
+ ### Fixed
177
+
178
+ - **The reflect size guard no longer flags a body that does not grow.** The
179
+ expansion ceiling is capped at 25,000 characters, so a source body longer than
180
+ that was flagged `EXCESSIVE_EXPANSION` even when the proposed body was its own
181
+ length (ratio 1.00), and went to review instead of the judge. A body no longer
182
+ than its source is never expansion; one that grows past the cap still is, by
183
+ any amount. The reflect prompt agrees: it told the model its body could be at
184
+ most 25,000 characters even when the source was longer, and now gives such a
185
+ source's own length.
186
+ - **An asset's or a task's own `tools:` can no longer name the model-work
187
+ policy.** In 0.9.25-alpha.1 exactly `read`, `edit`, `akm search`, `akm show`,
188
+ in that order, skipped `execution.allowedTools`. It is ordinary tools now:
189
+ only akm's own model-work callers ask for the policy.
190
+ - **An agent engine that names its model only in `args` is named in the usage
191
+ report,** where it showed `unattributed`.
192
+ - **`opencode` and `opencode-sdk` engines receive the XDG base-directory
193
+ variables.** Under a custom `XDG_CONFIG_HOME` the spawned opencode missed its
194
+ provider config and failed every dispatch with `Unexpected server error`.
195
+
9
196
  ## [0.9.25-alpha.1] - 2026-10-02
10
197
 
11
198
  ### Changed
12
199
 
13
- - **One schema instruction, appended by the shared agent request lowering.**
14
- Seven harness builders and the workflow engine each kept a copy of
15
- "Respond with ONLY a JSON value matching this JSON Schema (no prose, no code
16
- fences)". The lowering now appends it to every agent engine's prompt when a
17
- schema is requested, so `codex` gets it beside `--output-schema` outside
18
- workflows too. A workflow unit sends the same prompt bytes as before, less
19
- the duplicate fixed below; a direct-LLM unit still carries the instruction
20
- in its own prompt.
21
- - **Model work is bounded at 600 seconds on every engine kind.** An improve
22
- process, a quality or triage judge, or an index pass whose engine sets no
23
- `timeoutMs` now stops after 600 seconds. Before, an agent or `opencode-sdk`
24
- engine ran until it finished, and so did memory inference and consolidation
25
- on an LLM engine. An engine's own `timeoutMs` still applies.
200
+ - **Unattended model work runs under one tool policy, on any engine that
201
+ confines it.** The model may read, edit inside a scratch directory akm removes
202
+ after the dispatch, and run `akm search` and `akm show`; the stash stays
203
+ read-only. An LLM engine has no tools, `claude` confines the policy, and
204
+ `opencode` and `opencode-sdk` run an injected `akm-model-work` agent with
205
+ read and edit only. Every other harness refuses it before the request starts.
206
+ - **One config rule names the engines model work may use.** Every key it reads
207
+ its engine from (`defaults.llmEngine`, `index.*.engine`, a strategy's or
208
+ process's `engine`, an enabled triage `judgment.engine`, a
209
+ `qualityGate.engine`) must name an LLM engine or a `claude`, `opencode` or
210
+ `opencode-sdk` agent engine, or the config fails to load. `--require-engines`
211
+ and `akm health` check agent engines too.
212
+ - **Model work is bounded at 600 seconds on every engine kind** whose engine
213
+ sets no `timeoutMs`; agent and `opencode-sdk` engines ran until they finished.
214
+ - **`akm proposal new` returns the proposal as JSON on every engine kind,**
215
+ with no draft file and no live session. A reply that is not a proposal gets
216
+ one retry.
217
+ - **Reflect asks every engine kind for the same JSON reply and repairs it
218
+ once.** An agent or `opencode-sdk` engine gets reflect's JSON Schema and a
219
+ repair turn, as an LLM engine did. An LLM engine's requests are unchanged.
26
220
  - **An improve stage's reply that fails its JSON Schema gets one corrective
27
- retry.** The stage retries once with the validation errors, then reads the
28
- last reply with its own parser as before. `akm extract` now allows its
29
- corrective retry on every engine, not only on one without JSON Schema
30
- support. Reflect keeps its own repair turn.
31
- - **Agent and `opencode-sdk` dispatches leave a usage record.** Each one now
32
- writes one `llm_usage` record through the same sink and stage attribution as
33
- the LLM path, with the request's model and the tokens the runner reports. An
34
- LLM engine still records each HTTP attempt.
35
- - **`akm proposal new` gets the proposal as JSON from every engine kind, with
36
- no live session.** It told every engine to write the asset to a draft file
37
- and print no JSON, then parsed stdout as JSON. An LLM engine that followed
38
- the instruction could not succeed, and an agent engine succeeded only by
39
- writing the file: an agent CLI ran in a live terminal session whose output
40
- akm could not read, and otherwise failed with an "interactive mode" error.
41
- Every engine now returns the proposal as one JSON object on stdout, and the
42
- request carries its JSON Schema: as `response_format` for an LLM engine, as
43
- the schema instruction for an agent engine. akm captures the reply, unwraps
44
- a harness envelope such as claude's `--output-format json` result, and
45
- validates it. A reply that is not a proposal gets one corrective retry, then
46
- fails with an error that names the engine. An agent CLI now runs headless:
47
- you see the queued proposal, not the agent at work, and no draft file is
48
- written.
49
- - **Unattended model work has one tool policy, which each engine confines or
50
- refuses when the request is built.** The policy allows reading, editing only
51
- inside a scratch working directory that akm creates for the dispatch and
52
- removes after it, and running `akm search` and `akm show`. The stash stays
53
- read-only. An LLM engine has no tools. `claude` runs it with `--restricted`
54
- (its user, project and local settings, whose allow rules could pre-approve
55
- any command or path, are ignored, and its file tools stay in the working
56
- directory), `--strict-mcp-config`, `--tools Read,Edit,Bash`, `--allowedTools`
57
- for Read, Edit and the two `akm` commands, and `--permission-mode dontAsk`.
58
- `opencode` and `opencode-sdk` run an injected `akm-model-work` agent that
59
- can read and edit only inside the working directory and has no bash:
60
- opencode checks a bash rule against the command's words only, so
61
- `akm show x > ~/stash/asset.md` would pass an `akm show *` rule. The agent
62
- has its own short prompt in place of opencode's coding-assistant prompt, an
63
- 8-step limit, and no automatic compaction. opencode only asks the model to
64
- stop at that limit, so `opencode-sdk` aborts the session two steps past it,
65
- and aborts a session the dispatch times out on; on `opencode` the dispatch
66
- timeout is the bound. Every other harness refuses the policy. Such a
67
- dispatch ignores the engine's `args` and `workspace`, refuses to start when
68
- the temporary directory is inside a git repository, which opencode would
69
- treat as its working directory, and fails with `parse_error` when the agent
70
- ends with no answer, as opencode can at its step limit.
71
- - **Unattended model work runs on any engine that confines the model-work
72
- tool policy, and config checks that with one rule.**
73
- - **Who sends the policy.** The improve processes, the quality, triage and
74
- retrieval-gate judges, index passes and `akm remember --enrich` now send
75
- it, so they may run on an LLM engine or on a `claude`, `opencode` or
76
- `opencode-sdk` agent engine, where they could only use an LLM engine
77
- before.
78
- - **The one rule.** Every key model work reads its engine from must name
79
- such an engine. The keys are `defaults.llmEngine`, `index.defaults.engine`,
80
- `index.<pass>.engine`, a strategy's `engine`, a process's `engine`, an
81
- enabled triage `judgment.engine`, and a `qualityGate.engine`. A config
82
- that breaks the rule fails to load, naming the key, the engine and its
83
- platform. The rule replaces six checks that each required an LLM engine
84
- for some of those keys, or let the triage judgment use any agent.
85
- - **What else is gone.** The plan, reflect, the quality gate and index
86
- passes no longer turn an agent engine away. A triage judgment on an agent
87
- engine takes a strategy's `llm` overrides as `untranslated-field`
88
- notices, where it was refused before.
89
- - **Reflect on an agent engine.** The agent returns its proposal as JSON on
90
- stdout instead of writing a draft file, which the policy's scratch working
91
- directory would not keep.
92
- - **The answer is unwrapped.** A model-work reply from an agent engine goes
93
- through its harness's result extractor, so a stage call on `claude` gets
94
- the answer, not the `--output-format json` envelope around it.
95
- - **`--require-engines` checks agent engines too.** It checks an agent
96
- engine by its binary on PATH, and an `opencode-sdk` engine by its binary
97
- and its LLM fallback's endpoint. It probed only LLM connections before.
98
- The usage report and `akm health` count a process's calls whatever its
99
- engine's kind.
100
- - **An `opencode-sdk` engine gets its LLM fallback connection only from its own
101
- `llmEngine`. `defaults.llmEngine` no longer supplies one.** An SDK engine
102
- that set no `llmEngine` used to borrow `defaults.llmEngine`: its connection,
103
- its model (unless the engine set its own) and its timeout. `defaults.llmEngine`
104
- names the default engine for unattended model work, and since that work may
105
- now run on an agent engine it can be one too, so it is no longer also a
106
- connection that every SDK engine shares. **If you relied on the inheritance,
107
- set `llmEngine` on the SDK engine**, for example
108
- `"sdk": { "kind": "agent", "platform": "opencode-sdk", "llmEngine": "fast" }`.
109
- Without one, the SDK engine runs on opencode's own provider and auth, and on
110
- its own `model` if it sets one: akm sends it no connection. An SDK engine
111
- that sets `llmEngine` is unchanged, and so is a config with no `opencode-sdk`
112
- engine. The rule holds everywhere the fallback is read: dispatch, a
113
- workflow's frozen concurrency cap, `akm health`, and
114
- `akm improve --require-engines`, which now check an SDK engine's fallback
115
- endpoint only when it sets `llmEngine`.
116
- - **Inference reaches `opencode`, `opencode-sdk` and `claude` engines.** A
117
- request's `temperature`, `maxTokens`, `contextLength`, `enableThinking` and
118
- `reasoningEffort` came from the engine's own settings, an improve process's
119
- `llm` overlay, a task, command or agent asset's `inference`, a workflow's
120
- `llm:` and a `models.json` alias, and every agent engine dropped all of them
121
- with an `untranslated-field` notice. Each platform now translates what it can
122
- carry, and the nearest layer wins, field by field, as on an LLM engine. With
123
- no setting anywhere akm sends nothing of its own, so the model's configured
124
- default still applies, such as a `reasoningEffort: "none"` in your opencode
125
- config.
126
- - **`claude`** gets `reasoningEffort` as `--effort <level>`, passed as given
127
- (Claude Code 2.1.283 takes `low`, `medium`, `high`, `xhigh` and `max`).
128
- Its other four fields are still reported as untranslated.
129
- - **`opencode` and `opencode-sdk`** get inference as opencode config, which
130
- merges over your own opencode config for the same provider and model:
131
- `temperature` and `reasoningEffort` become `options.temperature` and
132
- `options.reasoningEffort` (opencode drops the snake_case spelling),
133
- `enableThinking` becomes both wire forms an LLM engine sends, and
134
- `maxTokens` with `contextLength` becomes `limit.output` and
135
- `limit.context`. Set both limit fields: opencode refuses half a limit, and
136
- a half would overwrite the other half of one you declared, so a lone one
137
- is reported as untranslated. Without `limit.output` opencode asks for
138
- `max_tokens: 32000`, which a small-context server rejects, and a wrong
139
- `limit.context` lets it build requests past the server's window. The fields
140
- need a `provider/model`: the request's model, or the `--model` an
141
- `opencode` engine's `args` name. Model work needs none for its options, so
142
- an `opencode-sdk` engine with no `model` and no `llmEngine` gets them too.
143
- - **Where it goes.** Model work puts the options on the `akm-model-work`
144
- agent that runs it, so opencode's own title call on the same model keeps
145
- the model's defaults. Any other dispatch runs your own agent, whose name
146
- akm cannot rely on, so its options go on the model and the title call sees
147
- them too. `opencode-sdk` makes no title call. An `opencode-sdk` engine
148
- declares its `llmEngine` fallback's model with the fallback's own
149
- inference, under the engine's and the request's; each distinct set starts
150
- its own `opencode serve`, as a different model does.
151
- - **An agent engine may set the inference fields its platform translates.**
152
- `engines.<name>` of `kind: "agent"` took none, so an engine could not carry a
153
- `temperature` or a `reasoningEffort` of its own. `opencode` and
154
- `opencode-sdk` may set `temperature`, `maxTokens`, `contextLength`,
155
- `enableThinking` and `reasoningEffort`, `claude` may set `reasoningEffort`,
156
- and every other platform none. A field its platform does not translate fails
157
- to load, naming the platform and the fields it does translate. `provider`,
158
- `endpoint`, `apiKey`, `apiKeyFile`, `concurrency` and `extraParams` stay
159
- invalid on an agent engine.
160
- - **Reasoning effort has one word in a request, `reasoningEffort`.** The
161
- starter `reasoning` alias, an alias in your `models.json` and an asset's
162
- `effort:` frontmatter said `effort`, and engines, opencode and the LLM
163
- request said `reasoningEffort`. `effort` is now read as `reasoningEffort`
164
- where layers are merged, so the nearest layer wins whichever word it used.
165
- On an LLM engine an alias's or asset's `effort` is therefore sent as
166
- `reasoning_effort`; it was reported as untranslated and dropped before. An
167
- LLM engine's own request is unchanged.
168
- - **Reflect asks every engine kind for the same JSON reply, checks it the same
169
- way and repairs it once.** An LLM engine was sent the reply's JSON Schema,
170
- and its reply was held to exact fields and repaired once. An agent or
171
- `opencode-sdk` engine got a looser contract in its prompt (`ref`, `content`
172
- and an optional `frontmatter`) with no schema and no repair, so one invalid
173
- reply failed the run. Every engine kind now runs the same iteration:
174
- - **The request.** The prompt carries the same output contract, and the
175
- reply's JSON Schema is the request's output schema: `response_format` for
176
- an LLM engine, as before, and the schema instruction at the end of the
177
- prompt for an agent engine. `claude` also gets `--output-format json`,
178
- which akm unwraps. An LLM endpoint that rejects JSON Schema still gets the
179
- framed-markdown contract.
180
- - **The reply.** An agent now returns `content`, `confidence` and a
181
- `frontmatterPatch` of `description` and `when_to_use`, and `ref` too when
182
- no asset was named, as an LLM does. akm derives a named asset's ref and
183
- merges the patch with the source's frontmatter, so an agent can no longer
184
- set other frontmatter keys or retarget the proposal.
185
- - **The repair.** A reply that fails the contract gets one repair turn that
186
- carries the first reply, shared across self-refine passes. A reply that is
187
- still invalid fails with `parse_error` and queues nothing. On an agent or
188
- `opencode-sdk` engine the error names the engine, as in
189
- `Engine "oc" reply was not a valid reflect proposal after 2 attempts: …`.
190
- An LLM engine's message is unchanged, because improve feeds it into later
191
- prompts as a pattern to avoid. A failed agent dispatch is still reported
192
- with its exit code and stderr.
193
- - **What reflect sends and reports on an agent engine.** Reflect asks every
194
- engine kind for no visible chain of thought (`enableThinking: false`).
195
- `opencode` and `opencode-sdk` carry it, as the inference entry above
196
- says; `claude` reports it as an `untranslated-field` notice, as it does
197
- for every other stage. `reflect_completed` carries `outputMode` and
198
- `repairAttempts` for every engine kind.
199
- - **Unchanged.** An LLM engine's requests are byte-identical to before:
200
- reflect's generation and repair, and its quality judge. So are the
201
- refine passes, the content budget (an LLM engine's context length), the
202
- protected frontmatter fields, the review routing and the judge selection.
221
+ retry,** then the stage reads the last reply with its own parser.
222
+ - **One schema instruction for every agent engine,** appended by the shared
223
+ request lowering: `opencode` and `opencode-sdk` now receive a requested
224
+ schema, and a workflow unit on seven harnesses no longer carries it twice.
225
+ - **Agent and `opencode-sdk` dispatches leave a usage record.**
226
+ - **An `opencode-sdk` engine's LLM fallback comes only from its own
227
+ `llmEngine`; `defaults.llmEngine` no longer supplies one.** If you relied on
228
+ that, set `llmEngine` on the SDK engine. Without one, opencode uses its own
229
+ provider, model and auth.
230
+ - **Inference reaches `opencode`, `opencode-sdk` and `claude` engines.**
231
+ `temperature`, `maxTokens`, `contextLength`, `enableThinking` and
232
+ `reasoningEffort` from an engine, an improve process's `llm` overlay, an
233
+ asset, a workflow or a `models.json` alias were dropped on every agent
234
+ engine. An agent engine may set them, and `claude` takes `reasoningEffort` as
235
+ `--effort`.
236
+ - **Reasoning effort has one word, `reasoningEffort`.** `effort` in a
237
+ `models.json` alias or an asset is read as it, so an LLM engine now sends it
238
+ as `reasoning_effort`.
203
239
 
204
240
  ### Removed
205
241
 
206
- - **Two harness metadata fields that nothing read.** Each of the ten harness
207
- descriptors carried an execution `pattern` and a `structuredOutput` tier.
208
- No code branched on either: the shared request lowering appends the schema
209
- instruction to every agent prompt, and a harness's own argv builder adds its
210
- native channel, as codex does with `--output-schema`. The fields, their two
211
- types and the tests that pinned their values are gone. Nothing you run
212
- changes.
213
- - **`resolveLlmEngineUse`'s swap of an agent engine for an LLM engine.** Given
214
- an agent engine, it used the engine's `llmEngine`, then `defaults.llmEngine`,
215
- and warned. Only the implicit SDK fallback above could reach it, so nothing
216
- you run changes. An agent engine passed to it is now an error.
217
- - **An `effort` hint on the agent dispatch request that nothing read.** The
218
- lowering set it from `inference.effort`, reserved for a workflow field, and
219
- no builder consumed it. `reasoningEffort` in the request's inference is read
220
- where it is translated, and the field is gone.
221
- - **The agent file-write contract.** An agent was once told to write its
222
- proposal to a draft file and print `DRAFT_WRITTEN confidence=<n>`. Reflect and
223
- `akm proposal new` had already stopped sending that instruction, because the
224
- model-work scratch directory does not outlive a dispatch. The instruction, the
225
- function that read the `DRAFT_WRITTEN` line and the `draftFilePath` prompt
226
- inputs that nothing passed any more are gone, with their tests. So is
227
- reflect's `ref_mismatch` check: a reflect that names an asset derives the
228
- proposal's ref, so an engine can no longer name another one. Nothing you run
229
- changes.
242
+ - **Dead code; nothing you run changes:** each harness's unused `pattern` and
243
+ `structuredOutput` fields, `resolveLlmEngineUse`'s swap of an agent engine
244
+ for an LLM engine, the agent request's unread `effort` hint, and the agent
245
+ file-write contract (`DRAFT_WRITTEN`, reflect's `ref_mismatch` check).
230
246
 
231
247
  ### Fixed
232
248
 
233
- - **An `opencode-sdk` engine with an LLM fallback now reaches its endpoint,
234
- and a failed dispatch is reported as a failure (#1015).** The `akm-custom`
235
- provider that akm generates for the fallback listed no models, so opencode
236
- could not find the model and failed every dispatch. akm now declares the
237
- routed model under the provider's `models`. The runner also ignored the
238
- error that the SDK client returns for an HTTP error, and the error opencode
239
- puts on a reply when the provider rejects a request, and reported
240
- `ok: true` with empty output. Both now give `ok: false`, with opencode's
241
- error name and message in `error` and `stderr`. An aborted message is
242
- `aborted`, a reply cut off at the output limit is `parse_error`, and any
243
- other error is `non_zero_exit`. A reply with several text parts now returns
244
- the last one, which is the answer, instead of the first.
245
- - **An LLM engine reports a provider error sent with HTTP 200 as a failure.**
246
- OpenRouter, for one, answers a request whose provider fails after the
247
- response has started with HTTP 200 and a body that holds only an `error`
248
- object and no `choices`. akm returned that as an empty reply with
249
- `ok: true`. It is now a provider error with the body in its message, as an
250
- error status is.
251
- - **An `opencode` engine runs a persona instead of failing.** akm passed the
252
- persona to `opencode run` as `--system-prompt`, which opencode 1.18 does
253
- not accept, so opencode printed its usage and exited 1 on every dispatch
254
- that carried a persona, including `akm agent <agent-ref> --engine opencode`.
255
- akm now composes the persona into the prompt in an `<AKM_PERSONA>` block,
256
- as it does for harnesses with no system-prompt option.
257
- - **`opencode` and `opencode-sdk` engines receive a requested output schema.**
258
- They dropped it with only an `untranslated-field` warning, so a schema from
259
- `akm agent`, `akm command run`, a command's frontmatter or a task's
260
- `output:` never reached the model. They now get it as the same instruction
261
- every other agent engine gets.
262
- - **An LLM engine sends a requested schema unless it opts out.** It sent
263
- `response_format` only when the engine set `supportsJsonSchema: true`, so an
264
- engine that left the flag unset never had its output constrained. It now
265
- sends it unless the engine sets `supportsJsonSchema: false`. An endpoint
266
- that rejects it with a 4xx is still retried once without it.
267
- - **A workflow unit on `claude`, `copilot`, `gemini`, `pi`, `aider`,
268
- `amazonq` or `openhands` gets the schema instruction once.** The unit prompt
269
- carried it and the harness builder appended a second copy.
270
- - **A feature gate's timeout now stops the call it bounds.** When an improve
271
- stage's gate timed out (600 seconds unless the call sets its own), the model
272
- call kept running in the background, on an engine with `timeoutMs: 900000`
273
- for up to five more minutes. The gate now aborts it.
274
- - **A stage call reports a timeout or abort by the dispatch's own reason.** A
275
- timed-out or aborted agent or `opencode-sdk` dispatch, and an LLM timeout
276
- inside a feature gate, came back as `error`. They now come back as `timeout`
277
- or `aborted`.
278
-
279
- - **`akm proposal new` keeps the reply's confidence.** The engine's
280
- self-rated `confidence` was parsed and then dropped, so a proposal from
281
- `proposal new` never carried the field the reference says it has.
282
- - **Model work on an agent or `opencode-sdk` engine that sets no `timeoutMs` now
283
- stops after 600 seconds, as documented.** Such an engine resolved to an
284
- explicit "no timeout", so the 600-second bound for model work never applied
285
- to it. An engine's own `timeoutMs`, `null` included, still applies, and
286
- other work on an agent engine still runs until it finishes.
287
- - **The announcement of the implicit `opencode-sdk` fallback is now true.** It
288
- says provider, model and auth come from opencode's own configuration, but the
289
- fallback engine borrowed `defaults.llmEngine`'s connection whenever one was
290
- set, so opencode got an akm-generated provider and model instead. It now runs
291
- on opencode's own configuration, as announced.
292
- - **`model: reasoning` set no effort on `claude`, `opencode` or `opencode-sdk`.**
293
- The starter alias supplies `effort: high` for all three, and each dropped it
294
- with an `untranslated-field` notice, so the alias chose a stronger model and
295
- nothing more. It now sets `--effort high` on `claude` and
296
- `options.reasoningEffort` on opencode (see "Inference reaches `opencode`,
297
- `opencode-sdk` and `claude` engines" above). An improve process's
298
- `llm.reasoningEffort` and `llm.temperature` overlay now reaches those
299
- engines the same way.
249
+ - **An `opencode-sdk` engine with an LLM fallback reaches its endpoint, and a
250
+ failed dispatch is a failure (#1015).** An SDK error or provider rejection is
251
+ `ok: false` with opencode's message, and a reply's last text part is its answer.
252
+ - **An LLM engine reports a provider error sent with HTTP 200 as a failure**
253
+ (OpenRouter does this), not as an empty reply.
254
+ - **An `opencode` engine runs a persona** instead of failing on
255
+ `--system-prompt`, which opencode 1.18 rejects.
256
+ - **An LLM engine sends a requested schema** unless it sets
257
+ `supportsJsonSchema: false`; it sent one only when it set `true`.
258
+ - **A feature gate's timeout stops the call it bounds,** and a stage call
259
+ reports a timeout or abort as `timeout` or `aborted`, not `error`.
260
+ - **`akm proposal new` keeps the reply's `confidence`.**
261
+ - **Model work on an agent or `opencode-sdk` engine with no `timeoutMs` stops
262
+ after 600 seconds** as documented; it resolved to no timeout.
300
263
 
301
264
  ## [0.9.24] - 2026-10-02
302
265
 
@@ -1 +1 @@
1
- Feedback describes what a reader found missing or wrong. It is a signal to investigate, not a fact to insert. Do not add claims, numbers, dates, paths, ports, or incidents that are not already present in the asset content. If feedback asks for information the asset lacks, leave the section unchanged.
1
+ Feedback describes what a reader found missing or wrong, often that the asset did not help with a task it was retrieved for. It is a signal, not a fact to insert. Change a field only when it is missing or broken, or when it claims more than the body covers, and then only to describe what the body covers. Feedback about a task the asset never claims to cover, or asking for information the asset lacks, needs no change. When the feedback says the asset is stale, outdated, superseded or historical, a `when_to_use` names the version or date the body records ("When working with the 0.1.0 client"), or stays as it is.
@@ -1,13 +1,6 @@
1
1
  Respond with exactly this plain-text frame, with no prose or code fence around it:
2
2
 
3
3
  {{REF_LINE}}AKM_REFLECT_CONFIDENCE: <number from 0 to 1>
4
- AKM_REFLECT_FRONTMATTER_PATCH: {"description": null, "when_to_use": null}
5
- AKM_REFLECT_CONTENT_BEGIN
6
- <complete improved markdown body>
7
- AKM_REFLECT_CONTENT_END
4
+ AKM_REFLECT_FRONTMATTER_PATCH: {"description": null, "when_to_use": null, "title": null}
8
5
 
9
- The first begin marker and final end marker delimit the body; marker lines between them are literal content. Put the complete markdown body between those outer markers. Quotes, Markdown fences, and backslashes inside the body are literal content; do not JSON-escape them. Emit the body only, without YAML frontmatter, because AKM preserves and merges the source frontmatter itself.
10
-
11
- The frontmatter patch must be a one-line JSON object with exactly `description` and `when_to_use`. Keep a field `null` when it should not change. Supply a non-empty string only when adding or correcting that field; AKM merges those values through its existing sanitizer.
12
-
13
- Never include the truncation marker (the literal text `{{TRUNCATION_MARKER}}`) or any other text from outside the fenced asset content shown to you, anywhere in the body.
6
+ The frontmatter patch must be a one-line JSON object with exactly `description`, `when_to_use` and `title`. Keep a field `null` when it should not change; otherwise give a non-empty single-line string. `title` is the text of a level-1 heading, without the leading `#`; AKM adds it only when the body has none. AKM applies the patch to the source asset and keeps the body itself.
@@ -1,5 +1,3 @@
1
1
  Respond only through the provider's native JSON schema. {{FIELD_RULE}}
2
2
 
3
- `content` must contain the complete improved markdown body only, without YAML frontmatter. `frontmatterPatch` must contain exactly `description` and `when_to_use`; set either field to `null` when it should not change, or to a non-empty string when adding or correcting it. AKM merges that narrow patch with the source frontmatter and preserves target identity itself. `confidence` is your honest self-rated quality confidence from 0 to 1. Do not add prose or Markdown fences around the JSON response.
4
-
5
- Never include the truncation marker (the literal text `{{TRUNCATION_MARKER}}`) or any other text from outside the quoted asset content shown to you, anywhere in `content`.
3
+ `frontmatterPatch` must contain exactly `description`, `when_to_use` and `title`; set a field to `null` when it should not change, or to a non-empty single-line string. `title` is the text of a level-1 heading, without the leading `#`; AKM adds it only when the body has none. AKM applies the patch to the source asset, keeps the body itself, and preserves target identity. `confidence` is your honest self-rated quality confidence from 0 to 1. Do not add prose or Markdown fences around the JSON response.
@@ -1,3 +1,3 @@
1
- Your previous response could not be extracted using the required output contract. Reformat that response exactly once using the contract below. Preserve its proposed markdown verbatim: do not revise, summarize, or add content. Return only the repaired envelope.
1
+ Your previous response could not be extracted using the required output contract. Reformat that response exactly once using the contract below. Keep its proposed values as they are: do not revise or add to them. Return only the repaired envelope.
2
2
 
3
3
  {{OUTPUT_CONTRACT}}
package/dist/cli.js CHANGED
@@ -177,7 +177,7 @@ const setupCommand = defineCommand({
177
177
  // the work, matching the `sync --push/--no-push` pattern. A flag
178
178
  // DECLARED as `no-init` can never be negated: `--no-init` parses as
179
179
  // "negate `init`", a name nothing declared, leaving the real key at its
180
- // default forever — see `search --no-track-usage`'s identical fix.
180
+ // default forever.
181
181
  init: {
182
182
  type: "boolean",
183
183
  default: true,
@@ -506,6 +506,7 @@ async function judgeOne(ctx, candidate) {
506
506
  signal: ctx.opts.signal,
507
507
  ...(ctx.chat ? { chat: ctx.chat } : {}),
508
508
  },
509
+ parse: parsePairJudgeResponse,
509
510
  ...(ctx.opts.onNotices ? { onNotices: ctx.opts.onNotices } : {}),
510
511
  });
511
512
  if (!outcome.ok)