akm-cli 0.9.25-alpha.1 → 0.9.25-alpha.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +159 -280
- package/dist/cli.js +1 -1
- package/dist/commands/improve/consolidate/pair-pass.js +1 -0
- package/dist/commands/improve/consolidate.js +7 -2
- package/dist/commands/improve/execution.js +2 -3
- package/dist/commands/improve/extract.js +1 -0
- package/dist/commands/improve/improve-cli.js +33 -1
- package/dist/commands/improve/loop-stages.js +3 -0
- package/dist/commands/improve/reflect-noise.js +125 -0
- package/dist/commands/improve/reflect.js +13 -16
- package/dist/commands/improve/retrieval-gate.js +7 -2
- package/dist/commands/improve/stage.js +31 -39
- package/dist/commands/proposal/drain.js +4 -7
- package/dist/commands/proposal/propose.js +2 -11
- package/dist/commands/proposal/validators/proposal-quality-validators.js +4 -2
- package/dist/commands/read/search-cli.js +0 -38
- package/dist/core/config/schema/engines.js +15 -33
- package/dist/core/config/schema/improve-processes.js +16 -0
- package/dist/core/redaction.js +4 -0
- package/dist/core/spawn-env.js +25 -0
- package/dist/core/structured.js +1 -1
- package/dist/execution/source.js +8 -12
- package/dist/integrations/agent/config.js +1 -3
- package/dist/integrations/agent/engine-resolution.js +0 -3
- package/dist/integrations/agent/execution.js +14 -13
- package/dist/integrations/agent/index.js +1 -1
- package/dist/integrations/agent/model-map.js +15 -16
- package/dist/integrations/agent/profiles.js +2 -2
- package/dist/integrations/agent/prompts.js +3 -39
- package/dist/integrations/agent/request-lowering.js +9 -7
- package/dist/integrations/agent/runner-dispatch.js +25 -31
- package/dist/integrations/harnesses/aider/agent-builder.js +1 -2
- package/dist/integrations/harnesses/amazonq/agent-builder.js +1 -2
- package/dist/integrations/harnesses/claude/agent-builder.js +4 -16
- package/dist/integrations/harnesses/codex/agent-builder.js +1 -2
- package/dist/integrations/harnesses/ids.js +10 -16
- package/dist/integrations/harnesses/opencode/agent-builder.js +14 -24
- package/dist/integrations/harnesses/opencode/model-config.js +15 -62
- package/dist/integrations/harnesses/opencode/model-work-agent.js +71 -36
- package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -6
- package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +32 -90
- package/dist/integrations/harnesses/openhands/agent-builder.js +1 -2
- package/dist/integrations/harnesses/pi/agent-builder.js +1 -2
- package/dist/llm/feature-gate.js +2 -5
- package/dist/llm/index-passes.js +2 -2
- package/dist/llm/structured-call.js +5 -5
- package/dist/output/shapes/passthrough.js +1 -0
- package/dist/scripts/akm-migrate-node.js +170 -194
- package/dist/scripts/akm-migrate.js +170 -194
- package/dist/workflows/exec/unit-dispatch.js +4 -13
- package/docs/reference/cli.md +12 -7
- package/docs/reference/configuration.md +78 -82
- package/docs/reference/data-and-telemetry.md +2 -3
- package/docs/reference/workflow-schema.md +6 -9
- package/package.json +1 -1
- package/schemas/akm-config.json +108 -36
package/CHANGELOG.md
CHANGED
|
@@ -6,297 +6,176 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.9.25-alpha.2] - 2026-10-03
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- **`akm improve judge`** runs reflect's quality judge on one revision, read as
|
|
14
|
+
`{"source", "candidate", "feedback", "ref"}` JSON from stdin, with the engine
|
|
15
|
+
`processes.reflect.qualityGate.engine` names, and prints the verdict. It
|
|
16
|
+
writes nothing: it tests a judge engine on revisions whose right answer you
|
|
17
|
+
know.
|
|
18
|
+
|
|
19
|
+
### Changed
|
|
20
|
+
|
|
21
|
+
- **Reflect refuses three kinds of defective revision before the judge runs,**
|
|
22
|
+
with the quality gate on or off: one that adds placeholder text ("please
|
|
23
|
+
confirm", "to be confirmed"; not `TODO`, `TBD` or `FIXME`), one that talks about
|
|
24
|
+
its own edit ("the feedback says", "this revision", "the source asset", a
|
|
25
|
+
quoted gate rejection), and one that copies frontmatter into its body (key
|
|
26
|
+
lines such as `sources:` or `updated:` outside code, or a `sources`,
|
|
27
|
+
`xrefs` or `contradictedBy` value). Each rule counts only what the revision
|
|
28
|
+
adds to its source. A hit is a `quality_rejected` refusal with no proposal
|
|
29
|
+
and no judge call, and the event names the rule (`reflectDefect`). Each
|
|
30
|
+
rule's wording is a list that `processes.reflect.defectFilter` can replace:
|
|
31
|
+
`placeholders` and `metaCommentary` are plain phrases, matched as whole words
|
|
32
|
+
in any case, and `frontmatterKeys` are exact key names. A list left out keeps
|
|
33
|
+
its default, and `[]` turns its rule off. On 396 labelled reflect edits the
|
|
34
|
+
default lists hit 22 of 313 bad edits and none of 83 good ones.
|
|
35
|
+
- **The reflect quality judge's rubric names what it kept missing.** Tuned on
|
|
36
|
+
the labelled judge-gate set (plain judge, qwen3.8-27b): a description with a
|
|
37
|
+
sentence split by a stray period or an unbalanced quote is broken text; a
|
|
38
|
+
missing title, description or `when_to_use` is a concrete problem whether or
|
|
39
|
+
not the feedback mentions it, while a bare `type:` or a provenance stamp is
|
|
40
|
+
not; PRESERVATION and QUALITY are judged line by line in the changed region,
|
|
41
|
+
so a fix elsewhere no longer excuses a dropped fact, an invented or
|
|
42
|
+
strengthened claim, a hedge or a placeholder. On a stratified 100-case sample
|
|
43
|
+
the gate passed 31 of 40 good edits instead of 18 and 3 of 60 bad instead of 2.
|
|
44
|
+
- **A reflect quality judge on an agent engine reads only to verify what a
|
|
45
|
+
revision adds.** Its prompt names the revised asset's ref and adds one
|
|
46
|
+
paragraph: read that asset, or one the changed region names, with `akm_show`
|
|
47
|
+
only to check a fact the revision adds or alters, at most twice, and never
|
|
48
|
+
search; before scoring, find each added statement in the asset or the
|
|
49
|
+
feedback (a step or cause that merely seems to follow does not count), each
|
|
50
|
+
source fact in the revision, and each feedback point in a change to the text
|
|
51
|
+
it is about. The plain judge's prompt is unchanged. On the same 100 cases
|
|
52
|
+
(qwen3.8-27b, thinking on) the agent judge passed 38 of 40 good edits and 14
|
|
53
|
+
of 60 bad, against 34 and 12 for the plain judge on that model. Give the
|
|
54
|
+
engine's `llmEngine` `enableThinking: true`: without thinking the model
|
|
55
|
+
looped on tool calls.
|
|
56
|
+
- **A failed reflect reply reads the same on every engine kind:** the parser's
|
|
57
|
+
own message. 0.9.25-alpha.1 named the engine on an agent engine.
|
|
58
|
+
- **Inference reaches opencode only where akm writes opencode's config:** an
|
|
59
|
+
improve process's `llm` overlay on model work's agent, and an `opencode-sdk`
|
|
60
|
+
engine's `llmEngine` fallback model. Set the rest in your opencode config; a
|
|
61
|
+
task's or workflow's `inference` is an `untranslated-field` notice, as before
|
|
62
|
+
0.9.25-alpha.1. `claude` takes no `--effort`.
|
|
63
|
+
- **An `opencode-sdk` session the dispatch times out on or aborts is aborted on
|
|
64
|
+
the server for every dispatch,** not only model work.
|
|
65
|
+
- **An improve stage retries a reply only when the stage cannot read it,** not
|
|
66
|
+
when it misses the JSON Schema. A reply it can read costs no second call.
|
|
67
|
+
- **opencode model work can read the stash and search it.** The `akm-model-work`
|
|
68
|
+
agent reads, greps and globs in the stash and its working directory, edits
|
|
69
|
+
only in the working directory, and has `akm_search` and `akm_show` from the
|
|
70
|
+
akm-opencode plugin (0.9.21 or later, in your own opencode config). The
|
|
71
|
+
plugin's curation, learning and write gate are off for these dispatches, and
|
|
72
|
+
its state goes to akm's state directory, not `~/.local/state/akm-opencode`.
|
|
73
|
+
|
|
74
|
+
### Removed
|
|
75
|
+
|
|
76
|
+
- **The agent-engine inference fields,** which 0.9.25-alpha.1 accepted: they
|
|
77
|
+
fail to load again.
|
|
78
|
+
- **The `opencode-sdk` step watcher and the git-repository refusal for model
|
|
79
|
+
work's scratch directory.**
|
|
80
|
+
- **opencode model work's step limit.** At the limit opencode sends a "maximum
|
|
81
|
+
steps" message as a trailing assistant message, which a qwen chat template
|
|
82
|
+
(LM Studio, llama-server) renders as the start of the model's reply: LM
|
|
83
|
+
Studio returned nothing, llama-server returned that message as the answer,
|
|
84
|
+
and the dispatch failed. The dispatch timeout bounds a run.
|
|
85
|
+
- **A prompt builder that nothing called** (`buildSchemaRepairPrompt`).
|
|
86
|
+
- **The `--track-usage` / `--no-track-usage` flag on `akm search`, `akm curate`
|
|
87
|
+
and `akm show`.** A successful read always records its usage event, stamped
|
|
88
|
+
with its source (`user`, `improve`, `task` or `audit`); only `user` events
|
|
89
|
+
feed ranking and eval, so machine reads never skew them. Either spelling now
|
|
90
|
+
fails as an unknown flag.
|
|
91
|
+
|
|
92
|
+
### Fixed
|
|
93
|
+
|
|
94
|
+
- **The reflect size guard no longer flags a body that does not grow.** The
|
|
95
|
+
expansion ceiling is capped at 25,000 characters, so a source body longer than
|
|
96
|
+
that was flagged `EXCESSIVE_EXPANSION` even when the proposed body was its own
|
|
97
|
+
length (ratio 1.00), and went to review instead of the judge. A body no longer
|
|
98
|
+
than its source is never expansion; one that grows past the cap still is, by
|
|
99
|
+
any amount. The reflect prompt agrees: it told the model its body could be at
|
|
100
|
+
most 25,000 characters even when the source was longer, and now gives such a
|
|
101
|
+
source's own length.
|
|
102
|
+
- **An asset's or a task's own `tools:` can no longer name the model-work
|
|
103
|
+
policy.** In 0.9.25-alpha.1 exactly `read`, `edit`, `akm search`, `akm show`,
|
|
104
|
+
in that order, skipped `execution.allowedTools`. It is ordinary tools now:
|
|
105
|
+
only akm's own model-work callers ask for the policy.
|
|
106
|
+
- **An agent engine that names its model only in `args` is named in the usage
|
|
107
|
+
report,** where it showed `unattributed`.
|
|
108
|
+
- **`opencode` and `opencode-sdk` engines receive the XDG base-directory
|
|
109
|
+
variables.** Under a custom `XDG_CONFIG_HOME` the spawned opencode missed its
|
|
110
|
+
provider config and failed every dispatch with `Unexpected server error`.
|
|
111
|
+
|
|
9
112
|
## [0.9.25-alpha.1] - 2026-10-02
|
|
10
113
|
|
|
11
114
|
### Changed
|
|
12
115
|
|
|
13
|
-
- **
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
`
|
|
24
|
-
|
|
25
|
-
|
|
116
|
+
- **Unattended model work runs under one tool policy, on any engine that
|
|
117
|
+
confines it.** The model may read, edit inside a scratch directory akm removes
|
|
118
|
+
after the dispatch, and run `akm search` and `akm show`; the stash stays
|
|
119
|
+
read-only. An LLM engine has no tools, `claude` confines the policy, and
|
|
120
|
+
`opencode` and `opencode-sdk` run an injected `akm-model-work` agent with
|
|
121
|
+
read and edit only. Every other harness refuses it before the request starts.
|
|
122
|
+
- **One config rule names the engines model work may use.** Every key it reads
|
|
123
|
+
its engine from (`defaults.llmEngine`, `index.*.engine`, a strategy's or
|
|
124
|
+
process's `engine`, an enabled triage `judgment.engine`, a
|
|
125
|
+
`qualityGate.engine`) must name an LLM engine or a `claude`, `opencode` or
|
|
126
|
+
`opencode-sdk` agent engine, or the config fails to load. `--require-engines`
|
|
127
|
+
and `akm health` check agent engines too.
|
|
128
|
+
- **Model work is bounded at 600 seconds on every engine kind** whose engine
|
|
129
|
+
sets no `timeoutMs`; agent and `opencode-sdk` engines ran until they finished.
|
|
130
|
+
- **`akm proposal new` returns the proposal as JSON on every engine kind,**
|
|
131
|
+
with no draft file and no live session. A reply that is not a proposal gets
|
|
132
|
+
one retry.
|
|
133
|
+
- **Reflect asks every engine kind for the same JSON reply and repairs it
|
|
134
|
+
once.** An agent or `opencode-sdk` engine gets reflect's JSON Schema and a
|
|
135
|
+
repair turn, as an LLM engine did. An LLM engine's requests are unchanged.
|
|
26
136
|
- **An improve stage's reply that fails its JSON Schema gets one corrective
|
|
27
|
-
retry
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
- **Agent and `opencode-sdk` dispatches leave a usage record.**
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
validates it. A reply that is not a proposal gets one corrective retry, then
|
|
46
|
-
fails with an error that names the engine. An agent CLI now runs headless:
|
|
47
|
-
you see the queued proposal, not the agent at work, and no draft file is
|
|
48
|
-
written.
|
|
49
|
-
- **Unattended model work has one tool policy, which each engine confines or
|
|
50
|
-
refuses when the request is built.** The policy allows reading, editing only
|
|
51
|
-
inside a scratch working directory that akm creates for the dispatch and
|
|
52
|
-
removes after it, and running `akm search` and `akm show`. The stash stays
|
|
53
|
-
read-only. An LLM engine has no tools. `claude` runs it with `--restricted`
|
|
54
|
-
(its user, project and local settings, whose allow rules could pre-approve
|
|
55
|
-
any command or path, are ignored, and its file tools stay in the working
|
|
56
|
-
directory), `--strict-mcp-config`, `--tools Read,Edit,Bash`, `--allowedTools`
|
|
57
|
-
for Read, Edit and the two `akm` commands, and `--permission-mode dontAsk`.
|
|
58
|
-
`opencode` and `opencode-sdk` run an injected `akm-model-work` agent that
|
|
59
|
-
can read and edit only inside the working directory and has no bash:
|
|
60
|
-
opencode checks a bash rule against the command's words only, so
|
|
61
|
-
`akm show x > ~/stash/asset.md` would pass an `akm show *` rule. The agent
|
|
62
|
-
has its own short prompt in place of opencode's coding-assistant prompt, an
|
|
63
|
-
8-step limit, and no automatic compaction. opencode only asks the model to
|
|
64
|
-
stop at that limit, so `opencode-sdk` aborts the session two steps past it,
|
|
65
|
-
and aborts a session the dispatch times out on; on `opencode` the dispatch
|
|
66
|
-
timeout is the bound. Every other harness refuses the policy. Such a
|
|
67
|
-
dispatch ignores the engine's `args` and `workspace`, refuses to start when
|
|
68
|
-
the temporary directory is inside a git repository, which opencode would
|
|
69
|
-
treat as its working directory, and fails with `parse_error` when the agent
|
|
70
|
-
ends with no answer, as opencode can at its step limit.
|
|
71
|
-
- **Unattended model work runs on any engine that confines the model-work
|
|
72
|
-
tool policy, and config checks that with one rule.**
|
|
73
|
-
- **Who sends the policy.** The improve processes, the quality, triage and
|
|
74
|
-
retrieval-gate judges, index passes and `akm remember --enrich` now send
|
|
75
|
-
it, so they may run on an LLM engine or on a `claude`, `opencode` or
|
|
76
|
-
`opencode-sdk` agent engine, where they could only use an LLM engine
|
|
77
|
-
before.
|
|
78
|
-
- **The one rule.** Every key model work reads its engine from must name
|
|
79
|
-
such an engine. The keys are `defaults.llmEngine`, `index.defaults.engine`,
|
|
80
|
-
`index.<pass>.engine`, a strategy's `engine`, a process's `engine`, an
|
|
81
|
-
enabled triage `judgment.engine`, and a `qualityGate.engine`. A config
|
|
82
|
-
that breaks the rule fails to load, naming the key, the engine and its
|
|
83
|
-
platform. The rule replaces six checks that each required an LLM engine
|
|
84
|
-
for some of those keys, or let the triage judgment use any agent.
|
|
85
|
-
- **What else is gone.** The plan, reflect, the quality gate and index
|
|
86
|
-
passes no longer turn an agent engine away. A triage judgment on an agent
|
|
87
|
-
engine takes a strategy's `llm` overrides as `untranslated-field`
|
|
88
|
-
notices, where it was refused before.
|
|
89
|
-
- **Reflect on an agent engine.** The agent returns its proposal as JSON on
|
|
90
|
-
stdout instead of writing a draft file, which the policy's scratch working
|
|
91
|
-
directory would not keep.
|
|
92
|
-
- **The answer is unwrapped.** A model-work reply from an agent engine goes
|
|
93
|
-
through its harness's result extractor, so a stage call on `claude` gets
|
|
94
|
-
the answer, not the `--output-format json` envelope around it.
|
|
95
|
-
- **`--require-engines` checks agent engines too.** It checks an agent
|
|
96
|
-
engine by its binary on PATH, and an `opencode-sdk` engine by its binary
|
|
97
|
-
and its LLM fallback's endpoint. It probed only LLM connections before.
|
|
98
|
-
The usage report and `akm health` count a process's calls whatever its
|
|
99
|
-
engine's kind.
|
|
100
|
-
- **An `opencode-sdk` engine gets its LLM fallback connection only from its own
|
|
101
|
-
`llmEngine`. `defaults.llmEngine` no longer supplies one.** An SDK engine
|
|
102
|
-
that set no `llmEngine` used to borrow `defaults.llmEngine`: its connection,
|
|
103
|
-
its model (unless the engine set its own) and its timeout. `defaults.llmEngine`
|
|
104
|
-
names the default engine for unattended model work, and since that work may
|
|
105
|
-
now run on an agent engine it can be one too, so it is no longer also a
|
|
106
|
-
connection that every SDK engine shares. **If you relied on the inheritance,
|
|
107
|
-
set `llmEngine` on the SDK engine**, for example
|
|
108
|
-
`"sdk": { "kind": "agent", "platform": "opencode-sdk", "llmEngine": "fast" }`.
|
|
109
|
-
Without one, the SDK engine runs on opencode's own provider and auth, and on
|
|
110
|
-
its own `model` if it sets one: akm sends it no connection. An SDK engine
|
|
111
|
-
that sets `llmEngine` is unchanged, and so is a config with no `opencode-sdk`
|
|
112
|
-
engine. The rule holds everywhere the fallback is read: dispatch, a
|
|
113
|
-
workflow's frozen concurrency cap, `akm health`, and
|
|
114
|
-
`akm improve --require-engines`, which now check an SDK engine's fallback
|
|
115
|
-
endpoint only when it sets `llmEngine`.
|
|
116
|
-
- **Inference reaches `opencode`, `opencode-sdk` and `claude` engines.** A
|
|
117
|
-
request's `temperature`, `maxTokens`, `contextLength`, `enableThinking` and
|
|
118
|
-
`reasoningEffort` came from the engine's own settings, an improve process's
|
|
119
|
-
`llm` overlay, a task, command or agent asset's `inference`, a workflow's
|
|
120
|
-
`llm:` and a `models.json` alias, and every agent engine dropped all of them
|
|
121
|
-
with an `untranslated-field` notice. Each platform now translates what it can
|
|
122
|
-
carry, and the nearest layer wins, field by field, as on an LLM engine. With
|
|
123
|
-
no setting anywhere akm sends nothing of its own, so the model's configured
|
|
124
|
-
default still applies, such as a `reasoningEffort: "none"` in your opencode
|
|
125
|
-
config.
|
|
126
|
-
- **`claude`** gets `reasoningEffort` as `--effort <level>`, passed as given
|
|
127
|
-
(Claude Code 2.1.283 takes `low`, `medium`, `high`, `xhigh` and `max`).
|
|
128
|
-
Its other four fields are still reported as untranslated.
|
|
129
|
-
- **`opencode` and `opencode-sdk`** get inference as opencode config, which
|
|
130
|
-
merges over your own opencode config for the same provider and model:
|
|
131
|
-
`temperature` and `reasoningEffort` become `options.temperature` and
|
|
132
|
-
`options.reasoningEffort` (opencode drops the snake_case spelling),
|
|
133
|
-
`enableThinking` becomes both wire forms an LLM engine sends, and
|
|
134
|
-
`maxTokens` with `contextLength` becomes `limit.output` and
|
|
135
|
-
`limit.context`. Set both limit fields: opencode refuses half a limit, and
|
|
136
|
-
a half would overwrite the other half of one you declared, so a lone one
|
|
137
|
-
is reported as untranslated. Without `limit.output` opencode asks for
|
|
138
|
-
`max_tokens: 32000`, which a small-context server rejects, and a wrong
|
|
139
|
-
`limit.context` lets it build requests past the server's window. The fields
|
|
140
|
-
need a `provider/model`: the request's model, or the `--model` an
|
|
141
|
-
`opencode` engine's `args` name. Model work needs none for its options, so
|
|
142
|
-
an `opencode-sdk` engine with no `model` and no `llmEngine` gets them too.
|
|
143
|
-
- **Where it goes.** Model work puts the options on the `akm-model-work`
|
|
144
|
-
agent that runs it, so opencode's own title call on the same model keeps
|
|
145
|
-
the model's defaults. Any other dispatch runs your own agent, whose name
|
|
146
|
-
akm cannot rely on, so its options go on the model and the title call sees
|
|
147
|
-
them too. `opencode-sdk` makes no title call. An `opencode-sdk` engine
|
|
148
|
-
declares its `llmEngine` fallback's model with the fallback's own
|
|
149
|
-
inference, under the engine's and the request's; each distinct set starts
|
|
150
|
-
its own `opencode serve`, as a different model does.
|
|
151
|
-
- **An agent engine may set the inference fields its platform translates.**
|
|
152
|
-
`engines.<name>` of `kind: "agent"` took none, so an engine could not carry a
|
|
153
|
-
`temperature` or a `reasoningEffort` of its own. `opencode` and
|
|
154
|
-
`opencode-sdk` may set `temperature`, `maxTokens`, `contextLength`,
|
|
155
|
-
`enableThinking` and `reasoningEffort`, `claude` may set `reasoningEffort`,
|
|
156
|
-
and every other platform none. A field its platform does not translate fails
|
|
157
|
-
to load, naming the platform and the fields it does translate. `provider`,
|
|
158
|
-
`endpoint`, `apiKey`, `apiKeyFile`, `concurrency` and `extraParams` stay
|
|
159
|
-
invalid on an agent engine.
|
|
160
|
-
- **Reasoning effort has one word in a request, `reasoningEffort`.** The
|
|
161
|
-
starter `reasoning` alias, an alias in your `models.json` and an asset's
|
|
162
|
-
`effort:` frontmatter said `effort`, and engines, opencode and the LLM
|
|
163
|
-
request said `reasoningEffort`. `effort` is now read as `reasoningEffort`
|
|
164
|
-
where layers are merged, so the nearest layer wins whichever word it used.
|
|
165
|
-
On an LLM engine an alias's or asset's `effort` is therefore sent as
|
|
166
|
-
`reasoning_effort`; it was reported as untranslated and dropped before. An
|
|
167
|
-
LLM engine's own request is unchanged.
|
|
168
|
-
- **Reflect asks every engine kind for the same JSON reply, checks it the same
|
|
169
|
-
way and repairs it once.** An LLM engine was sent the reply's JSON Schema,
|
|
170
|
-
and its reply was held to exact fields and repaired once. An agent or
|
|
171
|
-
`opencode-sdk` engine got a looser contract in its prompt (`ref`, `content`
|
|
172
|
-
and an optional `frontmatter`) with no schema and no repair, so one invalid
|
|
173
|
-
reply failed the run. Every engine kind now runs the same iteration:
|
|
174
|
-
- **The request.** The prompt carries the same output contract, and the
|
|
175
|
-
reply's JSON Schema is the request's output schema: `response_format` for
|
|
176
|
-
an LLM engine, as before, and the schema instruction at the end of the
|
|
177
|
-
prompt for an agent engine. `claude` also gets `--output-format json`,
|
|
178
|
-
which akm unwraps. An LLM endpoint that rejects JSON Schema still gets the
|
|
179
|
-
framed-markdown contract.
|
|
180
|
-
- **The reply.** An agent now returns `content`, `confidence` and a
|
|
181
|
-
`frontmatterPatch` of `description` and `when_to_use`, and `ref` too when
|
|
182
|
-
no asset was named, as an LLM does. akm derives a named asset's ref and
|
|
183
|
-
merges the patch with the source's frontmatter, so an agent can no longer
|
|
184
|
-
set other frontmatter keys or retarget the proposal.
|
|
185
|
-
- **The repair.** A reply that fails the contract gets one repair turn that
|
|
186
|
-
carries the first reply, shared across self-refine passes. A reply that is
|
|
187
|
-
still invalid fails with `parse_error` and queues nothing. On an agent or
|
|
188
|
-
`opencode-sdk` engine the error names the engine, as in
|
|
189
|
-
`Engine "oc" reply was not a valid reflect proposal after 2 attempts: …`.
|
|
190
|
-
An LLM engine's message is unchanged, because improve feeds it into later
|
|
191
|
-
prompts as a pattern to avoid. A failed agent dispatch is still reported
|
|
192
|
-
with its exit code and stderr.
|
|
193
|
-
- **What reflect sends and reports on an agent engine.** Reflect asks every
|
|
194
|
-
engine kind for no visible chain of thought (`enableThinking: false`).
|
|
195
|
-
`opencode` and `opencode-sdk` carry it, as the inference entry above
|
|
196
|
-
says; `claude` reports it as an `untranslated-field` notice, as it does
|
|
197
|
-
for every other stage. `reflect_completed` carries `outputMode` and
|
|
198
|
-
`repairAttempts` for every engine kind.
|
|
199
|
-
- **Unchanged.** An LLM engine's requests are byte-identical to before:
|
|
200
|
-
reflect's generation and repair, and its quality judge. So are the
|
|
201
|
-
refine passes, the content budget (an LLM engine's context length), the
|
|
202
|
-
protected frontmatter fields, the review routing and the judge selection.
|
|
137
|
+
retry,** then the stage reads the last reply with its own parser.
|
|
138
|
+
- **One schema instruction for every agent engine,** appended by the shared
|
|
139
|
+
request lowering: `opencode` and `opencode-sdk` now receive a requested
|
|
140
|
+
schema, and a workflow unit on seven harnesses no longer carries it twice.
|
|
141
|
+
- **Agent and `opencode-sdk` dispatches leave a usage record.**
|
|
142
|
+
- **An `opencode-sdk` engine's LLM fallback comes only from its own
|
|
143
|
+
`llmEngine`; `defaults.llmEngine` no longer supplies one.** If you relied on
|
|
144
|
+
that, set `llmEngine` on the SDK engine. Without one, opencode uses its own
|
|
145
|
+
provider, model and auth.
|
|
146
|
+
- **Inference reaches `opencode`, `opencode-sdk` and `claude` engines.**
|
|
147
|
+
`temperature`, `maxTokens`, `contextLength`, `enableThinking` and
|
|
148
|
+
`reasoningEffort` from an engine, an improve process's `llm` overlay, an
|
|
149
|
+
asset, a workflow or a `models.json` alias were dropped on every agent
|
|
150
|
+
engine. An agent engine may set them, and `claude` takes `reasoningEffort` as
|
|
151
|
+
`--effort`.
|
|
152
|
+
- **Reasoning effort has one word, `reasoningEffort`.** `effort` in a
|
|
153
|
+
`models.json` alias or an asset is read as it, so an LLM engine now sends it
|
|
154
|
+
as `reasoning_effort`.
|
|
203
155
|
|
|
204
156
|
### Removed
|
|
205
157
|
|
|
206
|
-
- **
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
native channel, as codex does with `--output-schema`. The fields, their two
|
|
211
|
-
types and the tests that pinned their values are gone. Nothing you run
|
|
212
|
-
changes.
|
|
213
|
-
- **`resolveLlmEngineUse`'s swap of an agent engine for an LLM engine.** Given
|
|
214
|
-
an agent engine, it used the engine's `llmEngine`, then `defaults.llmEngine`,
|
|
215
|
-
and warned. Only the implicit SDK fallback above could reach it, so nothing
|
|
216
|
-
you run changes. An agent engine passed to it is now an error.
|
|
217
|
-
- **An `effort` hint on the agent dispatch request that nothing read.** The
|
|
218
|
-
lowering set it from `inference.effort`, reserved for a workflow field, and
|
|
219
|
-
no builder consumed it. `reasoningEffort` in the request's inference is read
|
|
220
|
-
where it is translated, and the field is gone.
|
|
221
|
-
- **The agent file-write contract.** An agent was once told to write its
|
|
222
|
-
proposal to a draft file and print `DRAFT_WRITTEN confidence=<n>`. Reflect and
|
|
223
|
-
`akm proposal new` had already stopped sending that instruction, because the
|
|
224
|
-
model-work scratch directory does not outlive a dispatch. The instruction, the
|
|
225
|
-
function that read the `DRAFT_WRITTEN` line and the `draftFilePath` prompt
|
|
226
|
-
inputs that nothing passed any more are gone, with their tests. So is
|
|
227
|
-
reflect's `ref_mismatch` check: a reflect that names an asset derives the
|
|
228
|
-
proposal's ref, so an engine can no longer name another one. Nothing you run
|
|
229
|
-
changes.
|
|
158
|
+
- **Dead code; nothing you run changes:** each harness's unused `pattern` and
|
|
159
|
+
`structuredOutput` fields, `resolveLlmEngineUse`'s swap of an agent engine
|
|
160
|
+
for an LLM engine, the agent request's unread `effort` hint, and the agent
|
|
161
|
+
file-write contract (`DRAFT_WRITTEN`, reflect's `ref_mismatch` check).
|
|
230
162
|
|
|
231
163
|
### Fixed
|
|
232
164
|
|
|
233
|
-
- **An `opencode-sdk` engine with an LLM fallback
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
- **
|
|
246
|
-
|
|
247
|
-
response has started with HTTP 200 and a body that holds only an `error`
|
|
248
|
-
object and no `choices`. akm returned that as an empty reply with
|
|
249
|
-
`ok: true`. It is now a provider error with the body in its message, as an
|
|
250
|
-
error status is.
|
|
251
|
-
- **An `opencode` engine runs a persona instead of failing.** akm passed the
|
|
252
|
-
persona to `opencode run` as `--system-prompt`, which opencode 1.18 does
|
|
253
|
-
not accept, so opencode printed its usage and exited 1 on every dispatch
|
|
254
|
-
that carried a persona, including `akm agent <agent-ref> --engine opencode`.
|
|
255
|
-
akm now composes the persona into the prompt in an `<AKM_PERSONA>` block,
|
|
256
|
-
as it does for harnesses with no system-prompt option.
|
|
257
|
-
- **`opencode` and `opencode-sdk` engines receive a requested output schema.**
|
|
258
|
-
They dropped it with only an `untranslated-field` warning, so a schema from
|
|
259
|
-
`akm agent`, `akm command run`, a command's frontmatter or a task's
|
|
260
|
-
`output:` never reached the model. They now get it as the same instruction
|
|
261
|
-
every other agent engine gets.
|
|
262
|
-
- **An LLM engine sends a requested schema unless it opts out.** It sent
|
|
263
|
-
`response_format` only when the engine set `supportsJsonSchema: true`, so an
|
|
264
|
-
engine that left the flag unset never had its output constrained. It now
|
|
265
|
-
sends it unless the engine sets `supportsJsonSchema: false`. An endpoint
|
|
266
|
-
that rejects it with a 4xx is still retried once without it.
|
|
267
|
-
- **A workflow unit on `claude`, `copilot`, `gemini`, `pi`, `aider`,
|
|
268
|
-
`amazonq` or `openhands` gets the schema instruction once.** The unit prompt
|
|
269
|
-
carried it and the harness builder appended a second copy.
|
|
270
|
-
- **A feature gate's timeout now stops the call it bounds.** When an improve
|
|
271
|
-
stage's gate timed out (600 seconds unless the call sets its own), the model
|
|
272
|
-
call kept running in the background, on an engine with `timeoutMs: 900000`
|
|
273
|
-
for up to five more minutes. The gate now aborts it.
|
|
274
|
-
- **A stage call reports a timeout or abort by the dispatch's own reason.** A
|
|
275
|
-
timed-out or aborted agent or `opencode-sdk` dispatch, and an LLM timeout
|
|
276
|
-
inside a feature gate, came back as `error`. They now come back as `timeout`
|
|
277
|
-
or `aborted`.
|
|
278
|
-
|
|
279
|
-
- **`akm proposal new` keeps the reply's confidence.** The engine's
|
|
280
|
-
self-rated `confidence` was parsed and then dropped, so a proposal from
|
|
281
|
-
`proposal new` never carried the field the reference says it has.
|
|
282
|
-
- **Model work on an agent or `opencode-sdk` engine that sets no `timeoutMs` now
|
|
283
|
-
stops after 600 seconds, as documented.** Such an engine resolved to an
|
|
284
|
-
explicit "no timeout", so the 600-second bound for model work never applied
|
|
285
|
-
to it. An engine's own `timeoutMs`, `null` included, still applies, and
|
|
286
|
-
other work on an agent engine still runs until it finishes.
|
|
287
|
-
- **The announcement of the implicit `opencode-sdk` fallback is now true.** It
|
|
288
|
-
says provider, model and auth come from opencode's own configuration, but the
|
|
289
|
-
fallback engine borrowed `defaults.llmEngine`'s connection whenever one was
|
|
290
|
-
set, so opencode got an akm-generated provider and model instead. It now runs
|
|
291
|
-
on opencode's own configuration, as announced.
|
|
292
|
-
- **`model: reasoning` set no effort on `claude`, `opencode` or `opencode-sdk`.**
|
|
293
|
-
The starter alias supplies `effort: high` for all three, and each dropped it
|
|
294
|
-
with an `untranslated-field` notice, so the alias chose a stronger model and
|
|
295
|
-
nothing more. It now sets `--effort high` on `claude` and
|
|
296
|
-
`options.reasoningEffort` on opencode (see "Inference reaches `opencode`,
|
|
297
|
-
`opencode-sdk` and `claude` engines" above). An improve process's
|
|
298
|
-
`llm.reasoningEffort` and `llm.temperature` overlay now reaches those
|
|
299
|
-
engines the same way.
|
|
165
|
+
- **An `opencode-sdk` engine with an LLM fallback reaches its endpoint, and a
|
|
166
|
+
failed dispatch is a failure (#1015).** An SDK error or provider rejection is
|
|
167
|
+
`ok: false` with opencode's message, and a reply's last text part is its answer.
|
|
168
|
+
- **An LLM engine reports a provider error sent with HTTP 200 as a failure**
|
|
169
|
+
(OpenRouter does this), not as an empty reply.
|
|
170
|
+
- **An `opencode` engine runs a persona** instead of failing on
|
|
171
|
+
`--system-prompt`, which opencode 1.18 rejects.
|
|
172
|
+
- **An LLM engine sends a requested schema** unless it sets
|
|
173
|
+
`supportsJsonSchema: false`; it sent one only when it set `true`.
|
|
174
|
+
- **A feature gate's timeout stops the call it bounds,** and a stage call
|
|
175
|
+
reports a timeout or abort as `timeout` or `aborted`, not `error`.
|
|
176
|
+
- **`akm proposal new` keeps the reply's `confidence`.**
|
|
177
|
+
- **Model work on an agent or `opencode-sdk` engine with no `timeoutMs` stops
|
|
178
|
+
after 600 seconds** as documented; it resolved to no timeout.
|
|
300
179
|
|
|
301
180
|
## [0.9.24] - 2026-10-02
|
|
302
181
|
|
package/dist/cli.js
CHANGED
|
@@ -177,7 +177,7 @@ const setupCommand = defineCommand({
|
|
|
177
177
|
// the work, matching the `sync --push/--no-push` pattern. A flag
|
|
178
178
|
// DECLARED as `no-init` can never be negated: `--no-init` parses as
|
|
179
179
|
// "negate `init`", a name nothing declared, leaving the real key at its
|
|
180
|
-
// default forever
|
|
180
|
+
// default forever.
|
|
181
181
|
init: {
|
|
182
182
|
type: "boolean",
|
|
183
183
|
default: true,
|
|
@@ -506,6 +506,7 @@ async function judgeOne(ctx, candidate) {
|
|
|
506
506
|
signal: ctx.opts.signal,
|
|
507
507
|
...(ctx.chat ? { chat: ctx.chat } : {}),
|
|
508
508
|
},
|
|
509
|
+
parse: parsePairJudgeResponse,
|
|
509
510
|
...(ctx.opts.onNotices ? { onNotices: ctx.opts.onNotices } : {}),
|
|
510
511
|
});
|
|
511
512
|
if (!outcome.ok)
|
|
@@ -53,6 +53,10 @@ import { resolveImproveStrategy, resolveProcessEnabled } from "./improve-strateg
|
|
|
53
53
|
import { isContentDrivenRow, isLedgerBlocked, ledgerKey, loadLedgerSnapshot, recordLedgerAttempt } from "./ledger.js";
|
|
54
54
|
import { isInRetrievalScope, loadRetrievalScope } from "./retrieval-scope.js";
|
|
55
55
|
import { callStage, mintProposal, noticeSet, stageRunner } from "./stage.js";
|
|
56
|
+
function parsePlan(raw) {
|
|
57
|
+
const plan = parseEmbeddedJsonResponse(raw);
|
|
58
|
+
return plan && Array.isArray(plan.operations) ? { ...plan, operations: plan.operations } : undefined;
|
|
59
|
+
}
|
|
56
60
|
/** A plan op worth acting on. Retired advisory ops (merge/delete/contradict) are dropped, never thrown on. */
|
|
57
61
|
export function isValidOp(op) {
|
|
58
62
|
if (typeof op !== "object" || op === null)
|
|
@@ -570,6 +574,7 @@ async function judgeConsolidationChunks(args) {
|
|
|
570
574
|
...(Object.hasOwn(llmRunner, "timeoutMs") ? { timeoutMs: llmRunner.timeoutMs } : {}),
|
|
571
575
|
signal: opts.signal,
|
|
572
576
|
},
|
|
577
|
+
parse: parsePlan,
|
|
573
578
|
...(opts.onNotices ? { onNotices: opts.onNotices } : {}),
|
|
574
579
|
});
|
|
575
580
|
if (!outcome.ok) {
|
|
@@ -577,8 +582,8 @@ async function judgeConsolidationChunks(args) {
|
|
|
577
582
|
continue;
|
|
578
583
|
}
|
|
579
584
|
warnVerbose(`[akm:consolidate] ${label} raw response (first 500 chars): ${outcome.raw.slice(0, 500)}`);
|
|
580
|
-
const parsed =
|
|
581
|
-
if (!parsed
|
|
585
|
+
const parsed = parsePlan(outcome.raw);
|
|
586
|
+
if (!parsed) {
|
|
582
587
|
const hint = outcome.raw.trim() === "" ? " (empty response — if using a thinking model, disable thinking mode)" : "";
|
|
583
588
|
const msg = `Chunk ${chunkIdx + 1}: invalid plan from AI — skipping.${hint}`;
|
|
584
589
|
warn(msg);
|
|
@@ -2,7 +2,6 @@
|
|
|
2
2
|
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
3
|
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
4
|
import { deepMergeConfig } from "../../core/config/deep-merge.js";
|
|
5
|
-
import { MODEL_WORK_TOOLS } from "../../execution/source.js";
|
|
6
5
|
import { buildExecution, resolveExecution } from "../../integrations/agent/execution.js";
|
|
7
6
|
function own(value, key) {
|
|
8
7
|
return value !== undefined && Object.hasOwn(value, key);
|
|
@@ -36,12 +35,12 @@ export function resolveImproveExecution(options) {
|
|
|
36
35
|
if (selectedEngine === undefined || selectedEngine === null)
|
|
37
36
|
return null;
|
|
38
37
|
const invocationDefaults = mergeDefaults(mergeDefaults(defaultEngine ? { engine: defaultEngine } : {}, profileDefaults), indexDefaults);
|
|
39
|
-
const current = { ...mergeDefaults(processDefaults, currentDefaults), tools: MODEL_WORK_TOOLS };
|
|
40
38
|
const prepared = resolveExecution({
|
|
41
39
|
content: `improve ${options.processName} execution selection`,
|
|
42
40
|
config: options.config,
|
|
43
41
|
invocationDefaults,
|
|
44
|
-
current,
|
|
42
|
+
current: mergeDefaults(processDefaults, currentDefaults),
|
|
43
|
+
modelWork: true,
|
|
45
44
|
});
|
|
46
45
|
const lowered = buildExecution(prepared.request, prepared.runner);
|
|
47
46
|
return Object.freeze({ runner: lowered.runner, notices: lowered.notices });
|
|
@@ -741,6 +741,7 @@ function resolveExtractRun(options, config, process, activeProfile) {
|
|
|
741
741
|
...(options.signal ? { signal: options.signal } : {}),
|
|
742
742
|
...(options.chat ? { chat: options.chat } : {}),
|
|
743
743
|
},
|
|
744
|
+
parse: parseSessionSummary,
|
|
744
745
|
onNotices: notices.add,
|
|
745
746
|
});
|
|
746
747
|
return parseSessionSummary(outcome.ok ? outcome.raw : "");
|
|
@@ -20,13 +20,15 @@ import { collectEngineCredentialValues } from "../../integrations/agent/engine-r
|
|
|
20
20
|
import { probeLlmReachable } from "../../llm/client.js";
|
|
21
21
|
import { getOutputMode } from "../../output/context.js";
|
|
22
22
|
import { deliverRendered } from "../../output/html-render.js";
|
|
23
|
+
import { readStdin } from "../../runtime.js";
|
|
23
24
|
import { akmImprove, IMPROVE_TARGET_FLAG, resolveImproveReadSource } from "./improve.js";
|
|
24
25
|
import { runImproveReportQuery } from "./improve-report.js";
|
|
25
26
|
import { buildImproveRunId, recordImproveRunResult, recordTerminatedImproveRun, } from "./improve-result-file.js";
|
|
26
27
|
import { runImproveSession } from "./improve-session.js";
|
|
27
|
-
import { resolveImprovePlan, } from "./improve-strategies.js";
|
|
28
|
+
import { resolveImprovePlan, resolveImproveStrategy, } from "./improve-strategies.js";
|
|
28
29
|
import { formatUsageReportTable } from "./improve-usage-report.js";
|
|
29
30
|
import { renderReflectPromptPreview } from "./reflect.js";
|
|
31
|
+
import { resolveQualityGateJudge, runReflectQualityJudge } from "./stage.js";
|
|
30
32
|
let akmImproveForRun = akmImprove;
|
|
31
33
|
/** Swap the CLI's improve work implementation in deterministic subprocess tests. */
|
|
32
34
|
export function _setAkmImproveForTests(fake) {
|
|
@@ -188,6 +190,32 @@ function rejectReportOnlyFlags(args) {
|
|
|
188
190
|
return;
|
|
189
191
|
throw new UsageError(`\`${flag}\` only applies to \`akm improve report\`. Use \`akm improve report ${flag} <value>\` instead.`, "INVALID_FLAG_VALUE");
|
|
190
192
|
}
|
|
193
|
+
/**
|
|
194
|
+
* `akm improve judge`: reflect's quality judge on one revision, read as
|
|
195
|
+
* `{"source", "candidate", "feedback"}` JSON from stdin, with the engine the
|
|
196
|
+
* strategy's reflect quality gate names. It writes nothing.
|
|
197
|
+
*/
|
|
198
|
+
async function runImproveJudgeCli(strategyName) {
|
|
199
|
+
const input = process.stdin.isTTY
|
|
200
|
+
? {}
|
|
201
|
+
: JSON.parse((await readStdin()).toString("utf8"));
|
|
202
|
+
const { source, candidate, feedback, ref } = input;
|
|
203
|
+
if (typeof source !== "string" || typeof candidate !== "string") {
|
|
204
|
+
throw new UsageError('`akm improve judge` reads {"source": "...", "candidate": "...", "feedback": "...", "ref": "..."} JSON from stdin.', "MISSING_REQUIRED_ARGUMENT");
|
|
205
|
+
}
|
|
206
|
+
const config = loadConfig();
|
|
207
|
+
const judge = resolveQualityGateJudge(config, resolveImproveStrategy(strategyName, config).config, "reflect");
|
|
208
|
+
if (!judge) {
|
|
209
|
+
throw new ConfigError("`akm improve judge` judges with the reflect quality gate's engine. Set processes.reflect.qualityGate.engine.", "INVALID_CONFIG_FILE");
|
|
210
|
+
}
|
|
211
|
+
const notes = typeof feedback === "string" && feedback.trim() !== "" ? [feedback.trim()] : [];
|
|
212
|
+
const verdict = await runReflectQualityJudge(config, candidate, source, notes, undefined, {
|
|
213
|
+
runnerSelectionFrozen: true,
|
|
214
|
+
llmRunner: judge,
|
|
215
|
+
...(typeof ref === "string" && ref ? { ref } : {}),
|
|
216
|
+
});
|
|
217
|
+
output("improve-judge", { engine: judge.engine, ...verdict });
|
|
218
|
+
}
|
|
191
219
|
export const improveCommand = defineCommand({
|
|
192
220
|
meta: {
|
|
193
221
|
name: "improve",
|
|
@@ -270,6 +298,10 @@ export const improveCommand = defineCommand({
|
|
|
270
298
|
return;
|
|
271
299
|
}
|
|
272
300
|
rejectReportOnlyFlags(args);
|
|
301
|
+
if (getStringArg(args, "scope") === "judge") {
|
|
302
|
+
await runImproveJudgeCli(getStringArg(args, "strategy"));
|
|
303
|
+
return;
|
|
304
|
+
}
|
|
273
305
|
rejectRetiredImproveTargetFlag();
|
|
274
306
|
const jsonToStdout = args["json-to-stdout"];
|
|
275
307
|
const targetArg = getStringArg(args, "bundle");
|
|
@@ -149,6 +149,9 @@ async function runLoopReflectPass(planned, env, tally) {
|
|
|
149
149
|
...(reflectErrors.length > 0 ? { avoidPatterns: [...reflectErrors] } : {}),
|
|
150
150
|
eventSource: "improve",
|
|
151
151
|
lowValueFilter: improveProfile.processes?.reflect?.lowValueFilter?.enabled === true,
|
|
152
|
+
...(improveProfile.processes?.reflect?.defectFilter
|
|
153
|
+
? { defectFilter: improveProfile.processes.reflect.defectFilter }
|
|
154
|
+
: {}),
|
|
152
155
|
...(budgetMs > 0 ? { timeoutMs: budgetMs } : {}),
|
|
153
156
|
signal: env.budgetSignal,
|
|
154
157
|
eventsCtx: env.eventsCtx,
|