akm-cli 0.9.25-alpha.1 → 0.9.25-alpha.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/CHANGELOG.md +243 -280
  2. package/dist/assets/prompts/reflect-feedback-framing.md +1 -1
  3. package/dist/assets/prompts/reflect-llm-framed-contract.md +2 -9
  4. package/dist/assets/prompts/reflect-llm-schema-contract.md +1 -3
  5. package/dist/assets/prompts/reflect-output-repair.md +1 -1
  6. package/dist/cli.js +1 -1
  7. package/dist/commands/improve/consolidate/pair-pass.js +1 -0
  8. package/dist/commands/improve/consolidate.js +7 -2
  9. package/dist/commands/improve/execution.js +2 -3
  10. package/dist/commands/improve/extract-cli.js +3 -2
  11. package/dist/commands/improve/extract.js +2 -1
  12. package/dist/commands/improve/improve-cli.js +33 -1
  13. package/dist/commands/improve/loop-stages.js +3 -0
  14. package/dist/commands/improve/reflect-noise.js +125 -0
  15. package/dist/commands/improve/reflect.js +150 -333
  16. package/dist/commands/improve/retrieval-gate.js +7 -2
  17. package/dist/commands/improve/session-asset.js +6 -0
  18. package/dist/commands/improve/stage.js +31 -39
  19. package/dist/commands/proposal/drain.js +4 -7
  20. package/dist/commands/proposal/propose.js +2 -11
  21. package/dist/commands/proposal/validators/proposal-quality-validators.js +11 -5
  22. package/dist/commands/proposal/validators/proposal-validators.js +4 -5
  23. package/dist/commands/read/search-cli.js +0 -38
  24. package/dist/core/asset/asset-serialize.js +1 -1
  25. package/dist/core/config/schema/engines.js +15 -33
  26. package/dist/core/config/schema/improve-processes.js +16 -0
  27. package/dist/core/content-safety.js +0 -24
  28. package/dist/core/redaction.js +4 -0
  29. package/dist/core/spawn-env.js +25 -0
  30. package/dist/core/structured.js +1 -1
  31. package/dist/execution/source.js +8 -12
  32. package/dist/integrations/agent/config.js +1 -3
  33. package/dist/integrations/agent/engine-resolution.js +0 -3
  34. package/dist/integrations/agent/execution.js +14 -13
  35. package/dist/integrations/agent/index.js +1 -1
  36. package/dist/integrations/agent/model-map.js +15 -16
  37. package/dist/integrations/agent/profiles.js +2 -2
  38. package/dist/integrations/agent/prompts.js +51 -127
  39. package/dist/integrations/agent/request-lowering.js +9 -7
  40. package/dist/integrations/agent/runner-dispatch.js +25 -31
  41. package/dist/integrations/harnesses/aider/agent-builder.js +1 -2
  42. package/dist/integrations/harnesses/amazonq/agent-builder.js +1 -2
  43. package/dist/integrations/harnesses/claude/agent-builder.js +4 -16
  44. package/dist/integrations/harnesses/codex/agent-builder.js +1 -2
  45. package/dist/integrations/harnesses/codex/index.js +6 -11
  46. package/dist/integrations/harnesses/codex/session-log.js +211 -0
  47. package/dist/integrations/harnesses/ids.js +10 -16
  48. package/dist/integrations/harnesses/opencode/agent-builder.js +14 -24
  49. package/dist/integrations/harnesses/opencode/model-config.js +15 -62
  50. package/dist/integrations/harnesses/opencode/model-work-agent.js +71 -36
  51. package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -6
  52. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +32 -90
  53. package/dist/integrations/harnesses/openhands/agent-builder.js +1 -2
  54. package/dist/integrations/harnesses/pi/agent-builder.js +1 -2
  55. package/dist/integrations/harnesses/types.js +3 -3
  56. package/dist/llm/feature-gate.js +2 -5
  57. package/dist/llm/index-passes.js +2 -2
  58. package/dist/llm/structured-call.js +5 -5
  59. package/dist/output/shapes/passthrough.js +1 -0
  60. package/dist/scripts/akm-migrate-node.js +381 -239
  61. package/dist/scripts/akm-migrate.js +381 -239
  62. package/dist/workflows/exec/unit-dispatch.js +4 -13
  63. package/docs/reference/cli.md +29 -16
  64. package/docs/reference/configuration.md +79 -85
  65. package/docs/reference/data-and-telemetry.md +2 -3
  66. package/docs/reference/workflow-schema.md +6 -9
  67. package/package.json +1 -1
  68. package/schemas/akm-config.json +108 -36
@@ -5,8 +5,7 @@ import { ConfigError } from "../../core/errors.js";
5
5
  import { assertFrozenDirectoryContained } from "../../execution/directory-identity.js";
6
6
  import { canonicalResolvedExecutionRequest } from "../../execution/resolved-request.js";
7
7
  import { buildExecutionFromWire } from "../../integrations/agent/execution.js";
8
- import { runExecution } from "../../integrations/agent/runner-dispatch.js";
9
- import { getHarness } from "../../integrations/harnesses/index.js";
8
+ import { runExecution, unwrapHarnessReply } from "../../integrations/agent/runner-dispatch.js";
10
9
  /** Lower a frozen common request through its frozen runner only. */
11
10
  export function prepareWorkflowExecution(request, prompt = request.prompt) {
12
11
  const target = request.frozenTarget;
@@ -107,17 +106,9 @@ export async function dispatchWorkflowExecution(request, feedback) {
107
106
  ...(notices ? { notices } : {}),
108
107
  };
109
108
  }
110
- let text = result.stdout;
111
- let sessionId = result.sessionId;
112
- if (lowered.runner.kind === "agent" && result.ok) {
113
- const harness = getHarness(lowered.runner.profile.platform ?? lowered.runner.profile.name);
114
- if (harness?.resultExtractor) {
115
- const extraction = harness.resultExtractor(result);
116
- text = extraction.text;
117
- if (extraction.sessionId !== undefined)
118
- sessionId = extraction.sessionId;
119
- }
120
- }
109
+ const extraction = result.ok ? unwrapHarnessReply(lowered.runner, result) : undefined;
110
+ const text = extraction?.text ?? result.stdout;
111
+ const sessionId = extraction?.sessionId ?? result.sessionId;
121
112
  return {
122
113
  ok: result.ok,
123
114
  text,
@@ -517,7 +517,6 @@ kept.
517
517
  | `--filter` | `<key>=<value>` | _(none)_ | Scope filter — repeatable. Valid keys: `user`, `agent`, `run`, `channel`. Example: `--filter user=alice --filter channel=ops`. Narrows the result set; ranking is unchanged. |
518
518
  | `--include-proposed` | flag | `false` | Include entries with `quality: "proposed"` in the result set. Default search excludes them; `generated` and `curated` quality entries are always included. Unknown quality values warn once and remain searchable. |
519
519
  | `--belief` | `all`, `current`, `historical` | `all` | Memory belief filter. `current` keeps active memory beliefs; `historical` keeps contradicted/superseded/archived ones. |
520
- | `--track-usage`, `--no-track-usage` | flag | `true` | Record or suppress local usage events for this successful read |
521
520
  | `--include-sessions` | flag | `false` | Include session assets, which are excluded from default results via `config.search.defaultExcludeTypes` |
522
521
  | `--format` | `json`, `jsonl`, `yaml`, `text`, `md`, `html` | `json` | Output format |
523
522
  | `--detail` | `brief`, `normal`, `full` | `brief` | Output verbosity level |
@@ -593,7 +592,6 @@ akm curate "learn the release workflow" --from all --format text
593
592
  | `--type` | `skill`, `command`, `agent`, `knowledge`, `instruction`, `workflow`, `script`, `memory`, `env`, `secret`, `lesson`, `task`, `session`, `fact`, `any` | `any` | Filter curated results by asset type |
594
593
  | `--limit` | number | `4` | Maximum curated results |
595
594
  | `--from` | `local`, `registry`, `all` | `local` | Where to search before curating |
596
- | `--track-usage`, `--no-track-usage` | flag | `true` | Record or suppress local usage events for this successful read |
597
595
 
598
596
  `akm curate` takes the top `--limit` hits of one search, in search order, and
599
597
  enriches each with a preview, run details and up to two support refs: the
@@ -623,7 +621,6 @@ curated like any other.
623
621
  every prompt: it only ever reads the index as it currently stands (the same
624
622
  non-blocking `ensureIndex()` path `search` uses) and never waits on or
625
623
  contends with a full `akm index` rebuild in progress.
626
- Use `--no-track-usage` when this inspection must not record usage events.
627
624
 
628
625
  ### show
629
626
 
@@ -631,9 +628,6 @@ Display an asset by ref. On a markdown document `#fragment` selects one
631
628
  section by heading slug (falling back to case-insensitive heading text); an
632
629
  unmatched fragment lists the available slugs.
633
630
 
634
- Successful reads record local usage events by default; pass
635
- `--no-track-usage` to suppress them.
636
-
637
631
  ```sh
638
632
  akm show scripts/deploy.sh
639
633
  akm show skills/code-review
@@ -662,7 +656,6 @@ akm show memories/retro --filter user=alice --filter agent=claude
662
656
  | `--max-chars` | positive integer | `3200` for `lead` | Hard contextual content budget in characters; requires `--context lead` and is mutually exclusive with `--max-tokens`. |
663
657
  | `--max-tokens` | positive integer | _(none)_ | Approximate contextual budget using four characters per token; requires `--context lead` and is mutually exclusive with `--max-chars`. |
664
658
  | `--filter` | `<key>=<value>` | _(none)_ | Repeatable scope filter (`user`, `agent`, `run`, `channel`). |
665
- | `--track-usage`, `--no-track-usage` | flag | `true` | Record or suppress local usage events for this successful read. |
666
659
 
667
660
  `meta` is not an asset type — `[<origin>//]meta[:<name>]` direct-reads a
668
661
  human-authored orientation doc from a bundle's optional `.meta/` directory
@@ -2431,6 +2424,7 @@ akm improve lessons/my-lesson --show-prompt --format text # print the composed r
2431
2424
  akm improve report # LLM usage/routing report for the most recent real run
2432
2425
  akm improve report --run <id> # ...for one specific improve_runs id
2433
2426
  akm improve report --since 7d # ...aggregated over every real run started in the last 7 days
2427
+ akm improve judge < revision.json # reflect's quality judge on one revision; writes nothing
2434
2428
  ```
2435
2429
 
2436
2430
  | Flag | Description |
@@ -2606,16 +2600,16 @@ default probe-on behavior) to check whether a named engine actually answers.
2606
2600
 
2607
2601
  `--show-prompt` (#952) is the cheapest way to exercise reflect alone: it
2608
2602
  builds the exact prompt reflect would send for one asset — the same source
2609
- resolution, runner selection, feedback/schema-hint/related-lesson/rejected-
2610
- proposal gathering `akm improve`'s live reflect step uses — and prints it
2603
+ resolution, runner selection, feedback/schema-hint/rejected-proposal
2604
+ gathering `akm improve`'s live reflect step uses — and prints it
2611
2605
  without reading a credential, so it never calls an engine. An LLM engine
2612
2606
  receives the reply's JSON Schema as `response_format`, and an agent engine as
2613
2607
  an instruction that dispatch appends to this prompt. Add
2614
2608
  `--format text` (the default JSON/yaml envelope escapes the prompt into one
2615
2609
  line, which defeats a by-eye read) to confirm by eye that recent feedback is
2616
- framed as an unverified report to investigate (never a fact to insert
2617
- verbatim) and that the response contract tells the model never to emit the
2618
- truncation marker or any content from outside the shown asset.
2610
+ framed as a signal (never a fact to insert) and that the response contract asks
2611
+ only for `confidence` and a `frontmatterPatch` of `description`,
2612
+ `when_to_use` and `title`: akm keeps the body.
2619
2613
 
2620
2614
  When reinforced facts need promotion, `knowledge` is the higher-authority
2621
2615
  destination than `memory`.
@@ -2660,6 +2654,17 @@ that run's own `llm_usage` events instead of erroring, sets `noCalls` to `[]`
2660
2654
  `notes` entry saying so rather than fabricating precision the old row can't
2661
2655
  support.
2662
2656
 
2657
+ #### improve judge
2658
+
2659
+ `akm improve judge` runs reflect's quality judge on one revision and prints its
2660
+ verdict, for testing a judge engine on revisions whose right answer you know. It
2661
+ reads `{"source": "...", "candidate": "...", "feedback": "...", "ref": "..."}` JSON
2662
+ from stdin (`feedback` and `ref` are optional; `ref` names the revised asset,
2663
+ which a judge on an agent engine may read), judges with the engine the strategy's
2664
+ `processes.reflect.qualityGate.engine` names (`--strategy` picks the strategy),
2665
+ and prints `{ engine, pass, score, reason, criteria }` with the gate's prompt and
2666
+ pass rule. A `score` of `-1` means the judge gave no verdict. It writes nothing.
2667
+
2663
2668
  ### proposal
2664
2669
 
2665
2670
  Manage the proposal queue. The canonical grammar is `akm proposal <verb>`:
@@ -2687,21 +2692,23 @@ authenticates its root; it never falls back to an ambient write target.
2687
2692
  #### proposal extract
2688
2693
 
2689
2694
  Extract durable insights from native coding-agent session files (claude-code,
2690
- opencode) and queue them as proposals. This is the standalone entrypoint for
2691
- session extraction — it replaces the legacy session-checkpoint hook and runs
2692
- independently of the improve-stage extract toggle (see `improve` above).
2695
+ codex, opencode) and queue them as proposals. This is the standalone
2696
+ entrypoint for session extraction — it replaces the legacy session-checkpoint
2697
+ hook and runs independently of the improve-stage extract toggle (see `improve`
2698
+ above).
2693
2699
 
2694
2700
  ```sh
2695
2701
  akm proposal extract --type claude --session-id <id>
2696
2702
  akm proposal extract --type claude --since 24h
2697
2703
  akm proposal extract --type opencode --since 7d --dry-run
2704
+ akm proposal extract --type codex --since 24h
2698
2705
  akm proposal extract --auto # iterate every available harness
2699
2706
  akm proposal extract --type claude --location /custom/path --session-id <id>
2700
2707
  ```
2701
2708
 
2702
2709
  | Flag | Description |
2703
2710
  | --- | --- |
2704
- | `--type <harness>` | Harness name (`claude`, `opencode`). Required unless `--auto`. |
2711
+ | `--type <harness>` | Harness name (`claude`, `codex`, `opencode`). Required unless `--auto`. |
2705
2712
  | `--session-id <id>` | Process only this session ID. When absent, discover sessions via `--since`. |
2706
2713
  | `--location <path>` | Override the harness's default session-discovery location. |
2707
2714
  | `--since <cutoff>` | Discovery cutoff. ISO timestamp or duration (`24h`, `7d`, `30m`). Default `24h`. |
@@ -2719,6 +2726,12 @@ session-log location on the current machine — and returns an aggregated
2719
2726
  per-harness `results`); the run exits non-zero only when every harness
2720
2727
  failed.
2721
2728
 
2729
+ The `codex` harness reads Codex's rollout files under `$CODEX_HOME/sessions`
2730
+ (`~/.codex/sessions` by default; `--location` points at another rollout
2731
+ directory). It lists a person's sessions, `codex exec` runs included, and
2732
+ leaves out the rollouts Codex writes for subagents and its other internal agents:
2733
+ those are not sessions of their own.
2734
+
2722
2735
  There is no `akm proposal extract --watch`/`--debounce-ms` either (0.9.0:
2723
2736
  dropped — a foreground polling daemon in a one-shot CLI); the shipped
2724
2737
  `core/extract.yml` cron template (`akm proposal extract --auto` on a
@@ -126,10 +126,9 @@ value matching this JSON Schema (no prose, no code fences):` followed by the
126
126
  schema), plus the harness's own schema channel where it has one (codex
127
127
  `--output-schema`).
128
128
 
129
- An agent engine may set `bin`, `args`, `workspace`, `model`, and `timeoutMs`,
130
- and the inference fields its platform translates (see
131
- [Inference on an agent engine](#inference-on-an-agent-engine)).
132
- Only `platform: "opencode-sdk"` may set `llmEngine`; it names
129
+ An agent engine may set `bin`, `args`, `workspace`, `model`, and `timeoutMs`;
130
+ it takes no inference of its own (see
131
+ [Inference on an agent engine](#inference-on-an-agent-engine)). Only `platform: "opencode-sdk"` may set `llmEngine`; it names
133
132
  the LLM engine used as that SDK engine's fallback connection. With no
134
133
  `llmEngine`, an SDK engine has no fallback connection and opencode resolves
135
134
  provider, model and auth from its own configuration. `defaults.llmEngine` is
@@ -138,10 +137,10 @@ not a substitute.
138
137
  On an engine without `timeoutMs`, model work (an improve process, a quality or
139
138
  triage judge, or an index pass) stops after 600 seconds, whatever the engine's
140
139
  kind; other work on an agent engine runs until it finishes. An improve stage's
141
- reply that does not match the stage's JSON Schema gets one corrective retry
142
- before the stage reads it. Reflect holds its reply to its own contract the same
143
- way on every engine kind: a reply that is not the JSON object of reflect's
144
- schema gets one repair turn, and one still invalid fails with `parse_error`.
140
+ reply that the stage cannot read gets one corrective retry. Reflect holds its
141
+ reply to its own contract the same way on every engine kind: a reply that is not
142
+ the JSON object of reflect's schema gets one repair turn, and one still invalid
143
+ fails with `parse_error`.
145
144
 
146
145
  Executable assets may request tools, but the request is not authority. Configure
147
146
  the host-local `execution.allowedTools` list to define the ceiling; `"*"` is an
@@ -157,74 +156,28 @@ with `npm i -g opencode-ai` or opencode's own installer.
157
156
 
158
157
  ### Inference on an agent engine
159
158
 
160
- A request's inference (`temperature`, `maxTokens`, `contextLength`,
161
- `enableThinking`, `reasoningEffort`) reaches an agent engine's harness from
162
- every place it can come from: the engine's own settings, an improve process's
163
- `llm` overlay, a task, command or agent asset's `inference`, a workflow's
164
- `llm:`, and a `models.json` alias. The nearest layer wins, field by field, as
165
- for an LLM engine. With no setting anywhere akm sends nothing of its own, so
166
- the model's own default applies: a `reasoningEffort: "none"` that your opencode
167
- config sets on a model stays in force until a layer overrides it.
159
+ An agent engine sets no inference of its own. For `opencode`, set it on the
160
+ model in your own opencode config, which akm leaves as it is. akm carries
161
+ inference (`temperature`, `reasoningEffort`, `enableThinking`, `maxTokens`,
162
+ `contextLength`) into opencode only where it writes opencode's config itself:
168
163
 
169
- Each platform translates what it can carry:
164
+ - **Model work on `opencode` and `opencode-sdk`:** an improve process's `llm`
165
+ overlay becomes the options of the `akm-model-work` agent that runs the
166
+ dispatch, so opencode's own title call on the same model keeps its defaults.
167
+ - **An `opencode-sdk` engine's `llmEngine` fallback:** the model akm declares for
168
+ it under `akm-custom` carries the fallback's inference and the request's.
169
+ `maxTokens` and `contextLength` become `limit.output` and `limit.context`, and
170
+ only together: opencode refuses half a `limit`.
170
171
 
171
- | Platform | Translates | How |
172
- |---|---|---|
173
- | `claude` | `reasoningEffort` | `--effort <level>`, in the harness's own levels (`low`, `medium`, `high`, `xhigh`, `max` in Claude Code 2.1.283). The value is passed as given, so one Claude Code rejects fails the dispatch. |
174
- | `opencode`, `opencode-sdk` | `temperature`, `reasoningEffort`, `enableThinking`, and `maxTokens` with `contextLength` | Injected opencode config, below. |
175
- | every other platform | nothing | |
176
-
177
- An agent engine that sets a field its platform does not translate fails to
178
- load, with an error that names the platform and the fields it does translate.
179
- Inference that reaches the engine from an asset or a caller, and that the
180
- platform does not translate, is reported as an `untranslated-field` notice and
181
- dispatch continues, because the same asset may run on any engine.
182
-
183
- On `opencode` and `opencode-sdk`, inference goes into opencode's config for the
184
- model the dispatch names: the request's model, or the `--model` an `opencode`
185
- engine's `args` name. That must be a `provider/model`, except for an
186
- `opencode-sdk` engine with an `llmEngine` fallback, which routes the
187
- fallback's model through its own provider. Without a model to attach to the
188
- fields are reported as untranslated, except that model work's agent (below)
189
- carries its options whichever model opencode picks, so an `opencode-sdk` engine
190
- with no `model` and no `llmEngine` still gets them. The injected entry merges
191
- over your own opencode config for the same provider and model, so what the
192
- request does not set stays as you wrote it. A dispatch with no inference injects
193
- nothing. akm
194
- checked these shapes against opencode 1.18.25 and an OpenAI-compatible provider
195
- (`@ai-sdk/openai-compatible`); another provider package is given the same
196
- options and may ignore one.
197
-
198
- - `temperature` becomes `options.temperature`, and `reasoningEffort` becomes
199
- `options.reasoningEffort`. opencode reads the camelCase name: its
200
- `reasoning_effort` spelling is dropped, so a model option written that way in
201
- `opencode.jsonc` does nothing.
202
- - `enableThinking` becomes `options.chat_template_kwargs.enable_thinking` and
203
- `options.enable_thinking`, the two forms an LLM engine sends.
204
- - `maxTokens` becomes `limit.output` and `contextLength` becomes
205
- `limit.context`. Set them together. Without `limit.output` opencode asks for
206
- `max_tokens: 32000`, which a small-context server rejects, and a wrong
207
- `limit.context` lets opencode build requests longer than the server's window.
208
- opencode refuses a `limit` with only one of them, and a half would overwrite
209
- the other half of a limit you declared for the model, so akm declares `limit`
210
- only when it has both and reports a lone `maxTokens` or `contextLength` as
211
- untranslated.
212
- - For model work (below) the options go on the `akm-model-work` agent that runs
213
- the dispatch, so opencode's own call on the same model, a title for the
214
- session, keeps the model's defaults. Any other dispatch runs your own agent,
215
- whose name akm cannot rely on, so the options go on the model and the title
216
- call sees them too. `opencode-sdk` names its session, so it makes no title
217
- call.
218
- - `opencode-sdk` declares the model of its `llmEngine` fallback with the
219
- fallback's own inference, under the request's, field by field. Each distinct
220
- set of inference starts its own `opencode serve`, as a different model does.
172
+ Inference from an asset, a workflow's `llm:` or a `models.json` alias reaches an
173
+ LLM engine. On an agent engine it is reported as an `untranslated-field`
174
+ notice and dispatch continues; model work's agent carries `temperature`,
175
+ `reasoningEffort` and `enableThinking` without one.
221
176
 
222
177
  Reasoning effort has one word in a request, `reasoningEffort`. `effort`, as a
223
- `models.json` alias or an asset's `effort:` frontmatter spells it, is the same
224
- setting and is read as `reasoningEffort` wherever layers are merged, so the
225
- nearest layer wins whichever word it used. On an LLM engine an alias's or
226
- asset's `effort` is therefore sent as `reasoning_effort`; before, it was
227
- reported as untranslated.
178
+ `models.json` alias or an asset's `effort:` frontmatter spells it, is read as
179
+ `reasoningEffort` wherever layers are merged, so an LLM engine sends it as
180
+ `reasoning_effort`.
228
181
 
229
182
  ### Engines for unattended model work
230
183
 
@@ -248,7 +201,7 @@ cannot enforce:
248
201
  |---|---|
249
202
  | LLM | No tools. |
250
203
  | `claude` | Read and Edit inside the working directory, and Bash for `akm search` and `akm show` only. akm runs it with `--restricted`, so your user, project and local settings cannot widen that. |
251
- | `opencode`, `opencode-sdk` | Read and edit inside the working directory, through an injected `akm-model-work` agent. No bash, so no `akm search` or `akm show`, because opencode cannot stop a redirect such as `akm show x > file` from writing elsewhere. |
204
+ | `opencode`, `opencode-sdk` | Read, grep and glob in the working directory and the stash, edit in the working directory only, and `akm_search` and `akm_show`, through an injected `akm-model-work` agent. No bash, because opencode cannot stop a redirect such as `akm show x > file` from writing elsewhere. |
252
205
  | `codex`, `copilot`, `pi`, `gemini`, `aider`, `amazonq`, `openhands` | Cannot run model work: akm refuses the request before it starts. |
253
206
 
254
207
  **The one rule.** Every key model work reads its engine from must name an
@@ -268,12 +221,17 @@ key, the engine and its platform. Other engine keys, `defaults.engine` and
268
221
  **Further details:**
269
222
  - A model-work dispatch builds its own command, so the engine's `args` do not
270
223
  apply to it, except a `--model` they name, and neither does its `workspace`.
271
- - The temporary directory must not be inside a git repository, because
272
- opencode would treat the whole repository as its working directory. Point
273
- `TMPDIR` elsewhere if it is.
274
- - `llm` overrides that reach an agent engine are translated when its platform
275
- translates them (see [Inference on an agent engine](#inference-on-an-agent-engine))
276
- and reported as `untranslated-field` notices, not errors, when it does not.
224
+ - An improve process's `llm` overlay reaches the `akm-model-work` agent on
225
+ `opencode` and `opencode-sdk` (see
226
+ [Inference on an agent engine](#inference-on-an-agent-engine)); on any other
227
+ agent engine it is reported as `untranslated-field` notices, not errors.
228
+ - opencode model work may read the stash and write only its working directory.
229
+ `akm_search` and `akm_show` come from the akm-opencode plugin (0.9.21 or
230
+ later), which akm does not load: put it in your opencode config
231
+ (`"plugin": ["akm-opencode"]`). akm turns off the plugin's curation, learning
232
+ and write gate for these dispatches and keeps its state in akm's state
233
+ directory. The stash is protected from edits only while the temporary
234
+ directory is outside a git repository.
277
235
 
278
236
  ### Model-map files
279
237
 
@@ -320,9 +278,7 @@ partial Claude override above does. After overlay, every alias/engine entry
320
278
  must have a usable model. Unknown profile fields are rejected; JSON-safe
321
279
  fields inside `inference` are preserved for engine adapters to lower
322
280
  optimistically. An `inference.effort` is read as `reasoningEffort`
323
- (see [Inference on an agent engine](#inference-on-an-agent-engine)), so the
324
- starter's `reasoning` alias sets `--effort high` on `claude` and
325
- `options.reasoningEffort: "high"` on `opencode` and `opencode-sdk`.
281
+ (see [Inference on an agent engine](#inference-on-an-agent-engine)).
326
282
 
327
283
  A profile's `engine` field (0.9.15, #946) borrows a column's `model` (and, for
328
284
  an `llm`-kind engine, its inference defaults) from a configured
@@ -511,9 +467,7 @@ guidance. When enabled, engine selection is judgment → triage → strategy →
511
467
 
512
468
  `processes.reflect.qualityGate` and `processes.distill.qualityGate` control
513
469
  each process's LLM-as-judge quality gate. Each is on unless it sets
514
- `enabled: false`, and each follows only its own switch. A reflect revision
515
- that changes the body is never auto-accepted; when the judge passes it, it
516
- waits for review. With the gate off, it waits for review too. The judge is the
470
+ `enabled: false`, and each follows only its own switch. The judge is the
517
471
  process's own engine when that is an LLM engine, or the `defaults.llmEngine`
518
472
  engine when an agent generates. `engine`, `model`, `timeoutMs` and `llm` give the gate a judge of its own,
519
473
  resolved over the process's settings the way `triage.judgment` resolves over
@@ -549,6 +503,46 @@ judge's engine at the server directly.
549
503
  }
550
504
  ```
551
505
 
506
+ `processes.reflect.defectFilter` sets the wording of the checks reflect runs
507
+ before the judge. With the quality gate on or off, reflect refuses a revision
508
+ that adds placeholder text, talks about its own edit, or copies frontmatter into
509
+ its body: no proposal, and no judge call. Each rule counts only what the
510
+ revision adds to its source, so wording the asset already had and kept is not
511
+ held against it. Each of the three lists is optional. A list you set replaces
512
+ that rule's default list, and `[]` turns the rule off.
513
+
514
+ | List | Rule | Default |
515
+ |---|---|---|
516
+ | `placeholders` | `placeholder_added` | `please confirm`, `please verify`, `to be confirmed`, `to be determined`, `to be verified` |
517
+ | `metaCommentary` | `meta_commentary_added` | `feedback signal`, `feedback signals`, `feedback indicate`, `feedback indicates`, `feedback suggest`, `feedback suggests`, `feedback ask`, `feedback asks`, `feedback says`, `feedback report`, `feedback reports`, `feedback request`, `feedback requests`, `this revision`, `the source asset`, `the source note`, `the source memory`, `the original asset`, `the original note`, `the original memory`, `the original version of this`, `quality gate rejected`, `proposal rejected` |
518
+ | `frontmatterKeys` | `frontmatter_copied_into_body` | `sources`, `updated`, `inferenceProcessed`, `captureMode`, `beliefState`, `xrefs`, `contradictedBy`, `outcomeData`, `orderedActions`, `generated`, `verified`, `description`, `when_to_use`, `tags`, `searchHints`, `quality`, `salience`, `salienceInputs`, `lint_skip`, `type` |
519
+
520
+ `placeholders` and `metaCommentary` entries are plain phrases, not patterns:
521
+ whole words, in any case, with any run of whitespace between words.
522
+ `frontmatterKeys` entries are exact key names: a line outside a code fence that
523
+ starts with `key:` counts. The rule also refuses a `sources`, `xrefs` or
524
+ `contradictedBy` value copied into the body; `frontmatterKeys: []` turns that
525
+ off too. Every entry must be a non-empty string, or the config does not load.
526
+
527
+ ```jsonc
528
+ {
529
+ "improve": {
530
+ "strategies": {
531
+ "nightly": {
532
+ "processes": {
533
+ "reflect": {
534
+ "defectFilter": {
535
+ "placeholders": ["please confirm", "to be confirmed", "[draft]"],
536
+ "frontmatterKeys": []
537
+ }
538
+ }
539
+ }
540
+ }
541
+ }
542
+ }
543
+ }
544
+ ```
545
+
552
546
  No shipped strategy turns improve-stage session extraction on.
553
547
  `proactiveMaintenance` is on only in the `proactive-maintenance` preset; run
554
548
  `akm improve --strategy proactive-maintenance` to use that opt-in preset.
@@ -245,9 +245,8 @@ engagement, feedback signals, stable refs, and timestamps. It never leaves the
245
245
  machine unless you explicitly copy the database or send derived content to a
246
246
  configured endpoint.
247
247
 
248
- Successful `search`, `curate`, and `show` commands record usage by default.
249
- Pass `--no-track-usage` to any of those commands to leave local usage events
250
- unchanged.
248
+ Successful `search`, `curate`, and `show` commands always record usage. Machine
249
+ reads are stamped by source (below), so they never skew ranking or eval.
251
250
 
252
251
  Every runtime writer stamps provenance as `user`, `improve`, `task`, `audit`, or
253
252
  `unknown`. Direct interactive CLI traffic defaults to `user`; internal improve,
@@ -299,15 +299,12 @@ families) plus the orchestration keys:
299
299
  - `defaults` — run-level dispatch defaults (`engine`, `model`, `llm`,
300
300
  `timeout`, `on_error`), overridable per unit. `llm:` tuning reaches an engine
301
301
  of any kind, and a document-level `llm:` reaches EVERY step. An LLM engine
302
- sends it; an agent engine translates what its platform can carry
303
- (`temperature`, `reasoning_effort`, `enable_thinking`, and `max_tokens` with
304
- `context_length` on `opencode` and `opencode-sdk`; `reasoning_effort` on
305
- `claude`; see
306
- [Inference on an agent engine](configuration.md#inference-on-an-agent-engine))
307
- and reports the rest as an `untranslated-field` notice on that step. There is
308
- no per-step opt-out (`llm: {}` is a no-op and `llm: null` is a parse error),
309
- so in a mixed document put `llm:` on the `unit:` of each step it is meant for
310
- instead of in `defaults:`.
302
+ sends it; an agent engine reports it as an `untranslated-field` notice on that
303
+ step (see
304
+ [Inference on an agent engine](configuration.md#inference-on-an-agent-engine)).
305
+ There is no per-step opt-out (`llm: {}` is a no-op and `llm: null` is a parse
306
+ error), so in a mixed document put `llm:` on the `unit:` of each step it is
307
+ meant for instead of in `defaults:`.
311
308
  - `outputs` — name → `{ from, schema? }`, a run-level export projected from
312
309
  a step's own artifact (Markdown-only; see [Workflow
313
310
  outputs](#workflow-outputs) below).
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "akm-cli",
3
- "version": "0.9.25-alpha.1",
3
+ "version": "0.9.25-alpha.3",
4
4
  "type": "module",
5
5
  "description": "akm (Agent Knowledge Manager) — a portable, local-first capability library for AI agents. Discover, load, share, and improve reusable skills, scripts, workflows, and knowledge across any shell-capable coding agent, including Claude Code, OpenCode, and Cursor.",
6
6
  "keywords": [
@@ -146,24 +146,6 @@
146
146
  "type": "string",
147
147
  "maxLength": 63,
148
148
  "pattern": "^(?!akm-)[a-z][a-z0-9]*(?:-[a-z0-9]+)*$"
149
- },
150
- "temperature": {
151
- "type": "number"
152
- },
153
- "maxTokens": {
154
- "type": "integer",
155
- "exclusiveMinimum": 0
156
- },
157
- "contextLength": {
158
- "type": "integer",
159
- "exclusiveMinimum": 0
160
- },
161
- "enableThinking": {
162
- "type": "boolean"
163
- },
164
- "reasoningEffort": {
165
- "type": "string",
166
- "minLength": 1
167
149
  }
168
150
  },
169
151
  "required": [
@@ -837,6 +819,33 @@
837
819
  }
838
820
  },
839
821
  "additionalProperties": true
822
+ },
823
+ "defectFilter": {
824
+ "type": "object",
825
+ "properties": {
826
+ "placeholders": {
827
+ "type": "array",
828
+ "items": {
829
+ "type": "string",
830
+ "minLength": 1
831
+ }
832
+ },
833
+ "metaCommentary": {
834
+ "type": "array",
835
+ "items": {
836
+ "type": "string",
837
+ "minLength": 1
838
+ }
839
+ },
840
+ "frontmatterKeys": {
841
+ "type": "array",
842
+ "items": {
843
+ "type": "string",
844
+ "minLength": 1
845
+ }
846
+ }
847
+ },
848
+ "additionalProperties": true
840
849
  }
841
850
  },
842
851
  "additionalProperties": true
@@ -1816,24 +1825,6 @@
1816
1825
  "type": "string",
1817
1826
  "maxLength": 63,
1818
1827
  "pattern": "^(?!akm-)[a-z][a-z0-9]*(?:-[a-z0-9]+)*$"
1819
- },
1820
- "temperature": {
1821
- "type": "number"
1822
- },
1823
- "maxTokens": {
1824
- "type": "integer",
1825
- "exclusiveMinimum": 0
1826
- },
1827
- "contextLength": {
1828
- "type": "integer",
1829
- "exclusiveMinimum": 0
1830
- },
1831
- "enableThinking": {
1832
- "type": "boolean"
1833
- },
1834
- "reasoningEffort": {
1835
- "type": "string",
1836
- "minLength": 1
1837
1828
  }
1838
1829
  },
1839
1830
  "required": [
@@ -2507,6 +2498,33 @@
2507
2498
  }
2508
2499
  },
2509
2500
  "additionalProperties": true
2501
+ },
2502
+ "defectFilter": {
2503
+ "type": "object",
2504
+ "properties": {
2505
+ "placeholders": {
2506
+ "type": "array",
2507
+ "items": {
2508
+ "type": "string",
2509
+ "minLength": 1
2510
+ }
2511
+ },
2512
+ "metaCommentary": {
2513
+ "type": "array",
2514
+ "items": {
2515
+ "type": "string",
2516
+ "minLength": 1
2517
+ }
2518
+ },
2519
+ "frontmatterKeys": {
2520
+ "type": "array",
2521
+ "items": {
2522
+ "type": "string",
2523
+ "minLength": 1
2524
+ }
2525
+ }
2526
+ },
2527
+ "additionalProperties": true
2510
2528
  }
2511
2529
  },
2512
2530
  "additionalProperties": true
@@ -3620,6 +3638,33 @@
3620
3638
  },
3621
3639
  "additionalProperties": true
3622
3640
  },
3641
+ "defectFilter": {
3642
+ "type": "object",
3643
+ "properties": {
3644
+ "placeholders": {
3645
+ "type": "array",
3646
+ "items": {
3647
+ "type": "string",
3648
+ "minLength": 1
3649
+ }
3650
+ },
3651
+ "metaCommentary": {
3652
+ "type": "array",
3653
+ "items": {
3654
+ "type": "string",
3655
+ "minLength": 1
3656
+ }
3657
+ },
3658
+ "frontmatterKeys": {
3659
+ "type": "array",
3660
+ "items": {
3661
+ "type": "string",
3662
+ "minLength": 1
3663
+ }
3664
+ }
3665
+ },
3666
+ "additionalProperties": true
3667
+ },
3623
3668
  "requirePlannedRefs": {
3624
3669
  "type": "boolean"
3625
3670
  },
@@ -4009,6 +4054,33 @@
4009
4054
  }
4010
4055
  },
4011
4056
  "additionalProperties": true
4057
+ },
4058
+ "defectFilter": {
4059
+ "type": "object",
4060
+ "properties": {
4061
+ "placeholders": {
4062
+ "type": "array",
4063
+ "items": {
4064
+ "type": "string",
4065
+ "minLength": 1
4066
+ }
4067
+ },
4068
+ "metaCommentary": {
4069
+ "type": "array",
4070
+ "items": {
4071
+ "type": "string",
4072
+ "minLength": 1
4073
+ }
4074
+ },
4075
+ "frontmatterKeys": {
4076
+ "type": "array",
4077
+ "items": {
4078
+ "type": "string",
4079
+ "minLength": 1
4080
+ }
4081
+ }
4082
+ },
4083
+ "additionalProperties": true
4012
4084
  }
4013
4085
  },
4014
4086
  "additionalProperties": true