akm-cli 0.9.24 → 0.9.25-alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/CHANGELOG.md +173 -0
  2. package/dist/cli.js +1 -1
  3. package/dist/commands/health/checks.js +10 -11
  4. package/dist/commands/improve/consolidate/pair-pass.js +4 -2
  5. package/dist/commands/improve/consolidate.js +10 -4
  6. package/dist/commands/improve/execution.js +4 -11
  7. package/dist/commands/improve/extract-prompt.js +4 -4
  8. package/dist/commands/improve/extract.js +11 -13
  9. package/dist/commands/improve/improve-cli.js +65 -34
  10. package/dist/commands/improve/improve-strategies.js +49 -43
  11. package/dist/commands/improve/improve-usage-report.js +8 -17
  12. package/dist/commands/improve/loop-stages.js +3 -0
  13. package/dist/commands/improve/preparation.js +3 -1
  14. package/dist/commands/improve/reflect-noise.js +125 -0
  15. package/dist/commands/improve/reflect.js +105 -172
  16. package/dist/commands/improve/retrieval-gate.js +7 -2
  17. package/dist/commands/improve/stage.js +67 -24
  18. package/dist/commands/proposal/drain.js +11 -2
  19. package/dist/commands/proposal/proposal-cli.js +1 -5
  20. package/dist/commands/proposal/propose-cli.js +2 -2
  21. package/dist/commands/proposal/propose.js +72 -84
  22. package/dist/commands/proposal/validators/proposal-quality-validators.js +4 -2
  23. package/dist/commands/read/search-cli.js +0 -38
  24. package/dist/commands/remember.js +3 -3
  25. package/dist/commands/sources/schema-repair.js +1 -1
  26. package/dist/core/config/config-schema.js +37 -60
  27. package/dist/core/config/engine-semantics.js +15 -11
  28. package/dist/core/config/schema/improve-processes.js +18 -2
  29. package/dist/core/improve-result.js +3 -3
  30. package/dist/core/redaction.js +4 -0
  31. package/dist/core/spawn-env.js +25 -0
  32. package/dist/core/structured.js +11 -1
  33. package/dist/execution/source.js +10 -0
  34. package/dist/indexer/passes/memory-inference.js +2 -1
  35. package/dist/integrations/agent/builder-shared.js +15 -0
  36. package/dist/integrations/agent/config.js +1 -1
  37. package/dist/integrations/agent/engine-resolution.js +13 -31
  38. package/dist/integrations/agent/execution.js +48 -22
  39. package/dist/integrations/agent/index.js +1 -1
  40. package/dist/integrations/agent/profiles.js +2 -2
  41. package/dist/integrations/agent/prompts.js +55 -114
  42. package/dist/integrations/agent/request-lowering.js +21 -8
  43. package/dist/integrations/agent/runner-dispatch.js +96 -3
  44. package/dist/integrations/agent/runner.js +8 -2
  45. package/dist/integrations/harnesses/aider/agent-builder.js +10 -23
  46. package/dist/integrations/harnesses/aider/index.js +0 -5
  47. package/dist/integrations/harnesses/amazonq/agent-builder.js +9 -21
  48. package/dist/integrations/harnesses/amazonq/index.js +0 -5
  49. package/dist/integrations/harnesses/claude/agent-builder.js +47 -36
  50. package/dist/integrations/harnesses/claude/index.js +0 -14
  51. package/dist/integrations/harnesses/codex/agent-builder.js +3 -4
  52. package/dist/integrations/harnesses/codex/index.js +0 -4
  53. package/dist/integrations/harnesses/copilot/agent-builder.js +4 -13
  54. package/dist/integrations/harnesses/copilot/index.js +2 -7
  55. package/dist/integrations/harnesses/gemini/agent-builder.js +4 -13
  56. package/dist/integrations/harnesses/gemini/index.js +0 -5
  57. package/dist/integrations/harnesses/ids.js +12 -10
  58. package/dist/integrations/harnesses/opencode/agent-builder.js +41 -9
  59. package/dist/integrations/harnesses/opencode/index.js +0 -8
  60. package/dist/integrations/harnesses/opencode/model-config.js +33 -0
  61. package/dist/integrations/harnesses/opencode/model-work-agent.js +115 -0
  62. package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -7
  63. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +114 -28
  64. package/dist/integrations/harnesses/openhands/agent-builder.js +7 -21
  65. package/dist/integrations/harnesses/openhands/index.js +0 -5
  66. package/dist/integrations/harnesses/pi/agent-builder.js +7 -21
  67. package/dist/integrations/harnesses/pi/index.js +0 -5
  68. package/dist/llm/client.js +5 -0
  69. package/dist/llm/feature-gate.js +12 -9
  70. package/dist/llm/index-passes.js +2 -5
  71. package/dist/llm/memory-infer.js +6 -5
  72. package/dist/llm/structured-call.js +33 -11
  73. package/dist/output/shapes/passthrough.js +1 -0
  74. package/dist/scripts/akm-migrate-node.js +298 -228
  75. package/dist/scripts/akm-migrate.js +298 -228
  76. package/dist/workflows/exec/step-work.js +6 -5
  77. package/dist/workflows/exec/unit-dispatch.js +4 -13
  78. package/dist/workflows/freeze/step-values.js +1 -1
  79. package/docs/reference/cli.md +41 -18
  80. package/docs/reference/configuration.md +165 -12
  81. package/docs/reference/data-and-telemetry.md +2 -3
  82. package/docs/reference/workflow-schema.md +10 -9
  83. package/package.json +1 -1
  84. package/schemas/akm-config.json +108 -0
package/CHANGELOG.md CHANGED
@@ -4,6 +4,179 @@ All notable changes to this project will be documented in this file.
4
4
 
5
5
  The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
6
6
 
7
+ ## [Unreleased]
8
+
9
+ ## [0.9.25-alpha.2] - 2026-10-03
10
+
11
+ ### Added
12
+
13
+ - **`akm improve judge`** runs reflect's quality judge on one revision, read as
14
+ `{"source", "candidate", "feedback", "ref"}` JSON from stdin, with the engine
15
+ `processes.reflect.qualityGate.engine` names, and prints the verdict. It
16
+ writes nothing: it tests a judge engine on revisions whose right answer you
17
+ know.
18
+
19
+ ### Changed
20
+
21
+ - **Reflect refuses three kinds of defective revision before the judge runs,**
22
+ with the quality gate on or off: one that adds placeholder text ("please
23
+ confirm", "to be confirmed"; not `TODO`, `TBD` or `FIXME`), one that talks about
24
+ its own edit ("the feedback says", "this revision", "the source asset", a
25
+ quoted gate rejection), and one that copies frontmatter into its body (key
26
+ lines such as `sources:` or `updated:` outside code, or a `sources`,
27
+ `xrefs` or `contradictedBy` value). Each rule counts only what the revision
28
+ adds to its source. A hit is a `quality_rejected` refusal with no proposal
29
+ and no judge call, and the event names the rule (`reflectDefect`). Each
30
+ rule's wording is a list that `processes.reflect.defectFilter` can replace:
31
+ `placeholders` and `metaCommentary` are plain phrases, matched as whole words
32
+ in any case, and `frontmatterKeys` are exact key names. A list left out keeps
33
+ its default, and `[]` turns its rule off. On 396 labelled reflect edits the
34
+ default lists hit 22 of 313 bad edits and none of 83 good ones.
35
+ - **The reflect quality judge's rubric names what it kept missing.** Tuned on
36
+ the labelled judge-gate set (plain judge, qwen3.8-27b): a description with a
37
+ sentence split by a stray period or an unbalanced quote is broken text; a
38
+ missing title, description or `when_to_use` is a concrete problem whether or
39
+ not the feedback mentions it, while a bare `type:` or a provenance stamp is
40
+ not; PRESERVATION and QUALITY are judged line by line in the changed region,
41
+ so a fix elsewhere no longer excuses a dropped fact, an invented or
42
+ strengthened claim, a hedge or a placeholder. On a stratified 100-case sample
43
+ the gate passed 31 of 40 good edits instead of 18 and 3 of 60 bad instead of 2.
44
+ - **A reflect quality judge on an agent engine reads only to verify what a
45
+ revision adds.** Its prompt names the revised asset's ref and adds one
46
+ paragraph: read that asset, or one the changed region names, with `akm_show`
47
+ only to check a fact the revision adds or alters, at most twice, and never
48
+ search; before scoring, find each added statement in the asset or the
49
+ feedback (a step or cause that merely seems to follow does not count), each
50
+ source fact in the revision, and each feedback point in a change to the text
51
+ it is about. The plain judge's prompt is unchanged. On the same 100 cases
52
+ (qwen3.8-27b, thinking on) the agent judge passed 38 of 40 good edits and 14
53
+ of 60 bad, against 34 and 12 for the plain judge on that model. Give the
54
+ engine's `llmEngine` `enableThinking: true`: without thinking the model
55
+ looped on tool calls.
56
+ - **A failed reflect reply reads the same on every engine kind:** the parser's
57
+ own message. 0.9.25-alpha.1 named the engine on an agent engine.
58
+ - **Inference reaches opencode only where akm writes opencode's config:** an
59
+ improve process's `llm` overlay on model work's agent, and an `opencode-sdk`
60
+ engine's `llmEngine` fallback model. Set the rest in your opencode config; a
61
+ task's or workflow's `inference` is an `untranslated-field` notice, as before
62
+ 0.9.25-alpha.1. `claude` takes no `--effort`.
63
+ - **An `opencode-sdk` session the dispatch times out on or aborts is aborted on
64
+ the server for every dispatch,** not only model work.
65
+ - **An improve stage retries a reply only when the stage cannot read it,** not
66
+ when it misses the JSON Schema. A reply it can read costs no second call.
67
+ - **opencode model work can read the stash and search it.** The `akm-model-work`
68
+ agent reads, greps and globs in the stash and its working directory, edits
69
+ only in the working directory, and has `akm_search` and `akm_show` from the
70
+ akm-opencode plugin (0.9.21 or later, in your own opencode config). The
71
+ plugin's curation, learning and write gate are off for these dispatches, and
72
+ its state goes to akm's state directory, not `~/.local/state/akm-opencode`.
73
+
74
+ ### Removed
75
+
76
+ - **The agent-engine inference fields,** which 0.9.25-alpha.1 accepted: they
77
+ fail to load again.
78
+ - **The `opencode-sdk` step watcher and the git-repository refusal for model
79
+ work's scratch directory.**
80
+ - **opencode model work's step limit.** At the limit opencode sends a "maximum
81
+ steps" message as a trailing assistant message, which a qwen chat template
82
+ (LM Studio, llama-server) renders as the start of the model's reply: LM
83
+ Studio returned nothing, llama-server returned that message as the answer,
84
+ and the dispatch failed. The dispatch timeout bounds a run.
85
+ - **A prompt builder that nothing called** (`buildSchemaRepairPrompt`).
86
+ - **The `--track-usage` / `--no-track-usage` flag on `akm search`, `akm curate`
87
+ and `akm show`.** A successful read always records its usage event, stamped
88
+ with its source (`user`, `improve`, `task` or `audit`); only `user` events
89
+ feed ranking and eval, so machine reads never skew them. Either spelling now
90
+ fails as an unknown flag.
91
+
92
+ ### Fixed
93
+
94
+ - **The reflect size guard no longer flags a body that does not grow.** The
95
+ expansion ceiling is capped at 25,000 characters, so a source body longer than
96
+ that was flagged `EXCESSIVE_EXPANSION` even when the proposed body was its own
97
+ length (ratio 1.00), and went to review instead of the judge. A body no longer
98
+ than its source is never expansion; one that grows past the cap still is, by
99
+ any amount. The reflect prompt agrees: it told the model its body could be at
100
+ most 25,000 characters even when the source was longer, and now gives such a
101
+ source's own length.
102
+ - **An asset's or a task's own `tools:` can no longer name the model-work
103
+ policy.** In 0.9.25-alpha.1 exactly `read`, `edit`, `akm search`, `akm show`,
104
+ in that order, skipped `execution.allowedTools`. It is ordinary tools now:
105
+ only akm's own model-work callers ask for the policy.
106
+ - **An agent engine that names its model only in `args` is named in the usage
107
+ report,** where it showed `unattributed`.
108
+ - **`opencode` and `opencode-sdk` engines receive the XDG base-directory
109
+ variables.** Under a custom `XDG_CONFIG_HOME` the spawned opencode missed its
110
+ provider config and failed every dispatch with `Unexpected server error`.
111
+
112
+ ## [0.9.25-alpha.1] - 2026-10-02
113
+
114
+ ### Changed
115
+
116
+ - **Unattended model work runs under one tool policy, on any engine that
117
+ confines it.** The model may read, edit inside a scratch directory akm removes
118
+ after the dispatch, and run `akm search` and `akm show`; the stash stays
119
+ read-only. An LLM engine has no tools, `claude` confines the policy, and
120
+ `opencode` and `opencode-sdk` run an injected `akm-model-work` agent with
121
+ read and edit only. Every other harness refuses it before the request starts.
122
+ - **One config rule names the engines model work may use.** Every key it reads
123
+ its engine from (`defaults.llmEngine`, `index.*.engine`, a strategy's or
124
+ process's `engine`, an enabled triage `judgment.engine`, a
125
+ `qualityGate.engine`) must name an LLM engine or a `claude`, `opencode` or
126
+ `opencode-sdk` agent engine, or the config fails to load. `--require-engines`
127
+ and `akm health` check agent engines too.
128
+ - **Model work is bounded at 600 seconds on every engine kind** whose engine
129
+ sets no `timeoutMs`; agent and `opencode-sdk` engines ran until they finished.
130
+ - **`akm proposal new` returns the proposal as JSON on every engine kind,**
131
+ with no draft file and no live session. A reply that is not a proposal gets
132
+ one retry.
133
+ - **Reflect asks every engine kind for the same JSON reply and repairs it
134
+ once.** An agent or `opencode-sdk` engine gets reflect's JSON Schema and a
135
+ repair turn, as an LLM engine did. An LLM engine's requests are unchanged.
136
+ - **An improve stage's reply that fails its JSON Schema gets one corrective
137
+ retry,** then the stage reads the last reply with its own parser.
138
+ - **One schema instruction for every agent engine,** appended by the shared
139
+ request lowering: `opencode` and `opencode-sdk` now receive a requested
140
+ schema, and a workflow unit on seven harnesses no longer carries it twice.
141
+ - **Agent and `opencode-sdk` dispatches leave a usage record.**
142
+ - **An `opencode-sdk` engine's LLM fallback comes only from its own
143
+ `llmEngine`; `defaults.llmEngine` no longer supplies one.** If you relied on
144
+ that, set `llmEngine` on the SDK engine. Without one, opencode uses its own
145
+ provider, model and auth.
146
+ - **Inference reaches `opencode`, `opencode-sdk` and `claude` engines.**
147
+ `temperature`, `maxTokens`, `contextLength`, `enableThinking` and
148
+ `reasoningEffort` from an engine, an improve process's `llm` overlay, an
149
+ asset, a workflow or a `models.json` alias were dropped on every agent
150
+ engine. An agent engine may set them, and `claude` takes `reasoningEffort` as
151
+ `--effort`.
152
+ - **Reasoning effort has one word, `reasoningEffort`.** `effort` in a
153
+ `models.json` alias or an asset is read as it, so an LLM engine now sends it
154
+ as `reasoning_effort`.
155
+
156
+ ### Removed
157
+
158
+ - **Dead code; nothing you run changes:** each harness's unused `pattern` and
159
+ `structuredOutput` fields, `resolveLlmEngineUse`'s swap of an agent engine
160
+ for an LLM engine, the agent request's unread `effort` hint, and the agent
161
+ file-write contract (`DRAFT_WRITTEN`, reflect's `ref_mismatch` check).
162
+
163
+ ### Fixed
164
+
165
+ - **An `opencode-sdk` engine with an LLM fallback reaches its endpoint, and a
166
+ failed dispatch is a failure (#1015).** An SDK error or provider rejection is
167
+ `ok: false` with opencode's message, and a reply's last text part is its answer.
168
+ - **An LLM engine reports a provider error sent with HTTP 200 as a failure**
169
+ (OpenRouter does this), not as an empty reply.
170
+ - **An `opencode` engine runs a persona** instead of failing on
171
+ `--system-prompt`, which opencode 1.18 rejects.
172
+ - **An LLM engine sends a requested schema** unless it sets
173
+ `supportsJsonSchema: false`; it sent one only when it set `true`.
174
+ - **A feature gate's timeout stops the call it bounds,** and a stage call
175
+ reports a timeout or abort as `timeout` or `aborted`, not `error`.
176
+ - **`akm proposal new` keeps the reply's `confidence`.**
177
+ - **Model work on an agent or `opencode-sdk` engine with no `timeoutMs` stops
178
+ after 600 seconds** as documented; it resolved to no timeout.
179
+
7
180
  ## [0.9.24] - 2026-10-02
8
181
 
9
182
  ### Changed
package/dist/cli.js CHANGED
@@ -177,7 +177,7 @@ const setupCommand = defineCommand({
177
177
  // the work, matching the `sync --push/--no-push` pattern. A flag
178
178
  // DECLARED as `no-init` can never be negated: `--no-init` parses as
179
179
  // "negate `init`", a name nothing declared, leaving the real key at its
180
- // default forever — see `search --no-track-usage`'s identical fix.
180
+ // default forever.
181
181
  init: {
182
182
  type: "boolean",
183
183
  default: true,
@@ -3,7 +3,6 @@
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  import { spawnSync } from "node:child_process";
5
5
  import { loadConfig } from "../../core/config/config.js";
6
- import { IMPROVE_PROCESS_ENGINE_CAPABILITIES } from "../../core/config/engine-semantics.js";
7
6
  import { listEnvsRecursive } from "../../core/env-secret-ref.js";
8
7
  import { ConfigError } from "../../core/errors.js";
9
8
  import { EXTRACT_INFRASTRUCTURE_SKIP_REASONS } from "../../core/improve-types.js";
@@ -15,7 +14,7 @@ import { loadModelMap, mergeModelMapLayers, parseModelMapLayer, readInstalledMod
15
14
  import { probeEndpointOnce } from "../../llm/client.js";
16
15
  import { STATE_DB_FREELIST_WARN_RATIO, } from "../../storage/state-db-integrity.js";
17
16
  import { listKeys } from "../env/env.js";
18
- import { resolveImprovePlan } from "../improve/improve-strategies.js";
17
+ import { MODEL_CALLING_PROCESSES, resolveImprovePlan } from "../improve/improve-strategies.js";
19
18
  import { ENGINE_LAST_USED_LOOKBACK_DAYS } from "./engine-usage.js";
20
19
  import { ACTIVE_RUN_WARN_MS, TASK_FAIL_RATE_WARN, } from "./types.js";
21
20
  /** Probe one connection's reachability, once per endpoint; `undefined` when no probe seam is supplied. */
@@ -142,7 +141,7 @@ async function runConfiguredEngineProbe(checkName, engineName, config, deps, rea
142
141
  evidence: { engine: engineName, runtimeKind: "sdk", binaryAvailable: false },
143
142
  };
144
143
  }
145
- const fallbackEngine = configuredEngine.llmEngine ?? config.defaults?.llmEngine;
144
+ const fallbackEngine = configuredEngine.llmEngine;
146
145
  let fallback;
147
146
  let fallbackCredential;
148
147
  let fallbackApiKeyFile;
@@ -523,17 +522,17 @@ export function probeActiveImproveStrategy(deps = {}) {
523
522
  .sort(([a], [b]) => a.localeCompare(b))
524
523
  .map(([process, engine]) => `${process}: "${engine}"`)
525
524
  .join(", ");
526
- // #957: fail only when the strategy's LLM-backed work would be a total
527
- // no-op — every process the strategy actually enabled among the
528
- // `capability: "llm"` set (see IMPROVE_PROCESS_ENGINE_CAPABILITIES) ended
529
- // up unavailable. A partial failure (some processes still have a working
525
+ // #957: fail only when the strategy's model work would be a total no-op —
526
+ // every process the strategy actually enabled among the ones that call a
527
+ // model themselves (MODEL_CALLING_PROCESSES, any engine kind) ended up
528
+ // unavailable. A partial failure (some processes still have a working
530
529
  // engine) stays a `warn`, matching #914's "a credential warn stays a warn"
531
530
  // policy for the general per-engine probes; this is the strategy-scoped
532
531
  // "is the whole run a no-op" question instead.
533
- const llmProcessNames = Object.keys(IMPROVE_PROCESS_ENGINE_CAPABILITIES).filter((name) => IMPROVE_PROCESS_ENGINE_CAPABILITIES[name] === "llm");
534
- const requiredLlmProcessNames = llmProcessNames.filter((name) => plan.processes[name].enabled || plan.engineUnavailable.some((item) => item.process === name));
535
- const availableLlmProcessNames = llmProcessNames.filter((name) => plan.processes[name].enabled);
536
- const allRequiredUnavailable = requiredLlmProcessNames.length > 0 && availableLlmProcessNames.length === 0;
532
+ const modelProcessNames = [...MODEL_CALLING_PROCESSES];
533
+ const requiredModelProcessNames = modelProcessNames.filter((name) => plan.processes[name].enabled || plan.engineUnavailable.some((item) => item.process === name));
534
+ const availableModelProcessNames = modelProcessNames.filter((name) => plan.processes[name].enabled);
535
+ const allRequiredUnavailable = requiredModelProcessNames.length > 0 && availableModelProcessNames.length === 0;
537
536
  const status = unavailableProcesses.length === 0 ? "pass" : allRequiredUnavailable ? "fail" : "warn";
538
537
  return {
539
538
  check: {
@@ -36,6 +36,7 @@ import { concurrentMap } from "../../../core/concurrent.js";
36
36
  import { parseEmbeddedJsonResponse } from "../../../core/parse.js";
37
37
  import { DERIVED_SUFFIX } from "../../../core/recognition-util.js";
38
38
  import { warnOnce } from "../../../core/warn.js";
39
+ import { runnerLlmConnection } from "../../../integrations/agent/runner.js";
39
40
  import { assertRunnerCredentials } from "../../../integrations/agent/runner-dispatch.js";
40
41
  import { runGit } from "../../../sources/providers/git-install.js";
41
42
  import { closeDatabase, openExistingDatabase, openReadonlyExistingDatabase, } from "../../../storage/repositories/index-connection.js";
@@ -501,10 +502,11 @@ async function judgeOne(ctx, candidate) {
501
502
  request: {
502
503
  responseSchema: PAIR_JUDGE_JSON_SCHEMA,
503
504
  enableThinking: false,
504
- timeoutMs: ctx.llmRunner.timeoutMs,
505
+ ...(Object.hasOwn(ctx.llmRunner, "timeoutMs") ? { timeoutMs: ctx.llmRunner.timeoutMs } : {}),
505
506
  signal: ctx.opts.signal,
506
507
  ...(ctx.chat ? { chat: ctx.chat } : {}),
507
508
  },
509
+ parse: parsePairJudgeResponse,
508
510
  ...(ctx.opts.onNotices ? { onNotices: ctx.opts.onNotices } : {}),
509
511
  });
510
512
  if (!outcome.ok)
@@ -747,7 +749,7 @@ seams = {}) {
747
749
  // entirely and needs no credential.
748
750
  if (!seams.chat)
749
751
  assertRunnerCredentials(llmRunner);
750
- const results = await concurrentMap(judgeable, (candidate) => judgeOne(ctx, candidate), llmRunner.connection.concurrency ?? 1, { signal: opts.signal });
752
+ const results = await concurrentMap(judgeable, (candidate) => judgeOne(ctx, candidate), runnerLlmConnection(llmRunner)?.concurrency ?? 1, { signal: opts.signal });
751
753
  results.forEach((r, idx) => {
752
754
  // Must-fix 3: `concurrentMap` leaves an entry `undefined` for a call an
753
755
  // aborted run never sent at all — `r?.failed === true` reads that as
@@ -36,6 +36,7 @@ import { resolveWriteTarget } from "../../core/write-source.js";
36
36
  import { deriveInstallations } from "../../indexer/installations.js";
37
37
  import { resolveSourceEntries } from "../../indexer/search/search-source.js";
38
38
  import { USAGE_EVENT_RETENTION_DAYS } from "../../indexer/usage/usage-events.js";
39
+ import { runnerLlmConnection } from "../../integrations/agent/runner.js";
39
40
  import { assertRunnerCredentials } from "../../integrations/agent/runner-dispatch.js";
40
41
  import { cosineSimilarity, embedBatch, resolveEmbeddingModelId } from "../../llm/embedder.js";
41
42
  import { getBodyEmbeddings, upsertBodyEmbeddings } from "../../storage/repositories/embeddings-repository.js";
@@ -52,6 +53,10 @@ import { resolveImproveStrategy, resolveProcessEnabled } from "./improve-strateg
52
53
  import { isContentDrivenRow, isLedgerBlocked, ledgerKey, loadLedgerSnapshot, recordLedgerAttempt } from "./ledger.js";
53
54
  import { isInRetrievalScope, loadRetrievalScope } from "./retrieval-scope.js";
54
55
  import { callStage, mintProposal, noticeSet, stageRunner } from "./stage.js";
56
+ function parsePlan(raw) {
57
+ const plan = parseEmbeddedJsonResponse(raw);
58
+ return plan && Array.isArray(plan.operations) ? { ...plan, operations: plan.operations } : undefined;
59
+ }
55
60
  /** A plan op worth acting on. Retired advisory ops (merge/delete/contradict) are dropped, never thrown on. */
56
61
  export function isValidOp(op) {
57
62
  if (typeof op !== "object" || op === null)
@@ -566,9 +571,10 @@ async function judgeConsolidationChunks(args) {
566
571
  request: {
567
572
  responseSchema: CONSOLIDATE_PLAN_JSON_SCHEMA,
568
573
  enableThinking: false,
569
- timeoutMs: llmRunner.timeoutMs,
574
+ ...(Object.hasOwn(llmRunner, "timeoutMs") ? { timeoutMs: llmRunner.timeoutMs } : {}),
570
575
  signal: opts.signal,
571
576
  },
577
+ parse: parsePlan,
572
578
  ...(opts.onNotices ? { onNotices: opts.onNotices } : {}),
573
579
  });
574
580
  if (!outcome.ok) {
@@ -576,8 +582,8 @@ async function judgeConsolidationChunks(args) {
576
582
  continue;
577
583
  }
578
584
  warnVerbose(`[akm:consolidate] ${label} raw response (first 500 chars): ${outcome.raw.slice(0, 500)}`);
579
- const parsed = parseEmbeddedJsonResponse(outcome.raw);
580
- if (!parsed || !Array.isArray(parsed.operations)) {
585
+ const parsed = parsePlan(outcome.raw);
586
+ if (!parsed) {
581
587
  const hint = outcome.raw.trim() === "" ? " (empty response — if using a thinking model, disable thinking mode)" : "";
582
588
  const msg = `Chunk ${chunkIdx + 1}: invalid plan from AI — skipping.${hint}`;
583
589
  warn(msg);
@@ -619,7 +625,7 @@ async function planConsolidation(opts, config, stashDir, memories, warnings, sta
619
625
  const llmRunner = opts.llmRunner ?? undefined;
620
626
  // 500 body chars per memory keep the judgement useful; chunk size varies instead.
621
627
  const bodyTruncation = 500;
622
- const chunkSize = computeSafeChunkSize(llmRunner?.connection.contextLength ?? DEFAULT_CONTEXT_LENGTH_TOKENS, bodyTruncation, opts.maxChunkSize);
628
+ const chunkSize = computeSafeChunkSize((llmRunner && runnerLlmConnection(llmRunner)?.contextLength) ?? DEFAULT_CONTEXT_LENGTH_TOKENS, bodyTruncation, opts.maxChunkSize);
623
629
  const sourceName = opts.target ?? stashDir;
624
630
  let budgeted = memories;
625
631
  const budgetMs = opts.signal?.remainingBudgetMs;
@@ -22,6 +22,8 @@ function mergeDefaults(farther, nearer) {
22
22
  /**
23
23
  * Resolve improve-owned model work through the canonical execution cascade:
24
24
  * defaults.llmEngine -> strategy -> index.<pass> -> process -> current invocation.
25
+ * The engine may be of any kind that confines the model-work tool policy: one
26
+ * that cannot, chosen with `--engine` say, is refused here, before any work.
25
27
  */
26
28
  export function resolveImproveExecution(options) {
27
29
  const defaultEngine = options.config.defaults?.llmEngine;
@@ -33,22 +35,13 @@ export function resolveImproveExecution(options) {
33
35
  if (selectedEngine === undefined || selectedEngine === null)
34
36
  return null;
35
37
  const invocationDefaults = mergeDefaults(mergeDefaults(defaultEngine ? { engine: defaultEngine } : {}, profileDefaults), indexDefaults);
36
- const current = mergeDefaults(processDefaults, currentDefaults);
37
38
  const prepared = resolveExecution({
38
39
  content: `improve ${options.processName} execution selection`,
39
40
  config: options.config,
40
41
  invocationDefaults,
41
- current,
42
+ current: mergeDefaults(processDefaults, currentDefaults),
43
+ modelWork: true,
42
44
  });
43
45
  const lowered = buildExecution(prepared.request, prepared.runner);
44
46
  return Object.freeze({ runner: lowered.runner, notices: lowered.notices });
45
47
  }
46
- export function resolveImproveLlmExecution(options) {
47
- const resolved = resolveImproveExecution(options);
48
- if (!resolved)
49
- return null;
50
- if (resolved.runner.kind !== "llm") {
51
- return null;
52
- }
53
- return { runner: resolved.runner, notices: resolved.notices };
54
- }
@@ -9,9 +9,9 @@
9
9
  * session data into the markdown template loaded from
10
10
  * `src/assets/prompts/extract-session.md`.
11
11
  *
12
- * The schema is intentionally strict — providers with `supportsJsonSchema:
13
- * true` enforce shape upstream, so the parser only has to handle the
14
- * happy path. `additionalProperties: false` means any hallucinated keys
12
+ * The schema is intentionally strict — a provider that honours
13
+ * `response_format` enforces shape upstream, so the parser only has to handle
14
+ * the happy path. `additionalProperties: false` means any hallucinated keys
15
15
  * the model emits get dropped before we parse.
16
16
  */
17
17
  import promptTemplate from "../../assets/prompts/extract-session.md" with { type: "text" };
@@ -20,7 +20,7 @@ const EXTRACT_CANDIDATE_NAME_PATTERN = "^[a-z0-9](?:[a-z0-9-]*[a-z0-9])?(?:/[a-z
20
20
  const EXTRACT_CANDIDATE_NAME_RE = new RegExp(EXTRACT_CANDIDATE_NAME_PATTERN);
21
21
  /**
22
22
  * JSON Schema for the structured extract output. Passed to `chatCompletion`
23
- * when the configured LLM connection has `supportsJsonSchema: true`.
23
+ * unless the configured LLM connection sets `supportsJsonSchema: false`.
24
24
  *
25
25
  * Shape:
26
26
  * {
@@ -32,16 +32,15 @@ import { indexWrittenAssets } from "../../indexer/index-written-assets.js";
32
32
  import { assertRunnerCredentials } from "../../integrations/agent/runner-dispatch.js";
33
33
  import { getAvailableHarnesses } from "../../integrations/session-logs/index.js";
34
34
  import { preFilterSession } from "../../integrations/session-logs/pre-filter.js";
35
- import { isJsonSchemaKnownUnsupported } from "../../llm/client.js";
36
35
  import { getExtractedSessionsMap, getLastExtractRunAt, shouldSkipAlreadyExtractedSession, upsertExtractedSession, } from "../../storage/repositories/extract-sessions-repository.js";
37
36
  import { openSqliteReadSnapshot } from "../../storage/sqlite-read-snapshot.js";
38
37
  import { contentHash } from "./content-hash.js";
39
- import { resolveImproveLlmExecution } from "./execution.js";
38
+ import { resolveImproveExecution } from "./execution.js";
40
39
  import { buildExtractPrompt, EXTRACT_JSON_SCHEMA, parseExtractPayload, } from "./extract-prompt.js";
41
40
  import { cloneAndFreeze, resolveImproveStrategy, resolveProcessEnabled } from "./improve-strategies.js";
42
41
  import { isLedgerBlocked, ledgerKey, loadLedgerSnapshot } from "./ledger.js";
43
42
  import { buildSessionSummaryPrompt, parseSessionSummary, SESSION_SUMMARY_JSON_SCHEMA, sessionMeetsDurationGate, writeSessionAsset, } from "./session-asset.js";
44
- import { callStage, mintProposal, noticeSet } from "./stage.js";
43
+ import { callStage, callStageOnce, mintProposal, noticeSet } from "./stage.js";
45
44
  /** Minimum session duration (minutes) for writing a session asset. */
46
45
  const DEFAULT_MIN_SESSION_DURATION_MINUTES = 5;
47
46
  /** Raw session size (chars) below which the LLM call is skipped; only truly empty sessions are safe to skip. */
@@ -119,7 +118,7 @@ export function resolveStandaloneExtractPlan(config, selection) {
119
118
  }
120
119
  const selected = resolveImproveStrategy(selection.strategy, config);
121
120
  const process = cloneAndFreeze(getImproveProcessConfig("extract", selected.config) ?? {});
122
- const resolved = resolveImproveLlmExecution({
121
+ const resolved = resolveImproveExecution({
123
122
  config,
124
123
  profile: selected.config,
125
124
  process,
@@ -130,7 +129,7 @@ export function resolveStandaloneExtractPlan(config, selection) {
130
129
  processName: "extract",
131
130
  });
132
131
  if (!resolved) {
133
- throw new ConfigError("No LLM engine configured for extract. Set defaults.llmEngine, pass --engine, or select an improve strategy with processes.extract.engine.", "LLM_NOT_CONFIGURED");
132
+ throw new ConfigError("No engine configured for extract. Set defaults.llmEngine, pass --engine, or select an improve strategy with processes.extract.engine.", "LLM_NOT_CONFIGURED");
134
133
  }
135
134
  const runner = resolved.runner;
136
135
  return Object.freeze({
@@ -358,15 +357,16 @@ function planExtractSessions(args) {
358
357
  }
359
358
  const EXTRACT_LLM_UNAVAILABLE = Symbol("extract-llm-unavailable");
360
359
  /**
361
- * One session's extraction call. A connection without structured output gets
362
- * one corrective retry; configuration errors escape before any state is written.
360
+ * One session's extraction call, with one corrective retry; configuration
361
+ * errors escape before any state is written.
363
362
  */
364
363
  async function extractFromSession(run, prompt) {
365
364
  const { llmRunner } = run;
366
365
  try {
367
366
  const result = await runStructured({
368
367
  dispatch: async (feedback) => {
369
- const outcome = await callStage({
368
+ // This loop parses and repairs the reply itself, so each attempt is one unvalidated dispatch.
369
+ const outcome = await callStageOnce({
370
370
  feature: "session_extraction",
371
371
  runner: llmRunner,
372
372
  prompt: feedback ? `${prompt}\n\n## Corrective output instruction\n\n${feedback}` : prompt,
@@ -388,9 +388,6 @@ async function extractFromSession(run, prompt) {
388
388
  return payload.parseFailure ? undefined : payload;
389
389
  },
390
390
  validate: (payload) => ({ ok: true, value: payload }),
391
- maxAttempts: llmRunner.connection.supportsJsonSchema !== false && !isJsonSchemaKnownUnsupported(llmRunner.connection)
392
- ? 1
393
- : 2,
394
391
  buildFeedback: () => "Your previous response did not contain a valid extraction payload. Respond with ONLY a JSON object matching the requested schema, with a candidates array and no prose or code fences.",
395
392
  });
396
393
  if (result.ok)
@@ -714,13 +711,13 @@ function resolveExtractRun(options, config, process, activeProfile) {
714
711
  llmRunner = options.llmRunner;
715
712
  }
716
713
  else {
717
- const resolved = resolveImproveLlmExecution({ config, profile: activeProfile, process, processName: "extract" });
714
+ const resolved = resolveImproveExecution({ config, profile: activeProfile, process, processName: "extract" });
718
715
  llmRunner = resolved?.runner;
719
716
  if (resolved)
720
717
  notices.add(resolved.notices);
721
718
  }
722
719
  if (!llmRunner) {
723
- throw new ConfigError("No LLM engine configured for extract. Set defaults.llmEngine or improve.strategies.<name>.processes.extract.engine.", "LLM_NOT_CONFIGURED");
720
+ throw new ConfigError("No engine configured for extract. Set defaults.llmEngine or improve.strategies.<name>.processes.extract.engine.", "LLM_NOT_CONFIGURED");
724
721
  }
725
722
  const runner = llmRunner;
726
723
  const timeoutMs = options.resolvedPlan
@@ -744,6 +741,7 @@ function resolveExtractRun(options, config, process, activeProfile) {
744
741
  ...(options.signal ? { signal: options.signal } : {}),
745
742
  ...(options.chat ? { chat: options.chat } : {}),
746
743
  },
744
+ parse: parseSessionSummary,
747
745
  onNotices: notices.add,
748
746
  });
749
747
  return parseSessionSummary(outcome.ok ? outcome.raw : "");
@@ -15,17 +15,20 @@ import { redactSensitiveText } from "../../core/redaction.js";
15
15
  import { clearLogFile, setLogFile, warn } from "../../core/warn.js";
16
16
  import { resolveWriteTarget } from "../../core/write-source.js";
17
17
  import { DEFAULT_LLM_TIMEOUT_MS } from "../../integrations/agent/config.js";
18
+ import { defaultWhich } from "../../integrations/agent/detect.js";
18
19
  import { collectEngineCredentialValues } from "../../integrations/agent/engine-resolution.js";
19
20
  import { probeLlmReachable } from "../../llm/client.js";
20
21
  import { getOutputMode } from "../../output/context.js";
21
22
  import { deliverRendered } from "../../output/html-render.js";
23
+ import { readStdin } from "../../runtime.js";
22
24
  import { akmImprove, IMPROVE_TARGET_FLAG, resolveImproveReadSource } from "./improve.js";
23
25
  import { runImproveReportQuery } from "./improve-report.js";
24
26
  import { buildImproveRunId, recordImproveRunResult, recordTerminatedImproveRun, } from "./improve-result-file.js";
25
27
  import { runImproveSession } from "./improve-session.js";
26
- import { resolveImprovePlan, } from "./improve-strategies.js";
28
+ import { resolveImprovePlan, resolveImproveStrategy, } from "./improve-strategies.js";
27
29
  import { formatUsageReportTable } from "./improve-usage-report.js";
28
30
  import { renderReflectPromptPreview } from "./reflect.js";
31
+ import { resolveQualityGateJudge, runReflectQualityJudge } from "./stage.js";
29
32
  let akmImproveForRun = akmImprove;
30
33
  /** Swap the CLI's improve work implementation in deterministic subprocess tests. */
31
34
  export function _setAkmImproveForTests(fake) {
@@ -73,26 +76,19 @@ function assertRequiredEnginesAvailable(plan) {
73
76
  const lines = plan.engineUnavailable.map((item) => ` - ${item.process} (${item.configKey}): ${item.reason}`);
74
77
  throw new ConfigError(`--require-engines: ${plan.engineUnavailable.length} improve process${plan.engineUnavailable.length === 1 ? "" : "es"} cannot run because ${plan.engineUnavailable.length === 1 ? "its" : "their"} engine is unavailable:\n${lines.join("\n")}`, "LLM_NOT_CONFIGURED");
75
78
  }
76
- /** Every LLM connection the plan would dispatch to, triage's judgment engine included. */
79
+ /** Every engine the plan would dispatch to, triage's judgment engine included. */
77
80
  function collectRequiredEngineTargets(plan) {
78
- const targets = [];
79
- for (const [processName, process] of Object.entries(plan.processes)) {
80
- if (process.runner) {
81
- targets.push({
82
- process: processName,
83
- engine: process.runner.engine,
84
- connection: probeConnection(process.runner),
85
- });
86
- }
87
- }
88
- if (plan.triageJudgment?.kind === "llm") {
89
- targets.push({
90
- process: "triage.judgment",
91
- engine: plan.triageJudgment.engine,
92
- connection: probeConnection(plan.triageJudgment),
93
- });
94
- }
95
- return targets;
81
+ const runners = Object.entries(plan.processes)
82
+ .flatMap(([processName, process]) => (process.runner ? [[processName, process.runner]] : []))
83
+ .concat(plan.triageJudgment ? [["triage.judgment", plan.triageJudgment]] : []);
84
+ return runners.map(([processName, runner]) => ({
85
+ process: processName,
86
+ engine: runner.engine,
87
+ ...(runner.kind === "llm" ? { connection: probeConnection(runner) } : { bin: runner.profile.bin }),
88
+ ...(runner.kind === "sdk" && runner.fallbackConnection
89
+ ? { connection: probeConnection({ connection: runner.fallbackConnection, timeoutMs: runner.fallbackTimeoutMs }) }
90
+ : {}),
91
+ }));
96
92
  }
97
93
  /** The resolved engine keeps its request timeout beside the connection (the runtime merges it in); the probe needs it on the connection. */
98
94
  function probeConnection(runner) {
@@ -112,37 +108,42 @@ const REQUIRED_ENGINE_PROBE_MAX_MS = 120_000;
112
108
  /**
113
109
  * `--require-engines`, live: probe each connection's real completion path
114
110
  * (a gateway can list a model whose completion route is dead, #980), once per
115
- * endpoint + model, within {@link requiredEngineProbeTimeoutMs}. Returns each
116
- * target's latency for the run result (R17); an unreachable one fails the run.
111
+ * endpoint + model, within {@link requiredEngineProbeTimeoutMs}, and look each
112
+ * agent harness's binary up on PATH. Returns each target's latency for the run
113
+ * result (R17); an unreachable one fails the run.
117
114
  */
118
- export async function assertRequiredEnginesReachable(plan, probeReachable = (connection) => probeLlmReachable(connection, requiredEngineProbeTimeoutMs(connection))) {
115
+ export async function assertRequiredEnginesReachable(plan, probeReachable = (connection) => probeLlmReachable(connection, requiredEngineProbeTimeoutMs(connection)), which = defaultWhich) {
119
116
  const targets = collectRequiredEngineTargets(plan);
120
117
  if (targets.length === 0)
121
118
  return [];
122
119
  const probesByConnection = new Map();
123
- const probed = await Promise.all(targets.map(async (target) => {
124
- const key = `${target.connection.endpoint.replace(/\/+$/, "")}|${target.connection.model}`;
120
+ const probeConnectionOnce = (connection) => {
121
+ const key = `${connection.endpoint.replace(/\/+$/, "")}|${connection.model}`;
125
122
  let pending = probesByConnection.get(key);
126
123
  if (!pending) {
127
124
  const probeStartedAt = Date.now();
128
- pending = probeReachable(target.connection).then((reach) => ({
129
- reach,
130
- latencyMs: Date.now() - probeStartedAt,
131
- }));
125
+ pending = probeReachable(connection).then((reach) => ({ reach, latencyMs: Date.now() - probeStartedAt }));
132
126
  probesByConnection.set(key, pending);
133
127
  }
134
- const { reach, latencyMs } = await pending;
135
- return { ...target, reach, latencyMs };
128
+ return pending;
129
+ };
130
+ const probed = await Promise.all(targets.map(async (target) => {
131
+ if (target.bin !== undefined && which(target.bin) === undefined) {
132
+ return { ...target, reach: { reachable: false, error: `${target.bin} is not on PATH` }, latencyMs: 0 };
133
+ }
134
+ if (!target.connection)
135
+ return { ...target, reach: { reachable: true }, latencyMs: 0 };
136
+ return { ...target, ...(await probeConnectionOnce(target.connection)) };
136
137
  }));
137
138
  const unreachable = probed.filter((item) => !item.reach.reachable);
138
139
  if (unreachable.length > 0) {
139
- const lines = unreachable.map((item) => ` - ${item.process} (engine "${item.engine}", ${item.connection.endpoint}): ${item.reach.error ?? "did not respond"}`);
140
- throw new ConfigError(`--require-engines: ${unreachable.length} improve process${unreachable.length === 1 ? "" : "es"} cannot run because ${unreachable.length === 1 ? "its" : "their"} engine completion path is not reachable:\n${lines.join("\n")}`, "LLM_NOT_CONFIGURED", "Check that each listed endpoint is up and serves its model. The probe is one short completion, bounded by the engine's timeoutMs (at most two minutes).");
140
+ const lines = unreachable.map((item) => ` - ${item.process} (engine "${item.engine}", ${item.connection?.endpoint ?? item.bin}): ${item.reach.error ?? "did not respond"}`);
141
+ throw new ConfigError(`--require-engines: ${unreachable.length} improve process${unreachable.length === 1 ? "" : "es"} cannot run because ${unreachable.length === 1 ? "its" : "their"} engine completion path is not reachable:\n${lines.join("\n")}`, "LLM_NOT_CONFIGURED", "Check that each listed endpoint is up and serves its model, and that each listed agent binary is installed. The endpoint probe is one short completion, bounded by the engine's timeoutMs (at most two minutes).");
141
142
  }
142
143
  return probed.map((item) => ({
143
144
  process: item.process,
144
145
  engine: item.engine,
145
- endpoint: item.connection.endpoint,
146
+ endpoint: item.connection?.endpoint ?? item.bin,
146
147
  reachable: item.reach.reachable,
147
148
  latencyMs: item.latencyMs,
148
149
  }));
@@ -189,6 +190,32 @@ function rejectReportOnlyFlags(args) {
189
190
  return;
190
191
  throw new UsageError(`\`${flag}\` only applies to \`akm improve report\`. Use \`akm improve report ${flag} <value>\` instead.`, "INVALID_FLAG_VALUE");
191
192
  }
193
+ /**
194
+ * `akm improve judge`: reflect's quality judge on one revision, read as
195
+ * `{"source", "candidate", "feedback"}` JSON from stdin, with the engine the
196
+ * strategy's reflect quality gate names. It writes nothing.
197
+ */
198
+ async function runImproveJudgeCli(strategyName) {
199
+ const input = process.stdin.isTTY
200
+ ? {}
201
+ : JSON.parse((await readStdin()).toString("utf8"));
202
+ const { source, candidate, feedback, ref } = input;
203
+ if (typeof source !== "string" || typeof candidate !== "string") {
204
+ throw new UsageError('`akm improve judge` reads {"source": "...", "candidate": "...", "feedback": "...", "ref": "..."} JSON from stdin.', "MISSING_REQUIRED_ARGUMENT");
205
+ }
206
+ const config = loadConfig();
207
+ const judge = resolveQualityGateJudge(config, resolveImproveStrategy(strategyName, config).config, "reflect");
208
+ if (!judge) {
209
+ throw new ConfigError("`akm improve judge` judges with the reflect quality gate's engine. Set processes.reflect.qualityGate.engine.", "INVALID_CONFIG_FILE");
210
+ }
211
+ const notes = typeof feedback === "string" && feedback.trim() !== "" ? [feedback.trim()] : [];
212
+ const verdict = await runReflectQualityJudge(config, candidate, source, notes, undefined, {
213
+ runnerSelectionFrozen: true,
214
+ llmRunner: judge,
215
+ ...(typeof ref === "string" && ref ? { ref } : {}),
216
+ });
217
+ output("improve-judge", { engine: judge.engine, ...verdict });
218
+ }
192
219
  export const improveCommand = defineCommand({
193
220
  meta: {
194
221
  name: "improve",
@@ -271,6 +298,10 @@ export const improveCommand = defineCommand({
271
298
  return;
272
299
  }
273
300
  rejectReportOnlyFlags(args);
301
+ if (getStringArg(args, "scope") === "judge") {
302
+ await runImproveJudgeCli(getStringArg(args, "strategy"));
303
+ return;
304
+ }
274
305
  rejectRetiredImproveTargetFlag();
275
306
  const jsonToStdout = args["json-to-stdout"];
276
307
  const targetArg = getStringArg(args, "bundle");