akm-cli 0.9.24 → 0.9.25-alpha.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +173 -0
- package/dist/cli.js +1 -1
- package/dist/commands/health/checks.js +10 -11
- package/dist/commands/improve/consolidate/pair-pass.js +4 -2
- package/dist/commands/improve/consolidate.js +10 -4
- package/dist/commands/improve/execution.js +4 -11
- package/dist/commands/improve/extract-prompt.js +4 -4
- package/dist/commands/improve/extract.js +11 -13
- package/dist/commands/improve/improve-cli.js +65 -34
- package/dist/commands/improve/improve-strategies.js +49 -43
- package/dist/commands/improve/improve-usage-report.js +8 -17
- package/dist/commands/improve/loop-stages.js +3 -0
- package/dist/commands/improve/preparation.js +3 -1
- package/dist/commands/improve/reflect-noise.js +125 -0
- package/dist/commands/improve/reflect.js +105 -172
- package/dist/commands/improve/retrieval-gate.js +7 -2
- package/dist/commands/improve/stage.js +67 -24
- package/dist/commands/proposal/drain.js +11 -2
- package/dist/commands/proposal/proposal-cli.js +1 -5
- package/dist/commands/proposal/propose-cli.js +2 -2
- package/dist/commands/proposal/propose.js +72 -84
- package/dist/commands/proposal/validators/proposal-quality-validators.js +4 -2
- package/dist/commands/read/search-cli.js +0 -38
- package/dist/commands/remember.js +3 -3
- package/dist/commands/sources/schema-repair.js +1 -1
- package/dist/core/config/config-schema.js +37 -60
- package/dist/core/config/engine-semantics.js +15 -11
- package/dist/core/config/schema/improve-processes.js +18 -2
- package/dist/core/improve-result.js +3 -3
- package/dist/core/redaction.js +4 -0
- package/dist/core/spawn-env.js +25 -0
- package/dist/core/structured.js +11 -1
- package/dist/execution/source.js +10 -0
- package/dist/indexer/passes/memory-inference.js +2 -1
- package/dist/integrations/agent/builder-shared.js +15 -0
- package/dist/integrations/agent/config.js +1 -1
- package/dist/integrations/agent/engine-resolution.js +13 -31
- package/dist/integrations/agent/execution.js +48 -22
- package/dist/integrations/agent/index.js +1 -1
- package/dist/integrations/agent/profiles.js +2 -2
- package/dist/integrations/agent/prompts.js +55 -114
- package/dist/integrations/agent/request-lowering.js +21 -8
- package/dist/integrations/agent/runner-dispatch.js +96 -3
- package/dist/integrations/agent/runner.js +8 -2
- package/dist/integrations/harnesses/aider/agent-builder.js +10 -23
- package/dist/integrations/harnesses/aider/index.js +0 -5
- package/dist/integrations/harnesses/amazonq/agent-builder.js +9 -21
- package/dist/integrations/harnesses/amazonq/index.js +0 -5
- package/dist/integrations/harnesses/claude/agent-builder.js +47 -36
- package/dist/integrations/harnesses/claude/index.js +0 -14
- package/dist/integrations/harnesses/codex/agent-builder.js +3 -4
- package/dist/integrations/harnesses/codex/index.js +0 -4
- package/dist/integrations/harnesses/copilot/agent-builder.js +4 -13
- package/dist/integrations/harnesses/copilot/index.js +2 -7
- package/dist/integrations/harnesses/gemini/agent-builder.js +4 -13
- package/dist/integrations/harnesses/gemini/index.js +0 -5
- package/dist/integrations/harnesses/ids.js +12 -10
- package/dist/integrations/harnesses/opencode/agent-builder.js +41 -9
- package/dist/integrations/harnesses/opencode/index.js +0 -8
- package/dist/integrations/harnesses/opencode/model-config.js +33 -0
- package/dist/integrations/harnesses/opencode/model-work-agent.js +115 -0
- package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -7
- package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +114 -28
- package/dist/integrations/harnesses/openhands/agent-builder.js +7 -21
- package/dist/integrations/harnesses/openhands/index.js +0 -5
- package/dist/integrations/harnesses/pi/agent-builder.js +7 -21
- package/dist/integrations/harnesses/pi/index.js +0 -5
- package/dist/llm/client.js +5 -0
- package/dist/llm/feature-gate.js +12 -9
- package/dist/llm/index-passes.js +2 -5
- package/dist/llm/memory-infer.js +6 -5
- package/dist/llm/structured-call.js +33 -11
- package/dist/output/shapes/passthrough.js +1 -0
- package/dist/scripts/akm-migrate-node.js +298 -228
- package/dist/scripts/akm-migrate.js +298 -228
- package/dist/workflows/exec/step-work.js +6 -5
- package/dist/workflows/exec/unit-dispatch.js +4 -13
- package/dist/workflows/freeze/step-values.js +1 -1
- package/docs/reference/cli.md +41 -18
- package/docs/reference/configuration.md +165 -12
- package/docs/reference/data-and-telemetry.md +2 -3
- package/docs/reference/workflow-schema.md +10 -9
- package/package.json +1 -1
- package/schemas/akm-config.json +108 -0
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,179 @@ All notable changes to this project will be documented in this file.
|
|
|
4
4
|
|
|
5
5
|
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
|
|
6
6
|
|
|
7
|
+
## [Unreleased]
|
|
8
|
+
|
|
9
|
+
## [0.9.25-alpha.2] - 2026-10-03
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- **`akm improve judge`** runs reflect's quality judge on one revision, read as
|
|
14
|
+
`{"source", "candidate", "feedback", "ref"}` JSON from stdin, with the engine
|
|
15
|
+
`processes.reflect.qualityGate.engine` names, and prints the verdict. It
|
|
16
|
+
writes nothing: it tests a judge engine on revisions whose right answer you
|
|
17
|
+
know.
|
|
18
|
+
|
|
19
|
+
### Changed
|
|
20
|
+
|
|
21
|
+
- **Reflect refuses three kinds of defective revision before the judge runs,**
|
|
22
|
+
with the quality gate on or off: one that adds placeholder text ("please
|
|
23
|
+
confirm", "to be confirmed"; not `TODO`, `TBD` or `FIXME`), one that talks about
|
|
24
|
+
its own edit ("the feedback says", "this revision", "the source asset", a
|
|
25
|
+
quoted gate rejection), and one that copies frontmatter into its body (key
|
|
26
|
+
lines such as `sources:` or `updated:` outside code, or a `sources`,
|
|
27
|
+
`xrefs` or `contradictedBy` value). Each rule counts only what the revision
|
|
28
|
+
adds to its source. A hit is a `quality_rejected` refusal with no proposal
|
|
29
|
+
and no judge call, and the event names the rule (`reflectDefect`). Each
|
|
30
|
+
rule's wording is a list that `processes.reflect.defectFilter` can replace:
|
|
31
|
+
`placeholders` and `metaCommentary` are plain phrases, matched as whole words
|
|
32
|
+
in any case, and `frontmatterKeys` are exact key names. A list left out keeps
|
|
33
|
+
its default, and `[]` turns its rule off. On 396 labelled reflect edits the
|
|
34
|
+
default lists hit 22 of 313 bad edits and none of 83 good ones.
|
|
35
|
+
- **The reflect quality judge's rubric names what it kept missing.** Tuned on
|
|
36
|
+
the labelled judge-gate set (plain judge, qwen3.8-27b): a description with a
|
|
37
|
+
sentence split by a stray period or an unbalanced quote is broken text; a
|
|
38
|
+
missing title, description or `when_to_use` is a concrete problem whether or
|
|
39
|
+
not the feedback mentions it, while a bare `type:` or a provenance stamp is
|
|
40
|
+
not; PRESERVATION and QUALITY are judged line by line in the changed region,
|
|
41
|
+
so a fix elsewhere no longer excuses a dropped fact, an invented or
|
|
42
|
+
strengthened claim, a hedge or a placeholder. On a stratified 100-case sample
|
|
43
|
+
the gate passed 31 of 40 good edits instead of 18 and 3 of 60 bad instead of 2.
|
|
44
|
+
- **A reflect quality judge on an agent engine reads only to verify what a
|
|
45
|
+
revision adds.** Its prompt names the revised asset's ref and adds one
|
|
46
|
+
paragraph: read that asset, or one the changed region names, with `akm_show`
|
|
47
|
+
only to check a fact the revision adds or alters, at most twice, and never
|
|
48
|
+
search; before scoring, find each added statement in the asset or the
|
|
49
|
+
feedback (a step or cause that merely seems to follow does not count), each
|
|
50
|
+
source fact in the revision, and each feedback point in a change to the text
|
|
51
|
+
it is about. The plain judge's prompt is unchanged. On the same 100 cases
|
|
52
|
+
(qwen3.8-27b, thinking on) the agent judge passed 38 of 40 good edits and 14
|
|
53
|
+
of 60 bad, against 34 and 12 for the plain judge on that model. Give the
|
|
54
|
+
engine's `llmEngine` `enableThinking: true`: without thinking the model
|
|
55
|
+
looped on tool calls.
|
|
56
|
+
- **A failed reflect reply reads the same on every engine kind:** the parser's
|
|
57
|
+
own message. 0.9.25-alpha.1 named the engine on an agent engine.
|
|
58
|
+
- **Inference reaches opencode only where akm writes opencode's config:** an
|
|
59
|
+
improve process's `llm` overlay on model work's agent, and an `opencode-sdk`
|
|
60
|
+
engine's `llmEngine` fallback model. Set the rest in your opencode config; a
|
|
61
|
+
task's or workflow's `inference` is an `untranslated-field` notice, as before
|
|
62
|
+
0.9.25-alpha.1. `claude` takes no `--effort`.
|
|
63
|
+
- **An `opencode-sdk` session the dispatch times out on or aborts is aborted on
|
|
64
|
+
the server for every dispatch,** not only model work.
|
|
65
|
+
- **An improve stage retries a reply only when the stage cannot read it,** not
|
|
66
|
+
when it misses the JSON Schema. A reply it can read costs no second call.
|
|
67
|
+
- **opencode model work can read the stash and search it.** The `akm-model-work`
|
|
68
|
+
agent reads, greps and globs in the stash and its working directory, edits
|
|
69
|
+
only in the working directory, and has `akm_search` and `akm_show` from the
|
|
70
|
+
akm-opencode plugin (0.9.21 or later, in your own opencode config). The
|
|
71
|
+
plugin's curation, learning and write gate are off for these dispatches, and
|
|
72
|
+
its state goes to akm's state directory, not `~/.local/state/akm-opencode`.
|
|
73
|
+
|
|
74
|
+
### Removed
|
|
75
|
+
|
|
76
|
+
- **The agent-engine inference fields,** which 0.9.25-alpha.1 accepted: they
|
|
77
|
+
fail to load again.
|
|
78
|
+
- **The `opencode-sdk` step watcher and the git-repository refusal for model
|
|
79
|
+
work's scratch directory.**
|
|
80
|
+
- **opencode model work's step limit.** At the limit opencode sends a "maximum
|
|
81
|
+
steps" message as a trailing assistant message, which a qwen chat template
|
|
82
|
+
(LM Studio, llama-server) renders as the start of the model's reply: LM
|
|
83
|
+
Studio returned nothing, llama-server returned that message as the answer,
|
|
84
|
+
and the dispatch failed. The dispatch timeout bounds a run.
|
|
85
|
+
- **A prompt builder that nothing called** (`buildSchemaRepairPrompt`).
|
|
86
|
+
- **The `--track-usage` / `--no-track-usage` flag on `akm search`, `akm curate`
|
|
87
|
+
and `akm show`.** A successful read always records its usage event, stamped
|
|
88
|
+
with its source (`user`, `improve`, `task` or `audit`); only `user` events
|
|
89
|
+
feed ranking and eval, so machine reads never skew them. Either spelling now
|
|
90
|
+
fails as an unknown flag.
|
|
91
|
+
|
|
92
|
+
### Fixed
|
|
93
|
+
|
|
94
|
+
- **The reflect size guard no longer flags a body that does not grow.** The
|
|
95
|
+
expansion ceiling is capped at 25,000 characters, so a source body longer than
|
|
96
|
+
that was flagged `EXCESSIVE_EXPANSION` even when the proposed body was its own
|
|
97
|
+
length (ratio 1.00), and went to review instead of the judge. A body no longer
|
|
98
|
+
than its source is never expansion; one that grows past the cap still is, by
|
|
99
|
+
any amount. The reflect prompt agrees: it told the model its body could be at
|
|
100
|
+
most 25,000 characters even when the source was longer, and now gives such a
|
|
101
|
+
source's own length.
|
|
102
|
+
- **An asset's or a task's own `tools:` can no longer name the model-work
|
|
103
|
+
policy.** In 0.9.25-alpha.1 exactly `read`, `edit`, `akm search`, `akm show`,
|
|
104
|
+
in that order, skipped `execution.allowedTools`. It is ordinary tools now:
|
|
105
|
+
only akm's own model-work callers ask for the policy.
|
|
106
|
+
- **An agent engine that names its model only in `args` is named in the usage
|
|
107
|
+
report,** where it showed `unattributed`.
|
|
108
|
+
- **`opencode` and `opencode-sdk` engines receive the XDG base-directory
|
|
109
|
+
variables.** Under a custom `XDG_CONFIG_HOME` the spawned opencode missed its
|
|
110
|
+
provider config and failed every dispatch with `Unexpected server error`.
|
|
111
|
+
|
|
112
|
+
## [0.9.25-alpha.1] - 2026-10-02
|
|
113
|
+
|
|
114
|
+
### Changed
|
|
115
|
+
|
|
116
|
+
- **Unattended model work runs under one tool policy, on any engine that
|
|
117
|
+
confines it.** The model may read, edit inside a scratch directory akm removes
|
|
118
|
+
after the dispatch, and run `akm search` and `akm show`; the stash stays
|
|
119
|
+
read-only. An LLM engine has no tools, `claude` confines the policy, and
|
|
120
|
+
`opencode` and `opencode-sdk` run an injected `akm-model-work` agent with
|
|
121
|
+
read and edit only. Every other harness refuses it before the request starts.
|
|
122
|
+
- **One config rule names the engines model work may use.** Every key it reads
|
|
123
|
+
its engine from (`defaults.llmEngine`, `index.*.engine`, a strategy's or
|
|
124
|
+
process's `engine`, an enabled triage `judgment.engine`, a
|
|
125
|
+
`qualityGate.engine`) must name an LLM engine or a `claude`, `opencode` or
|
|
126
|
+
`opencode-sdk` agent engine, or the config fails to load. `--require-engines`
|
|
127
|
+
and `akm health` check agent engines too.
|
|
128
|
+
- **Model work is bounded at 600 seconds on every engine kind** whose engine
|
|
129
|
+
sets no `timeoutMs`; agent and `opencode-sdk` engines ran until they finished.
|
|
130
|
+
- **`akm proposal new` returns the proposal as JSON on every engine kind,**
|
|
131
|
+
with no draft file and no live session. A reply that is not a proposal gets
|
|
132
|
+
one retry.
|
|
133
|
+
- **Reflect asks every engine kind for the same JSON reply and repairs it
|
|
134
|
+
once.** An agent or `opencode-sdk` engine gets reflect's JSON Schema and a
|
|
135
|
+
repair turn, as an LLM engine did. An LLM engine's requests are unchanged.
|
|
136
|
+
- **An improve stage's reply that fails its JSON Schema gets one corrective
|
|
137
|
+
retry,** then the stage reads the last reply with its own parser.
|
|
138
|
+
- **One schema instruction for every agent engine,** appended by the shared
|
|
139
|
+
request lowering: `opencode` and `opencode-sdk` now receive a requested
|
|
140
|
+
schema, and a workflow unit on seven harnesses no longer carries it twice.
|
|
141
|
+
- **Agent and `opencode-sdk` dispatches leave a usage record.**
|
|
142
|
+
- **An `opencode-sdk` engine's LLM fallback comes only from its own
|
|
143
|
+
`llmEngine`; `defaults.llmEngine` no longer supplies one.** If you relied on
|
|
144
|
+
that, set `llmEngine` on the SDK engine. Without one, opencode uses its own
|
|
145
|
+
provider, model and auth.
|
|
146
|
+
- **Inference reaches `opencode`, `opencode-sdk` and `claude` engines.**
|
|
147
|
+
`temperature`, `maxTokens`, `contextLength`, `enableThinking` and
|
|
148
|
+
`reasoningEffort` from an engine, an improve process's `llm` overlay, an
|
|
149
|
+
asset, a workflow or a `models.json` alias were dropped on every agent
|
|
150
|
+
engine. An agent engine may set them, and `claude` takes `reasoningEffort` as
|
|
151
|
+
`--effort`.
|
|
152
|
+
- **Reasoning effort has one word, `reasoningEffort`.** `effort` in a
|
|
153
|
+
`models.json` alias or an asset is read as it, so an LLM engine now sends it
|
|
154
|
+
as `reasoning_effort`.
|
|
155
|
+
|
|
156
|
+
### Removed
|
|
157
|
+
|
|
158
|
+
- **Dead code; nothing you run changes:** each harness's unused `pattern` and
|
|
159
|
+
`structuredOutput` fields, `resolveLlmEngineUse`'s swap of an agent engine
|
|
160
|
+
for an LLM engine, the agent request's unread `effort` hint, and the agent
|
|
161
|
+
file-write contract (`DRAFT_WRITTEN`, reflect's `ref_mismatch` check).
|
|
162
|
+
|
|
163
|
+
### Fixed
|
|
164
|
+
|
|
165
|
+
- **An `opencode-sdk` engine with an LLM fallback reaches its endpoint, and a
|
|
166
|
+
failed dispatch is a failure (#1015).** An SDK error or provider rejection is
|
|
167
|
+
`ok: false` with opencode's message, and a reply's last text part is its answer.
|
|
168
|
+
- **An LLM engine reports a provider error sent with HTTP 200 as a failure**
|
|
169
|
+
(OpenRouter does this), not as an empty reply.
|
|
170
|
+
- **An `opencode` engine runs a persona** instead of failing on
|
|
171
|
+
`--system-prompt`, which opencode 1.18 rejects.
|
|
172
|
+
- **An LLM engine sends a requested schema** unless it sets
|
|
173
|
+
`supportsJsonSchema: false`; it sent one only when it set `true`.
|
|
174
|
+
- **A feature gate's timeout stops the call it bounds,** and a stage call
|
|
175
|
+
reports a timeout or abort as `timeout` or `aborted`, not `error`.
|
|
176
|
+
- **`akm proposal new` keeps the reply's `confidence`.**
|
|
177
|
+
- **Model work on an agent or `opencode-sdk` engine with no `timeoutMs` stops
|
|
178
|
+
after 600 seconds** as documented; it resolved to no timeout.
|
|
179
|
+
|
|
7
180
|
## [0.9.24] - 2026-10-02
|
|
8
181
|
|
|
9
182
|
### Changed
|
package/dist/cli.js
CHANGED
|
@@ -177,7 +177,7 @@ const setupCommand = defineCommand({
|
|
|
177
177
|
// the work, matching the `sync --push/--no-push` pattern. A flag
|
|
178
178
|
// DECLARED as `no-init` can never be negated: `--no-init` parses as
|
|
179
179
|
// "negate `init`", a name nothing declared, leaving the real key at its
|
|
180
|
-
// default forever
|
|
180
|
+
// default forever.
|
|
181
181
|
init: {
|
|
182
182
|
type: "boolean",
|
|
183
183
|
default: true,
|
|
@@ -3,7 +3,6 @@
|
|
|
3
3
|
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
4
|
import { spawnSync } from "node:child_process";
|
|
5
5
|
import { loadConfig } from "../../core/config/config.js";
|
|
6
|
-
import { IMPROVE_PROCESS_ENGINE_CAPABILITIES } from "../../core/config/engine-semantics.js";
|
|
7
6
|
import { listEnvsRecursive } from "../../core/env-secret-ref.js";
|
|
8
7
|
import { ConfigError } from "../../core/errors.js";
|
|
9
8
|
import { EXTRACT_INFRASTRUCTURE_SKIP_REASONS } from "../../core/improve-types.js";
|
|
@@ -15,7 +14,7 @@ import { loadModelMap, mergeModelMapLayers, parseModelMapLayer, readInstalledMod
|
|
|
15
14
|
import { probeEndpointOnce } from "../../llm/client.js";
|
|
16
15
|
import { STATE_DB_FREELIST_WARN_RATIO, } from "../../storage/state-db-integrity.js";
|
|
17
16
|
import { listKeys } from "../env/env.js";
|
|
18
|
-
import { resolveImprovePlan } from "../improve/improve-strategies.js";
|
|
17
|
+
import { MODEL_CALLING_PROCESSES, resolveImprovePlan } from "../improve/improve-strategies.js";
|
|
19
18
|
import { ENGINE_LAST_USED_LOOKBACK_DAYS } from "./engine-usage.js";
|
|
20
19
|
import { ACTIVE_RUN_WARN_MS, TASK_FAIL_RATE_WARN, } from "./types.js";
|
|
21
20
|
/** Probe one connection's reachability, once per endpoint; `undefined` when no probe seam is supplied. */
|
|
@@ -142,7 +141,7 @@ async function runConfiguredEngineProbe(checkName, engineName, config, deps, rea
|
|
|
142
141
|
evidence: { engine: engineName, runtimeKind: "sdk", binaryAvailable: false },
|
|
143
142
|
};
|
|
144
143
|
}
|
|
145
|
-
const fallbackEngine = configuredEngine.llmEngine
|
|
144
|
+
const fallbackEngine = configuredEngine.llmEngine;
|
|
146
145
|
let fallback;
|
|
147
146
|
let fallbackCredential;
|
|
148
147
|
let fallbackApiKeyFile;
|
|
@@ -523,17 +522,17 @@ export function probeActiveImproveStrategy(deps = {}) {
|
|
|
523
522
|
.sort(([a], [b]) => a.localeCompare(b))
|
|
524
523
|
.map(([process, engine]) => `${process}: "${engine}"`)
|
|
525
524
|
.join(", ");
|
|
526
|
-
// #957: fail only when the strategy's
|
|
527
|
-
//
|
|
528
|
-
//
|
|
529
|
-
//
|
|
525
|
+
// #957: fail only when the strategy's model work would be a total no-op —
|
|
526
|
+
// every process the strategy actually enabled among the ones that call a
|
|
527
|
+
// model themselves (MODEL_CALLING_PROCESSES, any engine kind) ended up
|
|
528
|
+
// unavailable. A partial failure (some processes still have a working
|
|
530
529
|
// engine) stays a `warn`, matching #914's "a credential warn stays a warn"
|
|
531
530
|
// policy for the general per-engine probes; this is the strategy-scoped
|
|
532
531
|
// "is the whole run a no-op" question instead.
|
|
533
|
-
const
|
|
534
|
-
const
|
|
535
|
-
const
|
|
536
|
-
const allRequiredUnavailable =
|
|
532
|
+
const modelProcessNames = [...MODEL_CALLING_PROCESSES];
|
|
533
|
+
const requiredModelProcessNames = modelProcessNames.filter((name) => plan.processes[name].enabled || plan.engineUnavailable.some((item) => item.process === name));
|
|
534
|
+
const availableModelProcessNames = modelProcessNames.filter((name) => plan.processes[name].enabled);
|
|
535
|
+
const allRequiredUnavailable = requiredModelProcessNames.length > 0 && availableModelProcessNames.length === 0;
|
|
537
536
|
const status = unavailableProcesses.length === 0 ? "pass" : allRequiredUnavailable ? "fail" : "warn";
|
|
538
537
|
return {
|
|
539
538
|
check: {
|
|
@@ -36,6 +36,7 @@ import { concurrentMap } from "../../../core/concurrent.js";
|
|
|
36
36
|
import { parseEmbeddedJsonResponse } from "../../../core/parse.js";
|
|
37
37
|
import { DERIVED_SUFFIX } from "../../../core/recognition-util.js";
|
|
38
38
|
import { warnOnce } from "../../../core/warn.js";
|
|
39
|
+
import { runnerLlmConnection } from "../../../integrations/agent/runner.js";
|
|
39
40
|
import { assertRunnerCredentials } from "../../../integrations/agent/runner-dispatch.js";
|
|
40
41
|
import { runGit } from "../../../sources/providers/git-install.js";
|
|
41
42
|
import { closeDatabase, openExistingDatabase, openReadonlyExistingDatabase, } from "../../../storage/repositories/index-connection.js";
|
|
@@ -501,10 +502,11 @@ async function judgeOne(ctx, candidate) {
|
|
|
501
502
|
request: {
|
|
502
503
|
responseSchema: PAIR_JUDGE_JSON_SCHEMA,
|
|
503
504
|
enableThinking: false,
|
|
504
|
-
timeoutMs: ctx.llmRunner.timeoutMs,
|
|
505
|
+
...(Object.hasOwn(ctx.llmRunner, "timeoutMs") ? { timeoutMs: ctx.llmRunner.timeoutMs } : {}),
|
|
505
506
|
signal: ctx.opts.signal,
|
|
506
507
|
...(ctx.chat ? { chat: ctx.chat } : {}),
|
|
507
508
|
},
|
|
509
|
+
parse: parsePairJudgeResponse,
|
|
508
510
|
...(ctx.opts.onNotices ? { onNotices: ctx.opts.onNotices } : {}),
|
|
509
511
|
});
|
|
510
512
|
if (!outcome.ok)
|
|
@@ -747,7 +749,7 @@ seams = {}) {
|
|
|
747
749
|
// entirely and needs no credential.
|
|
748
750
|
if (!seams.chat)
|
|
749
751
|
assertRunnerCredentials(llmRunner);
|
|
750
|
-
const results = await concurrentMap(judgeable, (candidate) => judgeOne(ctx, candidate), llmRunner
|
|
752
|
+
const results = await concurrentMap(judgeable, (candidate) => judgeOne(ctx, candidate), runnerLlmConnection(llmRunner)?.concurrency ?? 1, { signal: opts.signal });
|
|
751
753
|
results.forEach((r, idx) => {
|
|
752
754
|
// Must-fix 3: `concurrentMap` leaves an entry `undefined` for a call an
|
|
753
755
|
// aborted run never sent at all — `r?.failed === true` reads that as
|
|
@@ -36,6 +36,7 @@ import { resolveWriteTarget } from "../../core/write-source.js";
|
|
|
36
36
|
import { deriveInstallations } from "../../indexer/installations.js";
|
|
37
37
|
import { resolveSourceEntries } from "../../indexer/search/search-source.js";
|
|
38
38
|
import { USAGE_EVENT_RETENTION_DAYS } from "../../indexer/usage/usage-events.js";
|
|
39
|
+
import { runnerLlmConnection } from "../../integrations/agent/runner.js";
|
|
39
40
|
import { assertRunnerCredentials } from "../../integrations/agent/runner-dispatch.js";
|
|
40
41
|
import { cosineSimilarity, embedBatch, resolveEmbeddingModelId } from "../../llm/embedder.js";
|
|
41
42
|
import { getBodyEmbeddings, upsertBodyEmbeddings } from "../../storage/repositories/embeddings-repository.js";
|
|
@@ -52,6 +53,10 @@ import { resolveImproveStrategy, resolveProcessEnabled } from "./improve-strateg
|
|
|
52
53
|
import { isContentDrivenRow, isLedgerBlocked, ledgerKey, loadLedgerSnapshot, recordLedgerAttempt } from "./ledger.js";
|
|
53
54
|
import { isInRetrievalScope, loadRetrievalScope } from "./retrieval-scope.js";
|
|
54
55
|
import { callStage, mintProposal, noticeSet, stageRunner } from "./stage.js";
|
|
56
|
+
function parsePlan(raw) {
|
|
57
|
+
const plan = parseEmbeddedJsonResponse(raw);
|
|
58
|
+
return plan && Array.isArray(plan.operations) ? { ...plan, operations: plan.operations } : undefined;
|
|
59
|
+
}
|
|
55
60
|
/** A plan op worth acting on. Retired advisory ops (merge/delete/contradict) are dropped, never thrown on. */
|
|
56
61
|
export function isValidOp(op) {
|
|
57
62
|
if (typeof op !== "object" || op === null)
|
|
@@ -566,9 +571,10 @@ async function judgeConsolidationChunks(args) {
|
|
|
566
571
|
request: {
|
|
567
572
|
responseSchema: CONSOLIDATE_PLAN_JSON_SCHEMA,
|
|
568
573
|
enableThinking: false,
|
|
569
|
-
timeoutMs: llmRunner.timeoutMs,
|
|
574
|
+
...(Object.hasOwn(llmRunner, "timeoutMs") ? { timeoutMs: llmRunner.timeoutMs } : {}),
|
|
570
575
|
signal: opts.signal,
|
|
571
576
|
},
|
|
577
|
+
parse: parsePlan,
|
|
572
578
|
...(opts.onNotices ? { onNotices: opts.onNotices } : {}),
|
|
573
579
|
});
|
|
574
580
|
if (!outcome.ok) {
|
|
@@ -576,8 +582,8 @@ async function judgeConsolidationChunks(args) {
|
|
|
576
582
|
continue;
|
|
577
583
|
}
|
|
578
584
|
warnVerbose(`[akm:consolidate] ${label} raw response (first 500 chars): ${outcome.raw.slice(0, 500)}`);
|
|
579
|
-
const parsed =
|
|
580
|
-
if (!parsed
|
|
585
|
+
const parsed = parsePlan(outcome.raw);
|
|
586
|
+
if (!parsed) {
|
|
581
587
|
const hint = outcome.raw.trim() === "" ? " (empty response — if using a thinking model, disable thinking mode)" : "";
|
|
582
588
|
const msg = `Chunk ${chunkIdx + 1}: invalid plan from AI — skipping.${hint}`;
|
|
583
589
|
warn(msg);
|
|
@@ -619,7 +625,7 @@ async function planConsolidation(opts, config, stashDir, memories, warnings, sta
|
|
|
619
625
|
const llmRunner = opts.llmRunner ?? undefined;
|
|
620
626
|
// 500 body chars per memory keep the judgement useful; chunk size varies instead.
|
|
621
627
|
const bodyTruncation = 500;
|
|
622
|
-
const chunkSize = computeSafeChunkSize(llmRunner?.
|
|
628
|
+
const chunkSize = computeSafeChunkSize((llmRunner && runnerLlmConnection(llmRunner)?.contextLength) ?? DEFAULT_CONTEXT_LENGTH_TOKENS, bodyTruncation, opts.maxChunkSize);
|
|
623
629
|
const sourceName = opts.target ?? stashDir;
|
|
624
630
|
let budgeted = memories;
|
|
625
631
|
const budgetMs = opts.signal?.remainingBudgetMs;
|
|
@@ -22,6 +22,8 @@ function mergeDefaults(farther, nearer) {
|
|
|
22
22
|
/**
|
|
23
23
|
* Resolve improve-owned model work through the canonical execution cascade:
|
|
24
24
|
* defaults.llmEngine -> strategy -> index.<pass> -> process -> current invocation.
|
|
25
|
+
* The engine may be of any kind that confines the model-work tool policy: one
|
|
26
|
+
* that cannot, chosen with `--engine` say, is refused here, before any work.
|
|
25
27
|
*/
|
|
26
28
|
export function resolveImproveExecution(options) {
|
|
27
29
|
const defaultEngine = options.config.defaults?.llmEngine;
|
|
@@ -33,22 +35,13 @@ export function resolveImproveExecution(options) {
|
|
|
33
35
|
if (selectedEngine === undefined || selectedEngine === null)
|
|
34
36
|
return null;
|
|
35
37
|
const invocationDefaults = mergeDefaults(mergeDefaults(defaultEngine ? { engine: defaultEngine } : {}, profileDefaults), indexDefaults);
|
|
36
|
-
const current = mergeDefaults(processDefaults, currentDefaults);
|
|
37
38
|
const prepared = resolveExecution({
|
|
38
39
|
content: `improve ${options.processName} execution selection`,
|
|
39
40
|
config: options.config,
|
|
40
41
|
invocationDefaults,
|
|
41
|
-
current,
|
|
42
|
+
current: mergeDefaults(processDefaults, currentDefaults),
|
|
43
|
+
modelWork: true,
|
|
42
44
|
});
|
|
43
45
|
const lowered = buildExecution(prepared.request, prepared.runner);
|
|
44
46
|
return Object.freeze({ runner: lowered.runner, notices: lowered.notices });
|
|
45
47
|
}
|
|
46
|
-
export function resolveImproveLlmExecution(options) {
|
|
47
|
-
const resolved = resolveImproveExecution(options);
|
|
48
|
-
if (!resolved)
|
|
49
|
-
return null;
|
|
50
|
-
if (resolved.runner.kind !== "llm") {
|
|
51
|
-
return null;
|
|
52
|
-
}
|
|
53
|
-
return { runner: resolved.runner, notices: resolved.notices };
|
|
54
|
-
}
|
|
@@ -9,9 +9,9 @@
|
|
|
9
9
|
* session data into the markdown template loaded from
|
|
10
10
|
* `src/assets/prompts/extract-session.md`.
|
|
11
11
|
*
|
|
12
|
-
* The schema is intentionally strict —
|
|
13
|
-
*
|
|
14
|
-
* happy path. `additionalProperties: false` means any hallucinated keys
|
|
12
|
+
* The schema is intentionally strict — a provider that honours
|
|
13
|
+
* `response_format` enforces shape upstream, so the parser only has to handle
|
|
14
|
+
* the happy path. `additionalProperties: false` means any hallucinated keys
|
|
15
15
|
* the model emits get dropped before we parse.
|
|
16
16
|
*/
|
|
17
17
|
import promptTemplate from "../../assets/prompts/extract-session.md" with { type: "text" };
|
|
@@ -20,7 +20,7 @@ const EXTRACT_CANDIDATE_NAME_PATTERN = "^[a-z0-9](?:[a-z0-9-]*[a-z0-9])?(?:/[a-z
|
|
|
20
20
|
const EXTRACT_CANDIDATE_NAME_RE = new RegExp(EXTRACT_CANDIDATE_NAME_PATTERN);
|
|
21
21
|
/**
|
|
22
22
|
* JSON Schema for the structured extract output. Passed to `chatCompletion`
|
|
23
|
-
*
|
|
23
|
+
* unless the configured LLM connection sets `supportsJsonSchema: false`.
|
|
24
24
|
*
|
|
25
25
|
* Shape:
|
|
26
26
|
* {
|
|
@@ -32,16 +32,15 @@ import { indexWrittenAssets } from "../../indexer/index-written-assets.js";
|
|
|
32
32
|
import { assertRunnerCredentials } from "../../integrations/agent/runner-dispatch.js";
|
|
33
33
|
import { getAvailableHarnesses } from "../../integrations/session-logs/index.js";
|
|
34
34
|
import { preFilterSession } from "../../integrations/session-logs/pre-filter.js";
|
|
35
|
-
import { isJsonSchemaKnownUnsupported } from "../../llm/client.js";
|
|
36
35
|
import { getExtractedSessionsMap, getLastExtractRunAt, shouldSkipAlreadyExtractedSession, upsertExtractedSession, } from "../../storage/repositories/extract-sessions-repository.js";
|
|
37
36
|
import { openSqliteReadSnapshot } from "../../storage/sqlite-read-snapshot.js";
|
|
38
37
|
import { contentHash } from "./content-hash.js";
|
|
39
|
-
import {
|
|
38
|
+
import { resolveImproveExecution } from "./execution.js";
|
|
40
39
|
import { buildExtractPrompt, EXTRACT_JSON_SCHEMA, parseExtractPayload, } from "./extract-prompt.js";
|
|
41
40
|
import { cloneAndFreeze, resolveImproveStrategy, resolveProcessEnabled } from "./improve-strategies.js";
|
|
42
41
|
import { isLedgerBlocked, ledgerKey, loadLedgerSnapshot } from "./ledger.js";
|
|
43
42
|
import { buildSessionSummaryPrompt, parseSessionSummary, SESSION_SUMMARY_JSON_SCHEMA, sessionMeetsDurationGate, writeSessionAsset, } from "./session-asset.js";
|
|
44
|
-
import { callStage, mintProposal, noticeSet } from "./stage.js";
|
|
43
|
+
import { callStage, callStageOnce, mintProposal, noticeSet } from "./stage.js";
|
|
45
44
|
/** Minimum session duration (minutes) for writing a session asset. */
|
|
46
45
|
const DEFAULT_MIN_SESSION_DURATION_MINUTES = 5;
|
|
47
46
|
/** Raw session size (chars) below which the LLM call is skipped; only truly empty sessions are safe to skip. */
|
|
@@ -119,7 +118,7 @@ export function resolveStandaloneExtractPlan(config, selection) {
|
|
|
119
118
|
}
|
|
120
119
|
const selected = resolveImproveStrategy(selection.strategy, config);
|
|
121
120
|
const process = cloneAndFreeze(getImproveProcessConfig("extract", selected.config) ?? {});
|
|
122
|
-
const resolved =
|
|
121
|
+
const resolved = resolveImproveExecution({
|
|
123
122
|
config,
|
|
124
123
|
profile: selected.config,
|
|
125
124
|
process,
|
|
@@ -130,7 +129,7 @@ export function resolveStandaloneExtractPlan(config, selection) {
|
|
|
130
129
|
processName: "extract",
|
|
131
130
|
});
|
|
132
131
|
if (!resolved) {
|
|
133
|
-
throw new ConfigError("No
|
|
132
|
+
throw new ConfigError("No engine configured for extract. Set defaults.llmEngine, pass --engine, or select an improve strategy with processes.extract.engine.", "LLM_NOT_CONFIGURED");
|
|
134
133
|
}
|
|
135
134
|
const runner = resolved.runner;
|
|
136
135
|
return Object.freeze({
|
|
@@ -358,15 +357,16 @@ function planExtractSessions(args) {
|
|
|
358
357
|
}
|
|
359
358
|
const EXTRACT_LLM_UNAVAILABLE = Symbol("extract-llm-unavailable");
|
|
360
359
|
/**
|
|
361
|
-
* One session's extraction call
|
|
362
|
-
*
|
|
360
|
+
* One session's extraction call, with one corrective retry; configuration
|
|
361
|
+
* errors escape before any state is written.
|
|
363
362
|
*/
|
|
364
363
|
async function extractFromSession(run, prompt) {
|
|
365
364
|
const { llmRunner } = run;
|
|
366
365
|
try {
|
|
367
366
|
const result = await runStructured({
|
|
368
367
|
dispatch: async (feedback) => {
|
|
369
|
-
|
|
368
|
+
// This loop parses and repairs the reply itself, so each attempt is one unvalidated dispatch.
|
|
369
|
+
const outcome = await callStageOnce({
|
|
370
370
|
feature: "session_extraction",
|
|
371
371
|
runner: llmRunner,
|
|
372
372
|
prompt: feedback ? `${prompt}\n\n## Corrective output instruction\n\n${feedback}` : prompt,
|
|
@@ -388,9 +388,6 @@ async function extractFromSession(run, prompt) {
|
|
|
388
388
|
return payload.parseFailure ? undefined : payload;
|
|
389
389
|
},
|
|
390
390
|
validate: (payload) => ({ ok: true, value: payload }),
|
|
391
|
-
maxAttempts: llmRunner.connection.supportsJsonSchema !== false && !isJsonSchemaKnownUnsupported(llmRunner.connection)
|
|
392
|
-
? 1
|
|
393
|
-
: 2,
|
|
394
391
|
buildFeedback: () => "Your previous response did not contain a valid extraction payload. Respond with ONLY a JSON object matching the requested schema, with a candidates array and no prose or code fences.",
|
|
395
392
|
});
|
|
396
393
|
if (result.ok)
|
|
@@ -714,13 +711,13 @@ function resolveExtractRun(options, config, process, activeProfile) {
|
|
|
714
711
|
llmRunner = options.llmRunner;
|
|
715
712
|
}
|
|
716
713
|
else {
|
|
717
|
-
const resolved =
|
|
714
|
+
const resolved = resolveImproveExecution({ config, profile: activeProfile, process, processName: "extract" });
|
|
718
715
|
llmRunner = resolved?.runner;
|
|
719
716
|
if (resolved)
|
|
720
717
|
notices.add(resolved.notices);
|
|
721
718
|
}
|
|
722
719
|
if (!llmRunner) {
|
|
723
|
-
throw new ConfigError("No
|
|
720
|
+
throw new ConfigError("No engine configured for extract. Set defaults.llmEngine or improve.strategies.<name>.processes.extract.engine.", "LLM_NOT_CONFIGURED");
|
|
724
721
|
}
|
|
725
722
|
const runner = llmRunner;
|
|
726
723
|
const timeoutMs = options.resolvedPlan
|
|
@@ -744,6 +741,7 @@ function resolveExtractRun(options, config, process, activeProfile) {
|
|
|
744
741
|
...(options.signal ? { signal: options.signal } : {}),
|
|
745
742
|
...(options.chat ? { chat: options.chat } : {}),
|
|
746
743
|
},
|
|
744
|
+
parse: parseSessionSummary,
|
|
747
745
|
onNotices: notices.add,
|
|
748
746
|
});
|
|
749
747
|
return parseSessionSummary(outcome.ok ? outcome.raw : "");
|
|
@@ -15,17 +15,20 @@ import { redactSensitiveText } from "../../core/redaction.js";
|
|
|
15
15
|
import { clearLogFile, setLogFile, warn } from "../../core/warn.js";
|
|
16
16
|
import { resolveWriteTarget } from "../../core/write-source.js";
|
|
17
17
|
import { DEFAULT_LLM_TIMEOUT_MS } from "../../integrations/agent/config.js";
|
|
18
|
+
import { defaultWhich } from "../../integrations/agent/detect.js";
|
|
18
19
|
import { collectEngineCredentialValues } from "../../integrations/agent/engine-resolution.js";
|
|
19
20
|
import { probeLlmReachable } from "../../llm/client.js";
|
|
20
21
|
import { getOutputMode } from "../../output/context.js";
|
|
21
22
|
import { deliverRendered } from "../../output/html-render.js";
|
|
23
|
+
import { readStdin } from "../../runtime.js";
|
|
22
24
|
import { akmImprove, IMPROVE_TARGET_FLAG, resolveImproveReadSource } from "./improve.js";
|
|
23
25
|
import { runImproveReportQuery } from "./improve-report.js";
|
|
24
26
|
import { buildImproveRunId, recordImproveRunResult, recordTerminatedImproveRun, } from "./improve-result-file.js";
|
|
25
27
|
import { runImproveSession } from "./improve-session.js";
|
|
26
|
-
import { resolveImprovePlan, } from "./improve-strategies.js";
|
|
28
|
+
import { resolveImprovePlan, resolveImproveStrategy, } from "./improve-strategies.js";
|
|
27
29
|
import { formatUsageReportTable } from "./improve-usage-report.js";
|
|
28
30
|
import { renderReflectPromptPreview } from "./reflect.js";
|
|
31
|
+
import { resolveQualityGateJudge, runReflectQualityJudge } from "./stage.js";
|
|
29
32
|
let akmImproveForRun = akmImprove;
|
|
30
33
|
/** Swap the CLI's improve work implementation in deterministic subprocess tests. */
|
|
31
34
|
export function _setAkmImproveForTests(fake) {
|
|
@@ -73,26 +76,19 @@ function assertRequiredEnginesAvailable(plan) {
|
|
|
73
76
|
const lines = plan.engineUnavailable.map((item) => ` - ${item.process} (${item.configKey}): ${item.reason}`);
|
|
74
77
|
throw new ConfigError(`--require-engines: ${plan.engineUnavailable.length} improve process${plan.engineUnavailable.length === 1 ? "" : "es"} cannot run because ${plan.engineUnavailable.length === 1 ? "its" : "their"} engine is unavailable:\n${lines.join("\n")}`, "LLM_NOT_CONFIGURED");
|
|
75
78
|
}
|
|
76
|
-
/** Every
|
|
79
|
+
/** Every engine the plan would dispatch to, triage's judgment engine included. */
|
|
77
80
|
function collectRequiredEngineTargets(plan) {
|
|
78
|
-
const
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
targets.push({
|
|
90
|
-
process: "triage.judgment",
|
|
91
|
-
engine: plan.triageJudgment.engine,
|
|
92
|
-
connection: probeConnection(plan.triageJudgment),
|
|
93
|
-
});
|
|
94
|
-
}
|
|
95
|
-
return targets;
|
|
81
|
+
const runners = Object.entries(plan.processes)
|
|
82
|
+
.flatMap(([processName, process]) => (process.runner ? [[processName, process.runner]] : []))
|
|
83
|
+
.concat(plan.triageJudgment ? [["triage.judgment", plan.triageJudgment]] : []);
|
|
84
|
+
return runners.map(([processName, runner]) => ({
|
|
85
|
+
process: processName,
|
|
86
|
+
engine: runner.engine,
|
|
87
|
+
...(runner.kind === "llm" ? { connection: probeConnection(runner) } : { bin: runner.profile.bin }),
|
|
88
|
+
...(runner.kind === "sdk" && runner.fallbackConnection
|
|
89
|
+
? { connection: probeConnection({ connection: runner.fallbackConnection, timeoutMs: runner.fallbackTimeoutMs }) }
|
|
90
|
+
: {}),
|
|
91
|
+
}));
|
|
96
92
|
}
|
|
97
93
|
/** The resolved engine keeps its request timeout beside the connection (the runtime merges it in); the probe needs it on the connection. */
|
|
98
94
|
function probeConnection(runner) {
|
|
@@ -112,37 +108,42 @@ const REQUIRED_ENGINE_PROBE_MAX_MS = 120_000;
|
|
|
112
108
|
/**
|
|
113
109
|
* `--require-engines`, live: probe each connection's real completion path
|
|
114
110
|
* (a gateway can list a model whose completion route is dead, #980), once per
|
|
115
|
-
* endpoint + model, within {@link requiredEngineProbeTimeoutMs}
|
|
116
|
-
*
|
|
111
|
+
* endpoint + model, within {@link requiredEngineProbeTimeoutMs}, and look each
|
|
112
|
+
* agent harness's binary up on PATH. Returns each target's latency for the run
|
|
113
|
+
* result (R17); an unreachable one fails the run.
|
|
117
114
|
*/
|
|
118
|
-
export async function assertRequiredEnginesReachable(plan, probeReachable = (connection) => probeLlmReachable(connection, requiredEngineProbeTimeoutMs(connection))) {
|
|
115
|
+
export async function assertRequiredEnginesReachable(plan, probeReachable = (connection) => probeLlmReachable(connection, requiredEngineProbeTimeoutMs(connection)), which = defaultWhich) {
|
|
119
116
|
const targets = collectRequiredEngineTargets(plan);
|
|
120
117
|
if (targets.length === 0)
|
|
121
118
|
return [];
|
|
122
119
|
const probesByConnection = new Map();
|
|
123
|
-
const
|
|
124
|
-
const key = `${
|
|
120
|
+
const probeConnectionOnce = (connection) => {
|
|
121
|
+
const key = `${connection.endpoint.replace(/\/+$/, "")}|${connection.model}`;
|
|
125
122
|
let pending = probesByConnection.get(key);
|
|
126
123
|
if (!pending) {
|
|
127
124
|
const probeStartedAt = Date.now();
|
|
128
|
-
pending = probeReachable(
|
|
129
|
-
reach,
|
|
130
|
-
latencyMs: Date.now() - probeStartedAt,
|
|
131
|
-
}));
|
|
125
|
+
pending = probeReachable(connection).then((reach) => ({ reach, latencyMs: Date.now() - probeStartedAt }));
|
|
132
126
|
probesByConnection.set(key, pending);
|
|
133
127
|
}
|
|
134
|
-
|
|
135
|
-
|
|
128
|
+
return pending;
|
|
129
|
+
};
|
|
130
|
+
const probed = await Promise.all(targets.map(async (target) => {
|
|
131
|
+
if (target.bin !== undefined && which(target.bin) === undefined) {
|
|
132
|
+
return { ...target, reach: { reachable: false, error: `${target.bin} is not on PATH` }, latencyMs: 0 };
|
|
133
|
+
}
|
|
134
|
+
if (!target.connection)
|
|
135
|
+
return { ...target, reach: { reachable: true }, latencyMs: 0 };
|
|
136
|
+
return { ...target, ...(await probeConnectionOnce(target.connection)) };
|
|
136
137
|
}));
|
|
137
138
|
const unreachable = probed.filter((item) => !item.reach.reachable);
|
|
138
139
|
if (unreachable.length > 0) {
|
|
139
|
-
const lines = unreachable.map((item) => ` - ${item.process} (engine "${item.engine}", ${item.connection.
|
|
140
|
-
throw new ConfigError(`--require-engines: ${unreachable.length} improve process${unreachable.length === 1 ? "" : "es"} cannot run because ${unreachable.length === 1 ? "its" : "their"} engine completion path is not reachable:\n${lines.join("\n")}`, "LLM_NOT_CONFIGURED", "Check that each listed endpoint is up and serves its model. The probe is one short completion, bounded by the engine's timeoutMs (at most two minutes).");
|
|
140
|
+
const lines = unreachable.map((item) => ` - ${item.process} (engine "${item.engine}", ${item.connection?.endpoint ?? item.bin}): ${item.reach.error ?? "did not respond"}`);
|
|
141
|
+
throw new ConfigError(`--require-engines: ${unreachable.length} improve process${unreachable.length === 1 ? "" : "es"} cannot run because ${unreachable.length === 1 ? "its" : "their"} engine completion path is not reachable:\n${lines.join("\n")}`, "LLM_NOT_CONFIGURED", "Check that each listed endpoint is up and serves its model, and that each listed agent binary is installed. The endpoint probe is one short completion, bounded by the engine's timeoutMs (at most two minutes).");
|
|
141
142
|
}
|
|
142
143
|
return probed.map((item) => ({
|
|
143
144
|
process: item.process,
|
|
144
145
|
engine: item.engine,
|
|
145
|
-
endpoint: item.connection.
|
|
146
|
+
endpoint: item.connection?.endpoint ?? item.bin,
|
|
146
147
|
reachable: item.reach.reachable,
|
|
147
148
|
latencyMs: item.latencyMs,
|
|
148
149
|
}));
|
|
@@ -189,6 +190,32 @@ function rejectReportOnlyFlags(args) {
|
|
|
189
190
|
return;
|
|
190
191
|
throw new UsageError(`\`${flag}\` only applies to \`akm improve report\`. Use \`akm improve report ${flag} <value>\` instead.`, "INVALID_FLAG_VALUE");
|
|
191
192
|
}
|
|
193
|
+
/**
|
|
194
|
+
* `akm improve judge`: reflect's quality judge on one revision, read as
|
|
195
|
+
* `{"source", "candidate", "feedback"}` JSON from stdin, with the engine the
|
|
196
|
+
* strategy's reflect quality gate names. It writes nothing.
|
|
197
|
+
*/
|
|
198
|
+
async function runImproveJudgeCli(strategyName) {
|
|
199
|
+
const input = process.stdin.isTTY
|
|
200
|
+
? {}
|
|
201
|
+
: JSON.parse((await readStdin()).toString("utf8"));
|
|
202
|
+
const { source, candidate, feedback, ref } = input;
|
|
203
|
+
if (typeof source !== "string" || typeof candidate !== "string") {
|
|
204
|
+
throw new UsageError('`akm improve judge` reads {"source": "...", "candidate": "...", "feedback": "...", "ref": "..."} JSON from stdin.', "MISSING_REQUIRED_ARGUMENT");
|
|
205
|
+
}
|
|
206
|
+
const config = loadConfig();
|
|
207
|
+
const judge = resolveQualityGateJudge(config, resolveImproveStrategy(strategyName, config).config, "reflect");
|
|
208
|
+
if (!judge) {
|
|
209
|
+
throw new ConfigError("`akm improve judge` judges with the reflect quality gate's engine. Set processes.reflect.qualityGate.engine.", "INVALID_CONFIG_FILE");
|
|
210
|
+
}
|
|
211
|
+
const notes = typeof feedback === "string" && feedback.trim() !== "" ? [feedback.trim()] : [];
|
|
212
|
+
const verdict = await runReflectQualityJudge(config, candidate, source, notes, undefined, {
|
|
213
|
+
runnerSelectionFrozen: true,
|
|
214
|
+
llmRunner: judge,
|
|
215
|
+
...(typeof ref === "string" && ref ? { ref } : {}),
|
|
216
|
+
});
|
|
217
|
+
output("improve-judge", { engine: judge.engine, ...verdict });
|
|
218
|
+
}
|
|
192
219
|
export const improveCommand = defineCommand({
|
|
193
220
|
meta: {
|
|
194
221
|
name: "improve",
|
|
@@ -271,6 +298,10 @@ export const improveCommand = defineCommand({
|
|
|
271
298
|
return;
|
|
272
299
|
}
|
|
273
300
|
rejectReportOnlyFlags(args);
|
|
301
|
+
if (getStringArg(args, "scope") === "judge") {
|
|
302
|
+
await runImproveJudgeCli(getStringArg(args, "strategy"));
|
|
303
|
+
return;
|
|
304
|
+
}
|
|
274
305
|
rejectRetiredImproveTargetFlag();
|
|
275
306
|
const jsonToStdout = args["json-to-stdout"];
|
|
276
307
|
const targetArg = getStringArg(args, "bundle");
|