akm-cli 0.9.24 → 0.9.25-alpha.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +173 -0
- package/dist/cli.js +1 -1
- package/dist/commands/health/checks.js +10 -11
- package/dist/commands/improve/consolidate/pair-pass.js +4 -2
- package/dist/commands/improve/consolidate.js +10 -4
- package/dist/commands/improve/execution.js +4 -11
- package/dist/commands/improve/extract-prompt.js +4 -4
- package/dist/commands/improve/extract.js +11 -13
- package/dist/commands/improve/improve-cli.js +65 -34
- package/dist/commands/improve/improve-strategies.js +49 -43
- package/dist/commands/improve/improve-usage-report.js +8 -17
- package/dist/commands/improve/loop-stages.js +3 -0
- package/dist/commands/improve/preparation.js +3 -1
- package/dist/commands/improve/reflect-noise.js +125 -0
- package/dist/commands/improve/reflect.js +105 -172
- package/dist/commands/improve/retrieval-gate.js +7 -2
- package/dist/commands/improve/stage.js +67 -24
- package/dist/commands/proposal/drain.js +11 -2
- package/dist/commands/proposal/proposal-cli.js +1 -5
- package/dist/commands/proposal/propose-cli.js +2 -2
- package/dist/commands/proposal/propose.js +72 -84
- package/dist/commands/proposal/validators/proposal-quality-validators.js +4 -2
- package/dist/commands/read/search-cli.js +0 -38
- package/dist/commands/remember.js +3 -3
- package/dist/commands/sources/schema-repair.js +1 -1
- package/dist/core/config/config-schema.js +37 -60
- package/dist/core/config/engine-semantics.js +15 -11
- package/dist/core/config/schema/improve-processes.js +18 -2
- package/dist/core/improve-result.js +3 -3
- package/dist/core/redaction.js +4 -0
- package/dist/core/spawn-env.js +25 -0
- package/dist/core/structured.js +11 -1
- package/dist/execution/source.js +10 -0
- package/dist/indexer/passes/memory-inference.js +2 -1
- package/dist/integrations/agent/builder-shared.js +15 -0
- package/dist/integrations/agent/config.js +1 -1
- package/dist/integrations/agent/engine-resolution.js +13 -31
- package/dist/integrations/agent/execution.js +48 -22
- package/dist/integrations/agent/index.js +1 -1
- package/dist/integrations/agent/profiles.js +2 -2
- package/dist/integrations/agent/prompts.js +55 -114
- package/dist/integrations/agent/request-lowering.js +21 -8
- package/dist/integrations/agent/runner-dispatch.js +96 -3
- package/dist/integrations/agent/runner.js +8 -2
- package/dist/integrations/harnesses/aider/agent-builder.js +10 -23
- package/dist/integrations/harnesses/aider/index.js +0 -5
- package/dist/integrations/harnesses/amazonq/agent-builder.js +9 -21
- package/dist/integrations/harnesses/amazonq/index.js +0 -5
- package/dist/integrations/harnesses/claude/agent-builder.js +47 -36
- package/dist/integrations/harnesses/claude/index.js +0 -14
- package/dist/integrations/harnesses/codex/agent-builder.js +3 -4
- package/dist/integrations/harnesses/codex/index.js +0 -4
- package/dist/integrations/harnesses/copilot/agent-builder.js +4 -13
- package/dist/integrations/harnesses/copilot/index.js +2 -7
- package/dist/integrations/harnesses/gemini/agent-builder.js +4 -13
- package/dist/integrations/harnesses/gemini/index.js +0 -5
- package/dist/integrations/harnesses/ids.js +12 -10
- package/dist/integrations/harnesses/opencode/agent-builder.js +41 -9
- package/dist/integrations/harnesses/opencode/index.js +0 -8
- package/dist/integrations/harnesses/opencode/model-config.js +33 -0
- package/dist/integrations/harnesses/opencode/model-work-agent.js +115 -0
- package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -7
- package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +114 -28
- package/dist/integrations/harnesses/openhands/agent-builder.js +7 -21
- package/dist/integrations/harnesses/openhands/index.js +0 -5
- package/dist/integrations/harnesses/pi/agent-builder.js +7 -21
- package/dist/integrations/harnesses/pi/index.js +0 -5
- package/dist/llm/client.js +5 -0
- package/dist/llm/feature-gate.js +12 -9
- package/dist/llm/index-passes.js +2 -5
- package/dist/llm/memory-infer.js +6 -5
- package/dist/llm/structured-call.js +33 -11
- package/dist/output/shapes/passthrough.js +1 -0
- package/dist/scripts/akm-migrate-node.js +298 -228
- package/dist/scripts/akm-migrate.js +298 -228
- package/dist/workflows/exec/step-work.js +6 -5
- package/dist/workflows/exec/unit-dispatch.js +4 -13
- package/dist/workflows/freeze/step-values.js +1 -1
- package/docs/reference/cli.md +41 -18
- package/docs/reference/configuration.md +165 -12
- package/docs/reference/data-and-telemetry.md +2 -3
- package/docs/reference/workflow-schema.md +10 -9
- package/package.json +1 -1
- package/schemas/akm-config.json +108 -0
|
@@ -6,9 +6,9 @@ import { ConfigError } from "../../core/errors.js";
|
|
|
6
6
|
import { DURATION_UNITS, parseDuration } from "../../core/time.js";
|
|
7
7
|
import { EXECUTION_MAX_TIMEOUT_MS } from "../../execution/limits.js";
|
|
8
8
|
import { createInlineResolvedCommand, createResolvedExecutionRequest, decodeResolvedExecutionRequest, } from "../../execution/resolved-request.js";
|
|
9
|
-
import { cloneToolSelection, isPortableExecutionAgentSelector, } from "../../execution/source.js";
|
|
9
|
+
import { cloneToolSelection, isPortableExecutionAgentSelector, MODEL_WORK_POLICY_ID, } from "../../execution/source.js";
|
|
10
10
|
import { getHarness } from "../harnesses/index.js";
|
|
11
|
-
import {
|
|
11
|
+
import { DEFAULT_LLM_TIMEOUT_MS } from "./config.js";
|
|
12
12
|
import { FALLBACK_ENGINE_NAME, fallbackEngineConfig, NO_ENGINE_MESSAGE_SUFFIX, NO_ENGINE_REMEDY, } from "./engine-fallback.js";
|
|
13
13
|
import { configuredEngine, resolveEngine } from "./engine-resolution.js";
|
|
14
14
|
import { engineModelAndInference, loadModelMap, resolveModelMapAlias } from "./model-map.js";
|
|
@@ -62,21 +62,14 @@ function engineDefaults(name, engine, config) {
|
|
|
62
62
|
if (engine.platform !== "opencode-sdk") {
|
|
63
63
|
return { kind: "agent", platform: engine.platform, modelMapKey: engine.platform, values };
|
|
64
64
|
}
|
|
65
|
-
// An SDK engine runs its
|
|
66
|
-
|
|
65
|
+
// An SDK engine runs its own `llmEngine`'s model/inference/timeout unless it sets its own. With no
|
|
66
|
+
// `llmEngine` it has no fallback, and opencode picks the model: `defaults.llmEngine` is not borrowed.
|
|
67
|
+
const fallbackName = engine.llmEngine;
|
|
67
68
|
const fallback = fallbackName && config.engines && Object.hasOwn(config.engines, fallbackName)
|
|
68
69
|
? config.engines[fallbackName]
|
|
69
70
|
: undefined;
|
|
70
71
|
if (fallback?.kind !== "llm" || !fallbackName) {
|
|
71
|
-
return {
|
|
72
|
-
kind: "sdk",
|
|
73
|
-
platform: "opencode-sdk",
|
|
74
|
-
modelMapKey: "opencode-sdk",
|
|
75
|
-
values: {
|
|
76
|
-
...values,
|
|
77
|
-
timeout: has(values, "timeout") ? values.timeout : DEFAULT_AGENT_TIMEOUT_MS,
|
|
78
|
-
},
|
|
79
|
-
};
|
|
72
|
+
return { kind: "sdk", platform: "opencode-sdk", modelMapKey: "opencode-sdk", values };
|
|
80
73
|
}
|
|
81
74
|
const inherited = engineModelAndInference(fallback);
|
|
82
75
|
return {
|
|
@@ -114,7 +107,10 @@ function runnerDefaults(runner) {
|
|
|
114
107
|
const platform = runner.profile.platform ?? runner.profile.name;
|
|
115
108
|
const fallback = runner.kind === "sdk" ? runner.fallbackConnection : undefined;
|
|
116
109
|
const model = runner.profile.model ?? fallback?.model;
|
|
117
|
-
const
|
|
110
|
+
const fallbackInference = fallback ? inferenceOf(fallback) : undefined;
|
|
111
|
+
const inference = fallbackInference !== undefined || runner.profile.inference !== undefined
|
|
112
|
+
? { ...fallbackInference, ...runner.profile.inference }
|
|
113
|
+
: undefined;
|
|
118
114
|
return {
|
|
119
115
|
kind: runner.kind,
|
|
120
116
|
platform,
|
|
@@ -187,8 +183,19 @@ function requestedToolNames(tools) {
|
|
|
187
183
|
return undefined;
|
|
188
184
|
return Object.keys(policy).filter((tool) => policy[tool] === true);
|
|
189
185
|
}
|
|
190
|
-
/**
|
|
191
|
-
|
|
186
|
+
/**
|
|
187
|
+
* Assets may only narrow the host's `execution.allowedTools`; without a config
|
|
188
|
+
* nothing is allowed. The model-work policy is akm's own and always allowed:
|
|
189
|
+
* it confines an engine more tightly than leaving tools unset does.
|
|
190
|
+
*/
|
|
191
|
+
function authorizeTools(tools, config, modelWork) {
|
|
192
|
+
if (modelWork) {
|
|
193
|
+
return {
|
|
194
|
+
status: "allowed",
|
|
195
|
+
reason: "The model-work tool policy is akm's own.",
|
|
196
|
+
policy: { id: MODEL_WORK_POLICY_ID },
|
|
197
|
+
};
|
|
198
|
+
}
|
|
192
199
|
if (!hasToolSelection(tools))
|
|
193
200
|
return { status: "not-required" };
|
|
194
201
|
if (!config) {
|
|
@@ -239,14 +246,16 @@ function applyRequest(base, request) {
|
|
|
239
246
|
...timeout,
|
|
240
247
|
};
|
|
241
248
|
}
|
|
242
|
-
const { model: _model, workspace: _workspace, ...profile } = base.profile;
|
|
249
|
+
const { model: _model, workspace: _workspace, inference: ownInference, ...profile } = base.profile;
|
|
243
250
|
const workspace = request.runtime.workspace;
|
|
251
|
+
const inference = Object.hasOwn(request, "inference") ? request.inference : ownInference;
|
|
244
252
|
const next = {
|
|
245
253
|
...base,
|
|
246
254
|
profile: {
|
|
247
255
|
...profile,
|
|
248
256
|
...(model !== undefined ? { model } : {}),
|
|
249
257
|
...(typeof workspace === "string" ? { workspace } : {}),
|
|
258
|
+
...(inference ? { inference } : {}),
|
|
250
259
|
},
|
|
251
260
|
...timeout,
|
|
252
261
|
};
|
|
@@ -260,14 +269,29 @@ function applyRequest(base, request) {
|
|
|
260
269
|
}
|
|
261
270
|
return next;
|
|
262
271
|
}
|
|
263
|
-
|
|
272
|
+
/**
|
|
273
|
+
* Reasoning effort has one word in a request: `reasoningEffort`, which is what
|
|
274
|
+
* engines, opencode and the LLM request body call it. `effort` is the same
|
|
275
|
+
* setting as a `models.json` alias or an asset's `effort:` frontmatter spells
|
|
276
|
+
* it, and becomes `reasoningEffort` here, in the one place every layer's
|
|
277
|
+
* inference is merged, so the nearest layer wins whichever word it used. When
|
|
278
|
+
* one inference object has both, `reasoningEffort` wins.
|
|
279
|
+
*/
|
|
280
|
+
function withReasoningEffort(inference) {
|
|
281
|
+
if (!Object.hasOwn(inference, "effort"))
|
|
282
|
+
return inference;
|
|
283
|
+
const { effort, ...rest } = inference;
|
|
284
|
+
return Object.hasOwn(rest, "reasoningEffort") ? rest : { ...rest, reasoningEffort: effort };
|
|
285
|
+
}
|
|
286
|
+
function mergeInference(current, layerInference, source, provenance) {
|
|
264
287
|
provenance["/inference"] = source;
|
|
265
|
-
if (
|
|
288
|
+
if (layerInference === null) {
|
|
266
289
|
for (const key of Object.keys(provenance))
|
|
267
290
|
if (key.startsWith("/inference/"))
|
|
268
291
|
delete provenance[key];
|
|
269
292
|
return null;
|
|
270
293
|
}
|
|
294
|
+
const next = withReasoningEffort(layerInference);
|
|
271
295
|
for (const key of Object.keys(next)) {
|
|
272
296
|
provenance[`/inference/${key.replaceAll("~", "~0").replaceAll("/", "~1")}`] = source;
|
|
273
297
|
}
|
|
@@ -369,13 +393,14 @@ export function resolveExecution(input) {
|
|
|
369
393
|
return layer;
|
|
370
394
|
};
|
|
371
395
|
const schemaLayer = select("outputSchema", "outputSchema");
|
|
372
|
-
const
|
|
396
|
+
const modelWork = input.modelWork === true;
|
|
397
|
+
const toolsLayer = modelWork ? undefined : select("tools", "tools");
|
|
373
398
|
const timeoutLayer = select("timeout", "runtime.timeoutMs");
|
|
374
399
|
const workspaceLayer = select("workspace", "runtime.workspace");
|
|
375
400
|
const environmentLayer = select("environment", "runtime.environment");
|
|
376
401
|
const settingsLayer = select("runtime", "runtime.settings");
|
|
377
402
|
const tools = toolsLayer ? cloneToolSelection(toolsLayer.values.tools ?? null, "tools") : undefined;
|
|
378
|
-
const authorization = authorizeTools(tools, input.runner ? undefined : input.config);
|
|
403
|
+
const authorization = authorizeTools(tools, input.runner ? undefined : input.config, modelWork);
|
|
379
404
|
provenance.authorization = {
|
|
380
405
|
layer: typeof authorization.policy?.id === "string" ? authorization.policy.id : "not-required",
|
|
381
406
|
kind: "authorization",
|
|
@@ -427,8 +452,9 @@ function buildLlm(request, runner, options) {
|
|
|
427
452
|
skip(`inference.${key}`);
|
|
428
453
|
}
|
|
429
454
|
const chatOptions = {};
|
|
455
|
+
// Sent unless the engine opts out; chatCompletion drops it once if the provider rejects it.
|
|
430
456
|
if (request.outputSchema) {
|
|
431
|
-
if (runner.connection.supportsJsonSchema
|
|
457
|
+
if (runner.connection.supportsJsonSchema !== false)
|
|
432
458
|
chatOptions.responseSchema = request.outputSchema;
|
|
433
459
|
else
|
|
434
460
|
skip("outputSchema");
|
|
@@ -4,4 +4,4 @@
|
|
|
4
4
|
export { DEFAULT_AGENT_TIMEOUT_MS } from "./config.js";
|
|
5
5
|
export { _setAgentDetectForTests, defaultWhich, detectAgentCliProfiles, pickDefaultAgentProfile } from "./detect.js";
|
|
6
6
|
export { BUILTIN_AGENT_PROFILE_NAMES, getBuiltinAgentProfile, listBuiltinAgentProfiles, } from "./profiles.js";
|
|
7
|
-
export { buildProposePrompt, buildReflectPrompt,
|
|
7
|
+
export { buildProposePrompt, buildReflectPrompt, parseAgentProposalPayload } from "./prompts.js";
|
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
* coding-agent CLI. Named engines lower canonical harness metadata into this
|
|
9
9
|
* intentionally small internal shape. The wrapper is in `./spawn.ts`.
|
|
10
10
|
*/
|
|
11
|
-
import { COMMON_SPAWN_ENV_PASSTHROUGH } from "../../core/spawn-env.js";
|
|
11
|
+
import { COMMON_SPAWN_ENV_PASSTHROUGH, XDG_BASE_DIR_ENV_PASSTHROUGH } from "../../core/spawn-env.js";
|
|
12
12
|
// AKM_EVENT_SOURCE carries usage-event provenance (improve/task) so that akm
|
|
13
13
|
// invocations a spawned agent makes are recorded as machine traffic, not user
|
|
14
14
|
// demand (DRIFT-6). Without it in the passthrough whitelist, buildChildEnv drops
|
|
@@ -30,7 +30,7 @@ const BUILTINS = {
|
|
|
30
30
|
bin: "opencode",
|
|
31
31
|
args: ["run"],
|
|
32
32
|
stdio: "interactive",
|
|
33
|
-
envPassthrough: [...COMMON_PASSTHROUGH, "OPENCODE_API_KEY", "OPENCODE_CONFIG"],
|
|
33
|
+
envPassthrough: [...COMMON_PASSTHROUGH, "OPENCODE_API_KEY", "OPENCODE_CONFIG", ...XDG_BASE_DIR_ENV_PASSTHROUGH],
|
|
34
34
|
parseOutput: "text",
|
|
35
35
|
},
|
|
36
36
|
claude: {
|
|
@@ -4,15 +4,14 @@
|
|
|
4
4
|
/**
|
|
5
5
|
* Shared prompt builders for proposal-producing agent commands (#226).
|
|
6
6
|
*
|
|
7
|
-
* `akm reflect` and `akm
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
* shell-out prompts targeting an agent CLI, not in-tree LLM API calls.
|
|
7
|
+
* `akm reflect` and `akm proposal new` both ask the configured engine for a
|
|
8
|
+
* structured proposal payload. The prompts are intentionally similar and share
|
|
9
|
+
* their construction here. `proposal new` asks every engine kind for the JSON
|
|
10
|
+
* object below, with {@link PROPOSAL_JSON_SCHEMA} as the request's output
|
|
11
|
+
* schema. Reflect asks every engine kind for its own contract, JSON Schema or
|
|
12
|
+
* framed markdown ({@link ReflectOutputMode}).
|
|
14
13
|
*
|
|
15
|
-
* The
|
|
14
|
+
* The output an engine must produce for `proposal new` is a strict JSON object:
|
|
16
15
|
*
|
|
17
16
|
* ```json
|
|
18
17
|
* {
|
|
@@ -32,6 +31,7 @@ import reflectOutputRepair from "../../assets/prompts/reflect-output-repair.md"
|
|
|
32
31
|
import { placementTypes } from "../../core/asset/asset-placement.js";
|
|
33
32
|
import { parseRefInput } from "../../core/asset/resolve-ref.js";
|
|
34
33
|
import { authoringRulesForType, DESCRIPTION_MAX_CHARS, DESCRIPTION_MIN_CHARS, requiresDescription, } from "../../core/authoring-rules.js";
|
|
34
|
+
import { isRecord } from "../../core/common.js";
|
|
35
35
|
import { parseEmbeddedJsonResponse, stripCodeFences, stripThinkBlocks } from "../../core/parse.js";
|
|
36
36
|
/**
|
|
37
37
|
* Per-asset-type frontmatter / authoring hints surfaced in the prompt so
|
|
@@ -77,10 +77,9 @@ export const REFLECT_CONTENT_CAP = 12_000;
|
|
|
77
77
|
*/
|
|
78
78
|
export const REFLECT_TRUNCATION_MARKER = "... [truncated — focus on the visible portion]";
|
|
79
79
|
/**
|
|
80
|
-
*
|
|
81
|
-
*
|
|
82
|
-
*
|
|
83
|
-
* error.
|
|
80
|
+
* The JSON envelope `proposal new` asks the engine to honour. The wrapper code
|
|
81
|
+
* uses `JSON.parse(stdout)` to extract the payload — anything outside the JSON
|
|
82
|
+
* object will be treated as a parse error.
|
|
84
83
|
*/
|
|
85
84
|
const RESPONSE_CONTRACT_JSON = [
|
|
86
85
|
"Respond ONLY with a single JSON object. No prose before or after.",
|
|
@@ -98,48 +97,25 @@ const RESPONSE_CONTRACT_JSON = [
|
|
|
98
97
|
"Reviewers and the triage judge read this score when adjudicating the proposal queue. Overclaiming erodes trust in your proposals; underclaiming buries good ones. Be honest.",
|
|
99
98
|
].join("\n");
|
|
100
99
|
/**
|
|
101
|
-
*
|
|
102
|
-
*
|
|
103
|
-
*
|
|
100
|
+
* The JSON Schema of {@link RESPONSE_CONTRACT_JSON}'s object, the output
|
|
101
|
+
* schema `proposal new` sends with its request: an LLM engine gets it as
|
|
102
|
+
* `response_format`, codex as `--output-schema`, every agent engine as the
|
|
103
|
+
* shared schema instruction. It is in the strict form those native channels
|
|
104
|
+
* accept (every property required, no others), so it leaves out the optional
|
|
105
|
+
* `frontmatter`, which `content` already carries. The reply is held only to
|
|
106
|
+
* what {@link validateProposalPayload} needs.
|
|
104
107
|
*/
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
"
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
" • Below 0.50 — not confident; prefer not writing changes at all.",
|
|
117
|
-
"Reviewers and the triage judge read this score during adjudication. Overclaim → trust erodes; underclaim → good changes buried.",
|
|
118
|
-
].join("\n");
|
|
119
|
-
}
|
|
120
|
-
/**
|
|
121
|
-
* Extract a confidence score from a `DRAFT_WRITTEN confidence=<n>` line emitted
|
|
122
|
-
* by an agent following {@link fileWriteContract}. Tolerates trailing prose,
|
|
123
|
-
* surrounding log lines, and missing/invalid confidence (returns `undefined`
|
|
124
|
-
* so callers can keep the proposal without a score).
|
|
125
|
-
*
|
|
126
|
-
* Matched forms (case-insensitive, anywhere in stdout):
|
|
127
|
-
* - `DRAFT_WRITTEN confidence=0.85`
|
|
128
|
-
* - `DRAFT_WRITTEN confidence=0.85 ...trailing`
|
|
129
|
-
* - `DRAFT_WRITTEN` (no confidence — returns `undefined`)
|
|
130
|
-
*/
|
|
131
|
-
export function extractDraftConfidence(stdout) {
|
|
132
|
-
if (!stdout)
|
|
133
|
-
return undefined;
|
|
134
|
-
const match = stdout.match(/\bDRAFT_WRITTEN\b[^\S\r\n]+confidence=([0-9]*\.?[0-9]+)/i);
|
|
135
|
-
if (!match)
|
|
136
|
-
return undefined;
|
|
137
|
-
const value = Number.parseFloat(match[1] ?? "");
|
|
138
|
-
if (!Number.isFinite(value) || value < 0 || value > 1)
|
|
139
|
-
return undefined;
|
|
140
|
-
return value;
|
|
141
|
-
}
|
|
142
|
-
export function reflectLlmResponseContract(mode, targetScoped) {
|
|
108
|
+
export const PROPOSAL_JSON_SCHEMA = {
|
|
109
|
+
type: "object",
|
|
110
|
+
required: ["ref", "content", "confidence"],
|
|
111
|
+
additionalProperties: false,
|
|
112
|
+
properties: {
|
|
113
|
+
ref: { type: "string", description: "The new asset's ref as a subdir-qualified conceptId." },
|
|
114
|
+
content: { type: "string", description: "The full file contents that will be written if accepted." },
|
|
115
|
+
confidence: { type: "number", minimum: 0, maximum: 1, description: "Self-rated confidence in [0, 1]." },
|
|
116
|
+
},
|
|
117
|
+
};
|
|
118
|
+
export function reflectResponseContract(mode, targetScoped) {
|
|
143
119
|
if (mode === "json_schema") {
|
|
144
120
|
return reflectLlmSchemaContract
|
|
145
121
|
.replace("{{FIELD_RULE}}", targetScoped
|
|
@@ -155,14 +131,7 @@ export function reflectLlmResponseContract(mode, targetScoped) {
|
|
|
155
131
|
.trim();
|
|
156
132
|
}
|
|
157
133
|
export function buildReflectOutputRepairPrompt(mode, targetScoped) {
|
|
158
|
-
return reflectOutputRepair.replace("{{OUTPUT_CONTRACT}}",
|
|
159
|
-
}
|
|
160
|
-
function reflectResponseContract(input) {
|
|
161
|
-
if (input.draftFilePath)
|
|
162
|
-
return fileWriteContract(input.draftFilePath);
|
|
163
|
-
if (input.outputMode)
|
|
164
|
-
return reflectLlmResponseContract(input.outputMode, input.ref !== undefined);
|
|
165
|
-
return RESPONSE_CONTRACT_JSON;
|
|
134
|
+
return reflectOutputRepair.replace("{{OUTPUT_CONTRACT}}", reflectResponseContract(mode, targetScoped)).trim();
|
|
166
135
|
}
|
|
167
136
|
/**
|
|
168
137
|
* Whether the source asset content has a non-empty `description:` key in its
|
|
@@ -372,7 +341,8 @@ export function buildReflectPrompt(input) {
|
|
|
372
341
|
// Embed concrete counts only when the gate will actually fire (source >= 200 chars).
|
|
373
342
|
const showCharBounds = sourceBodyLen >= 200;
|
|
374
343
|
const minChars = Math.max(Math.round(0.5 * sourceBodyLen), 150);
|
|
375
|
-
|
|
344
|
+
// A source already past the 25000 cap may not grow, and need not shrink to the cap.
|
|
345
|
+
const maxChars = Math.max(Math.min(Math.max(Math.round(2.5 * sourceBodyLen), 2500), 25000), sourceBodyLen);
|
|
376
346
|
sections.push([
|
|
377
347
|
"## Content preservation rules (MUST follow)",
|
|
378
348
|
"1. PRESERVE ALL concrete content: code blocks, fenced snippets, CLI commands, numbered/bulleted checklists, tables, YAML/JSON examples, file paths, configuration keys, environment variable names, and CSS/HTML selectors. These are load-bearing — do NOT replace them with prose summaries.",
|
|
@@ -381,22 +351,18 @@ export function buildReflectPrompt(input) {
|
|
|
381
351
|
? `3. DO NOT shrink the asset. Your body must be at least ${minChars} characters (source body is ${sourceBodyLen} chars; floor is 50%). If you genuinely need to remove a major section, explain why in a comment line at the top of the body (e.g. \`<!-- removed obsolete section X because ... -->\`).`
|
|
382
352
|
: "3. DO NOT shrink the asset dramatically. The improved body must be at least 50% of the source body length. If you genuinely need to remove a major section, explain why in a comment line at the top of the body (e.g. `<!-- removed obsolete section X because ... -->`).",
|
|
383
353
|
showCharBounds
|
|
384
|
-
? `4. DO NOT pad the asset with speculative material. Your body must be at most ${maxChars} characters (source body is ${sourceBodyLen} chars; ceiling is 250%). Do not add invented sections, hypothetical examples, or padding prose.`
|
|
354
|
+
? `4. DO NOT pad the asset with speculative material. Your body must be at most ${maxChars} characters (source body is ${sourceBodyLen} chars; ceiling is ${maxChars === sourceBodyLen ? "100%" : "250%"}). Do not add invented sections, hypothetical examples, or padding prose.`
|
|
385
355
|
: "4. DO NOT pad the asset with speculative material. The improved body must be at most 250% of the source body length unless the feedback explicitly requests added sections.",
|
|
386
356
|
"5. Improve clarity of surrounding prose, fix structural issues, add missing required frontmatter fields. Do NOT rewrite a runbook into an essay.",
|
|
387
357
|
].join("\n"));
|
|
388
358
|
}
|
|
389
|
-
|
|
390
|
-
// Reinforce that the `ref` field is mandatory and must exactly match the target.
|
|
391
|
-
// Small models frequently omit `ref` from the response JSON, causing parse errors.
|
|
392
|
-
sections.push(`IMPORTANT: The JSON "ref" field is REQUIRED. It MUST be exactly: "${input.ref}"`);
|
|
393
|
-
}
|
|
394
|
-
sections.push(reflectResponseContract(input));
|
|
359
|
+
sections.push(reflectResponseContract(input.outputMode ?? "json_schema", input.ref !== undefined));
|
|
395
360
|
return { prompt: sections.join("\n\n") };
|
|
396
361
|
}
|
|
397
362
|
/**
|
|
398
|
-
* Build the prompt for `akm
|
|
399
|
-
*
|
|
363
|
+
* Build the prompt for `akm proposal new <type> <name> --task ...`. Asks the
|
|
364
|
+
* engine to author a brand-new asset of the given type fulfilling `task`, and
|
|
365
|
+
* to return it as the JSON object of {@link RESPONSE_CONTRACT_JSON}.
|
|
400
366
|
*/
|
|
401
367
|
export function buildProposePrompt(input) {
|
|
402
368
|
const sections = [];
|
|
@@ -420,44 +386,7 @@ export function buildProposePrompt(input) {
|
|
|
420
386
|
}
|
|
421
387
|
}
|
|
422
388
|
sections.push("Produce a single proposal that, if accepted, would land as the asset described above.");
|
|
423
|
-
sections.push(
|
|
424
|
-
return sections.join("\n\n");
|
|
425
|
-
}
|
|
426
|
-
/**
|
|
427
|
-
* Build the prompt for the schema repair pass in `akm improve`. Asks the
|
|
428
|
-
* agent to add the minimal required frontmatter to an asset that failed
|
|
429
|
-
* validation — without rewriting the body.
|
|
430
|
-
*/
|
|
431
|
-
export function buildSchemaRepairPrompt(input) {
|
|
432
|
-
const sections = [];
|
|
433
|
-
sections.push(`This ${input.type} asset failed schema validation with the error: "${input.reason}". ` +
|
|
434
|
-
`Your task is to fix the schema issue by adding or correcting the missing/invalid field(s) ` +
|
|
435
|
-
`while preserving all existing content.`);
|
|
436
|
-
sections.push(`Target ref: ${input.ref}`);
|
|
437
|
-
sections.push(`Schema requirements for ${input.type} assets: ${hintForType(input.type)}`);
|
|
438
|
-
if (input.standardsContext?.trim()) {
|
|
439
|
-
sections.push("Standards to follow (the rulebook for this target):");
|
|
440
|
-
sections.push(input.standardsContext.trim());
|
|
441
|
-
}
|
|
442
|
-
{
|
|
443
|
-
const authoringRules = authoringRulesForType(input.type);
|
|
444
|
-
if (authoringRules) {
|
|
445
|
-
sections.push(authoringRules);
|
|
446
|
-
}
|
|
447
|
-
}
|
|
448
|
-
const CONTENT_CAP = 3000;
|
|
449
|
-
const body = input.assetContent.trimEnd();
|
|
450
|
-
const truncated = body.length > CONTENT_CAP;
|
|
451
|
-
sections.push("Current asset content (first 3000 chars — sufficient to generate missing frontmatter):");
|
|
452
|
-
sections.push("```");
|
|
453
|
-
sections.push(truncated ? `${body.slice(0, CONTENT_CAP)}\n... [truncated]` : body);
|
|
454
|
-
sections.push("```");
|
|
455
|
-
sections.push("Produce the minimal fix: add ONLY the missing required frontmatter field(s). " +
|
|
456
|
-
"Do not rewrite the body unless it is empty. " +
|
|
457
|
-
"If `description` is missing, generate a concise one-sentence description from the content. " +
|
|
458
|
-
"If `when_to_use` is missing, generate a one-line trigger sentence. " +
|
|
459
|
-
"Preserve all existing frontmatter keys and the full body verbatim.");
|
|
460
|
-
sections.push(input.draftFilePath ? fileWriteContract(input.draftFilePath) : RESPONSE_CONTRACT_JSON);
|
|
389
|
+
sections.push(RESPONSE_CONTRACT_JSON);
|
|
461
390
|
return sections.join("\n\n");
|
|
462
391
|
}
|
|
463
392
|
/**
|
|
@@ -487,19 +416,31 @@ export function parseAgentProposalPayload(stdout) {
|
|
|
487
416
|
throw directErr;
|
|
488
417
|
parsed = embedded;
|
|
489
418
|
}
|
|
419
|
+
const verdict = validateProposalPayload(parsed);
|
|
420
|
+
if (!verdict.ok)
|
|
421
|
+
throw new Error(verdict.errors.join("; "));
|
|
422
|
+
return verdict.value;
|
|
423
|
+
}
|
|
424
|
+
/**
|
|
425
|
+
* The proposal in a parsed reply, or why the reply is not one: `ref` and
|
|
426
|
+
* `content` must be non-empty strings. A malformed optional field is dropped,
|
|
427
|
+
* never refused.
|
|
428
|
+
*/
|
|
429
|
+
export function validateProposalPayload(parsed) {
|
|
430
|
+
if (!isRecord(parsed))
|
|
431
|
+
return { ok: false, errors: ["agent response is not a JSON object"] };
|
|
490
432
|
if (typeof parsed.ref !== "string" || !parsed.ref.trim()) {
|
|
491
|
-
|
|
433
|
+
return { ok: false, errors: ['agent response missing required string field "ref"'] };
|
|
492
434
|
}
|
|
493
435
|
if (typeof parsed.content !== "string" || !parsed.content.trim()) {
|
|
494
|
-
|
|
436
|
+
return { ok: false, errors: ['agent response missing required string field "content"'] };
|
|
495
437
|
}
|
|
496
438
|
const out = {
|
|
497
439
|
ref: parsed.ref.trim(),
|
|
498
440
|
content: parsed.content,
|
|
499
441
|
};
|
|
500
|
-
if (
|
|
442
|
+
if (isRecord(parsed.frontmatter))
|
|
501
443
|
out.frontmatter = parsed.frontmatter;
|
|
502
|
-
}
|
|
503
444
|
// Phase 6A: extract optional `confidence` (number in [0, 1]). Clamp gently
|
|
504
445
|
// rather than reject — a model that returns 1.0 or 0 with extra precision
|
|
505
446
|
// (e.g. 1.0000001) should still surface a usable score. Anything that isn't
|
|
@@ -509,5 +450,5 @@ export function parseAgentProposalPayload(stdout) {
|
|
|
509
450
|
const clamped = Math.max(0, Math.min(1, parsed.confidence));
|
|
510
451
|
out.confidence = clamped;
|
|
511
452
|
}
|
|
512
|
-
return out;
|
|
453
|
+
return { ok: true, value: out };
|
|
513
454
|
}
|
|
@@ -2,6 +2,9 @@
|
|
|
2
2
|
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
3
|
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
4
|
import { ConfigError } from "../../core/errors.js";
|
|
5
|
+
import { withSchemaInstruction } from "../../core/structured.js";
|
|
6
|
+
import { MODEL_WORK_POLICY_ID } from "../../execution/source.js";
|
|
7
|
+
import { HARNESS_MODEL_WORK_IDS } from "../harnesses/ids.js";
|
|
5
8
|
import { composeConversationFallbackPrompt } from "./conversation-fallback.js";
|
|
6
9
|
import { composePersonaFallbackPrompt } from "./persona-fallback.js";
|
|
7
10
|
/** A tool selection that actually names tools (not omitted, null, or empty). */
|
|
@@ -47,6 +50,7 @@ export function extensionFields(request) {
|
|
|
47
50
|
export function createAgentRequestLowerer(options) {
|
|
48
51
|
const supportedInference = new Set(options.inference ?? []);
|
|
49
52
|
return (_profile, request) => {
|
|
53
|
+
const modelWork = request.authorization.policy?.id === MODEL_WORK_POLICY_ID;
|
|
50
54
|
const notices = [];
|
|
51
55
|
const skip = (field) => {
|
|
52
56
|
notices.push(untranslated(options.adapter, field));
|
|
@@ -78,24 +82,33 @@ export function createAgentRequestLowerer(options) {
|
|
|
78
82
|
}
|
|
79
83
|
dispatch.agent = request.agent;
|
|
80
84
|
}
|
|
85
|
+
// Every agent transport gets the schema as the one instruction; a harness
|
|
86
|
+
// with a native channel (codex --output-schema) also reads dispatch.schema.
|
|
87
|
+
if (request.outputSchema) {
|
|
88
|
+
prompt = withSchemaInstruction(prompt, request.outputSchema);
|
|
89
|
+
dispatch.schema = request.outputSchema;
|
|
90
|
+
}
|
|
81
91
|
dispatch.prompt = prompt;
|
|
82
92
|
if (request.model)
|
|
83
93
|
dispatch.model = request.model.resolved;
|
|
84
94
|
if (Object.hasOwn(request, "inference")) {
|
|
85
95
|
dispatch.inference = request.inference ?? null;
|
|
86
96
|
for (const key of Object.keys(request.inference ?? {}).sort()) {
|
|
87
|
-
if (!supportedInference.has(key))
|
|
97
|
+
if (!(modelWork && supportedInference.has(key)))
|
|
88
98
|
skip(`inference.${key}`);
|
|
89
99
|
}
|
|
90
|
-
if (typeof request.inference?.effort === "string")
|
|
91
|
-
dispatch.effort = request.inference.effort;
|
|
92
100
|
}
|
|
93
|
-
if (
|
|
94
|
-
if (!options.
|
|
95
|
-
|
|
96
|
-
|
|
101
|
+
if (modelWork) {
|
|
102
|
+
if (!HARNESS_MODEL_WORK_IDS.has(options.adapter)) {
|
|
103
|
+
throw new ConfigError(`The ${options.adapter} transport cannot enforce the model-work tool policy.`, "INVALID_CONFIG_FILE");
|
|
104
|
+
}
|
|
105
|
+
// The builder selects its own confined agent or flags; another agent would replace them.
|
|
106
|
+
if (dispatch.agent) {
|
|
107
|
+
throw new ConfigError(`The ${options.adapter} transport cannot run native agent ${JSON.stringify(dispatch.agent)} under the model-work tool policy.`, "INVALID_CONFIG_FILE");
|
|
108
|
+
}
|
|
109
|
+
dispatch.modelWork = true;
|
|
97
110
|
}
|
|
98
|
-
if (request.tools !== undefined) {
|
|
111
|
+
else if (request.tools !== undefined) {
|
|
99
112
|
// An explicit empty selection still reaches the builder (e.g. an empty allowlist).
|
|
100
113
|
if (hasToolSelection(request.tools) && !translatesTools(options.tools, request.tools)) {
|
|
101
114
|
throw new ConfigError(`The ${options.adapter} transport cannot enforce the resolved tool policy.`, "INVALID_CONFIG_FILE");
|
|
@@ -8,11 +8,18 @@
|
|
|
8
8
|
* credential and passthrough value that could reach the child is redacted
|
|
9
9
|
* from the result.
|
|
10
10
|
*/
|
|
11
|
+
import fs from "node:fs";
|
|
12
|
+
import os from "node:os";
|
|
13
|
+
import path from "node:path";
|
|
11
14
|
import { assertNever } from "../../core/assert.js";
|
|
12
15
|
import { UsageError } from "../../core/errors.js";
|
|
13
16
|
import { collectSensitiveValues, isEnvPassthroughValueSafeToExpose, redactSensitiveText, redactSensitiveValue, } from "../../core/redaction.js";
|
|
17
|
+
import { MODEL_WORK_POLICY_ID } from "../../execution/source.js";
|
|
14
18
|
import { chatCompletion, LlmCallError } from "../../llm/client.js";
|
|
19
|
+
import { emitLlmUsage } from "../../llm/usage-telemetry.js";
|
|
20
|
+
import { getHarness } from "../harnesses/index.js";
|
|
15
21
|
import { closeServer as disposeOpencodeSdkServers, runOpencodeSdk } from "../harnesses/opencode-sdk/sdk-runner.js";
|
|
22
|
+
import { modelFromArgs } from "./builder-shared.js";
|
|
16
23
|
import { lookupApiKeyFileValue, lookupApiKeySecretRefValue, lookupCredentialFromEnv, resolveEngine, } from "./engine-resolution.js";
|
|
17
24
|
import { materializeLlmRunnerConnection, materializeSdkFallbackConnection } from "./runner.js";
|
|
18
25
|
import { runAgent } from "./spawn.js";
|
|
@@ -74,6 +81,42 @@ function llmFailureReason(error) {
|
|
|
74
81
|
return "spawn_failed";
|
|
75
82
|
}
|
|
76
83
|
}
|
|
84
|
+
const USAGE_ERROR_CODES = {
|
|
85
|
+
timeout: "timeout",
|
|
86
|
+
aborted: "aborted",
|
|
87
|
+
parse_error: "parse_error",
|
|
88
|
+
llm_rate_limit: "rate_limited",
|
|
89
|
+
};
|
|
90
|
+
/**
|
|
91
|
+
* The model an agent or SDK dispatch ran: the request's, else the one an agent
|
|
92
|
+
* CLI's own `args` select, which its command carries when the request names
|
|
93
|
+
* none. An SDK server never sees `args`, so an SDK engine with no model named
|
|
94
|
+
* has none to report: opencode picks it.
|
|
95
|
+
*/
|
|
96
|
+
function dispatchedModel({ request, runner }) {
|
|
97
|
+
return request.model?.resolved ?? (runner.kind === "agent" ? modelFromArgs(runner.profile.args) : undefined);
|
|
98
|
+
}
|
|
99
|
+
/**
|
|
100
|
+
* One usage record for an agent or SDK dispatch, through the same sink and
|
|
101
|
+
* ambient `withLlmStage` attribution as the LLM transport's per-HTTP-attempt
|
|
102
|
+
* records: the model it ran, and tokens when the runner reported them.
|
|
103
|
+
*/
|
|
104
|
+
function recordDispatchUsage(execution, result) {
|
|
105
|
+
const { inputTokens, outputTokens, reasoningTokens } = result.usage ?? {};
|
|
106
|
+
const reported = [inputTokens, outputTokens, reasoningTokens].filter((count) => count !== undefined);
|
|
107
|
+
const model = dispatchedModel(execution);
|
|
108
|
+
emitLlmUsage({
|
|
109
|
+
outcome: result.ok ? "success" : "error",
|
|
110
|
+
modelSource: "configured",
|
|
111
|
+
...(model ? { model } : {}),
|
|
112
|
+
durationMs: result.durationMs,
|
|
113
|
+
...(inputTokens !== undefined ? { promptTokens: inputTokens } : {}),
|
|
114
|
+
...(outputTokens !== undefined ? { completionTokens: outputTokens } : {}),
|
|
115
|
+
...(reasoningTokens !== undefined ? { reasoningTokens } : {}),
|
|
116
|
+
...(reported.length > 0 ? { totalTokens: reported.reduce((sum, count) => sum + count, 0) } : {}),
|
|
117
|
+
...(result.ok ? {} : { errorCode: (result.reason && USAGE_ERROR_CODES[result.reason]) ?? "unknown_error" }),
|
|
118
|
+
});
|
|
119
|
+
}
|
|
77
120
|
async function dispatchRunner(runner, prompt, opts, seams, llm) {
|
|
78
121
|
const envSource = opts.envSource ?? process.env;
|
|
79
122
|
const secrets = collectDispatchSensitiveValues(runner, opts, envSource);
|
|
@@ -103,9 +146,53 @@ async function dispatchRunner(runner, prompt, opts, seams, llm) {
|
|
|
103
146
|
}
|
|
104
147
|
return redactResult(result, collectSensitiveValues(secrets));
|
|
105
148
|
}
|
|
106
|
-
/**
|
|
149
|
+
/**
|
|
150
|
+
* The scratch working directory for one model-work dispatch on an agent or SDK
|
|
151
|
+
* engine, so the edit the model-work tool policy grants never reaches the
|
|
152
|
+
* stash.
|
|
153
|
+
*/
|
|
154
|
+
function createModelWorkDirectory() {
|
|
155
|
+
return fs.mkdtempSync(path.join(os.tmpdir(), "akm-model-work-"));
|
|
156
|
+
}
|
|
157
|
+
/**
|
|
158
|
+
* Run a built execution. Credentials are read here, once per call, and never
|
|
159
|
+
* returned. Model work on an agent or SDK engine runs in a scratch working
|
|
160
|
+
* directory that akm creates for the dispatch and removes after it, and must
|
|
161
|
+
* end with an answer: an agent that stops with none (opencode at its step
|
|
162
|
+
* limit, for one) has failed with `parse_error`.
|
|
163
|
+
*/
|
|
107
164
|
export async function runExecution(execution, options = {}) {
|
|
108
|
-
const
|
|
165
|
+
const scratch = execution.runner.kind !== "llm" && execution.request.authorization.policy?.id === MODEL_WORK_POLICY_ID
|
|
166
|
+
? createModelWorkDirectory()
|
|
167
|
+
: undefined;
|
|
168
|
+
try {
|
|
169
|
+
return await runBuiltExecution(execution, options, scratch);
|
|
170
|
+
}
|
|
171
|
+
finally {
|
|
172
|
+
if (scratch)
|
|
173
|
+
fs.rmSync(scratch, { recursive: true, force: true });
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
/** A successful reply's text, unwrapped from its harness's framing (claude's `--output-format json` result envelope, for one). */
|
|
177
|
+
export function unwrapHarnessReply(runner, result) {
|
|
178
|
+
const extractor = runner.kind === "agent" ? getHarness(runner.profile.platform ?? runner.profile.name)?.resultExtractor : undefined;
|
|
179
|
+
return extractor ? extractor(result) : { text: result.stdout };
|
|
180
|
+
}
|
|
181
|
+
/** A model-work reply's answer: its harness's framing stripped, and no answer is a `parse_error`. */
|
|
182
|
+
function modelWorkAnswer(runner, result) {
|
|
183
|
+
const extracted = unwrapHarnessReply(runner, result);
|
|
184
|
+
const answer = {
|
|
185
|
+
...result,
|
|
186
|
+
stdout: extracted.text,
|
|
187
|
+
...(extracted.sessionId ? { sessionId: extracted.sessionId } : {}),
|
|
188
|
+
};
|
|
189
|
+
if (extracted.text.trim() !== "")
|
|
190
|
+
return answer;
|
|
191
|
+
return { ...answer, ok: false, reason: "parse_error", error: `Engine "${runner.engine}" returned no answer.` };
|
|
192
|
+
}
|
|
193
|
+
/** `scratch` is the model-work working directory, set only for model work on an agent or SDK engine. */
|
|
194
|
+
async function runBuiltExecution(execution, options, scratch) {
|
|
195
|
+
const opts = { ...execution.options, ...(scratch ? { cwd: scratch } : {}) };
|
|
109
196
|
const operational = options.runOptions ?? {};
|
|
110
197
|
for (const key of OPERATIONAL_OPTIONS) {
|
|
111
198
|
if (operational[key] !== undefined)
|
|
@@ -140,7 +227,13 @@ export async function runExecution(execution, options = {}) {
|
|
|
140
227
|
};
|
|
141
228
|
}
|
|
142
229
|
};
|
|
143
|
-
|
|
230
|
+
let result = await dispatchRunner(execution.runner, execution.prompt, opts, options, llm);
|
|
231
|
+
if (scratch !== undefined && result.ok)
|
|
232
|
+
result = modelWorkAnswer(execution.runner, result);
|
|
233
|
+
// The LLM transport records each HTTP attempt itself.
|
|
234
|
+
if (execution.runner.kind !== "llm")
|
|
235
|
+
recordDispatchUsage(execution, result);
|
|
236
|
+
return result;
|
|
144
237
|
}
|
|
145
238
|
/** `akm agent`: launch an agent engine interactively, with no prompt of its own. */
|
|
146
239
|
export async function executeInteractiveAgentInvocation(input, seams = {}) {
|
|
@@ -67,6 +67,12 @@ export function decodeFrozenRunnerSpec(value) {
|
|
|
67
67
|
export function runnerIsLlm(runner) {
|
|
68
68
|
return runner.kind === "llm";
|
|
69
69
|
}
|
|
70
|
-
|
|
71
|
-
|
|
70
|
+
/**
|
|
71
|
+
* The LLM connection behind a runner, whatever its kind: an llm runner's own,
|
|
72
|
+
* an sdk runner's provider fallback, none for an agent CLI.
|
|
73
|
+
*/
|
|
74
|
+
export function runnerLlmConnection(runner) {
|
|
75
|
+
if (runner.kind === "llm")
|
|
76
|
+
return runner.connection;
|
|
77
|
+
return runner.kind === "sdk" ? runner.fallbackConnection : undefined;
|
|
72
78
|
}
|