akm-cli 0.9.24 → 0.9.25-alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/CHANGELOG.md +173 -0
  2. package/dist/cli.js +1 -1
  3. package/dist/commands/health/checks.js +10 -11
  4. package/dist/commands/improve/consolidate/pair-pass.js +4 -2
  5. package/dist/commands/improve/consolidate.js +10 -4
  6. package/dist/commands/improve/execution.js +4 -11
  7. package/dist/commands/improve/extract-prompt.js +4 -4
  8. package/dist/commands/improve/extract.js +11 -13
  9. package/dist/commands/improve/improve-cli.js +65 -34
  10. package/dist/commands/improve/improve-strategies.js +49 -43
  11. package/dist/commands/improve/improve-usage-report.js +8 -17
  12. package/dist/commands/improve/loop-stages.js +3 -0
  13. package/dist/commands/improve/preparation.js +3 -1
  14. package/dist/commands/improve/reflect-noise.js +125 -0
  15. package/dist/commands/improve/reflect.js +105 -172
  16. package/dist/commands/improve/retrieval-gate.js +7 -2
  17. package/dist/commands/improve/stage.js +67 -24
  18. package/dist/commands/proposal/drain.js +11 -2
  19. package/dist/commands/proposal/proposal-cli.js +1 -5
  20. package/dist/commands/proposal/propose-cli.js +2 -2
  21. package/dist/commands/proposal/propose.js +72 -84
  22. package/dist/commands/proposal/validators/proposal-quality-validators.js +4 -2
  23. package/dist/commands/read/search-cli.js +0 -38
  24. package/dist/commands/remember.js +3 -3
  25. package/dist/commands/sources/schema-repair.js +1 -1
  26. package/dist/core/config/config-schema.js +37 -60
  27. package/dist/core/config/engine-semantics.js +15 -11
  28. package/dist/core/config/schema/improve-processes.js +18 -2
  29. package/dist/core/improve-result.js +3 -3
  30. package/dist/core/redaction.js +4 -0
  31. package/dist/core/spawn-env.js +25 -0
  32. package/dist/core/structured.js +11 -1
  33. package/dist/execution/source.js +10 -0
  34. package/dist/indexer/passes/memory-inference.js +2 -1
  35. package/dist/integrations/agent/builder-shared.js +15 -0
  36. package/dist/integrations/agent/config.js +1 -1
  37. package/dist/integrations/agent/engine-resolution.js +13 -31
  38. package/dist/integrations/agent/execution.js +48 -22
  39. package/dist/integrations/agent/index.js +1 -1
  40. package/dist/integrations/agent/profiles.js +2 -2
  41. package/dist/integrations/agent/prompts.js +55 -114
  42. package/dist/integrations/agent/request-lowering.js +21 -8
  43. package/dist/integrations/agent/runner-dispatch.js +96 -3
  44. package/dist/integrations/agent/runner.js +8 -2
  45. package/dist/integrations/harnesses/aider/agent-builder.js +10 -23
  46. package/dist/integrations/harnesses/aider/index.js +0 -5
  47. package/dist/integrations/harnesses/amazonq/agent-builder.js +9 -21
  48. package/dist/integrations/harnesses/amazonq/index.js +0 -5
  49. package/dist/integrations/harnesses/claude/agent-builder.js +47 -36
  50. package/dist/integrations/harnesses/claude/index.js +0 -14
  51. package/dist/integrations/harnesses/codex/agent-builder.js +3 -4
  52. package/dist/integrations/harnesses/codex/index.js +0 -4
  53. package/dist/integrations/harnesses/copilot/agent-builder.js +4 -13
  54. package/dist/integrations/harnesses/copilot/index.js +2 -7
  55. package/dist/integrations/harnesses/gemini/agent-builder.js +4 -13
  56. package/dist/integrations/harnesses/gemini/index.js +0 -5
  57. package/dist/integrations/harnesses/ids.js +12 -10
  58. package/dist/integrations/harnesses/opencode/agent-builder.js +41 -9
  59. package/dist/integrations/harnesses/opencode/index.js +0 -8
  60. package/dist/integrations/harnesses/opencode/model-config.js +33 -0
  61. package/dist/integrations/harnesses/opencode/model-work-agent.js +115 -0
  62. package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -7
  63. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +114 -28
  64. package/dist/integrations/harnesses/openhands/agent-builder.js +7 -21
  65. package/dist/integrations/harnesses/openhands/index.js +0 -5
  66. package/dist/integrations/harnesses/pi/agent-builder.js +7 -21
  67. package/dist/integrations/harnesses/pi/index.js +0 -5
  68. package/dist/llm/client.js +5 -0
  69. package/dist/llm/feature-gate.js +12 -9
  70. package/dist/llm/index-passes.js +2 -5
  71. package/dist/llm/memory-infer.js +6 -5
  72. package/dist/llm/structured-call.js +33 -11
  73. package/dist/output/shapes/passthrough.js +1 -0
  74. package/dist/scripts/akm-migrate-node.js +298 -228
  75. package/dist/scripts/akm-migrate.js +298 -228
  76. package/dist/workflows/exec/step-work.js +6 -5
  77. package/dist/workflows/exec/unit-dispatch.js +4 -13
  78. package/dist/workflows/freeze/step-values.js +1 -1
  79. package/docs/reference/cli.md +41 -18
  80. package/docs/reference/configuration.md +165 -12
  81. package/docs/reference/data-and-telemetry.md +2 -3
  82. package/docs/reference/workflow-schema.md +10 -9
  83. package/package.json +1 -1
  84. package/schemas/akm-config.json +108 -0
@@ -6,9 +6,9 @@ import { ConfigError } from "../../core/errors.js";
6
6
  import { DURATION_UNITS, parseDuration } from "../../core/time.js";
7
7
  import { EXECUTION_MAX_TIMEOUT_MS } from "../../execution/limits.js";
8
8
  import { createInlineResolvedCommand, createResolvedExecutionRequest, decodeResolvedExecutionRequest, } from "../../execution/resolved-request.js";
9
- import { cloneToolSelection, isPortableExecutionAgentSelector, } from "../../execution/source.js";
9
+ import { cloneToolSelection, isPortableExecutionAgentSelector, MODEL_WORK_POLICY_ID, } from "../../execution/source.js";
10
10
  import { getHarness } from "../harnesses/index.js";
11
- import { DEFAULT_AGENT_TIMEOUT_MS, DEFAULT_LLM_TIMEOUT_MS } from "./config.js";
11
+ import { DEFAULT_LLM_TIMEOUT_MS } from "./config.js";
12
12
  import { FALLBACK_ENGINE_NAME, fallbackEngineConfig, NO_ENGINE_MESSAGE_SUFFIX, NO_ENGINE_REMEDY, } from "./engine-fallback.js";
13
13
  import { configuredEngine, resolveEngine } from "./engine-resolution.js";
14
14
  import { engineModelAndInference, loadModelMap, resolveModelMapAlias } from "./model-map.js";
@@ -62,21 +62,14 @@ function engineDefaults(name, engine, config) {
62
62
  if (engine.platform !== "opencode-sdk") {
63
63
  return { kind: "agent", platform: engine.platform, modelMapKey: engine.platform, values };
64
64
  }
65
- // An SDK engine runs its LLM fallback's model/inference/timeout unless it sets its own.
66
- const fallbackName = engine.llmEngine ?? config.defaults?.llmEngine;
65
+ // An SDK engine runs its own `llmEngine`'s model/inference/timeout unless it sets its own. With no
66
+ // `llmEngine` it has no fallback, and opencode picks the model: `defaults.llmEngine` is not borrowed.
67
+ const fallbackName = engine.llmEngine;
67
68
  const fallback = fallbackName && config.engines && Object.hasOwn(config.engines, fallbackName)
68
69
  ? config.engines[fallbackName]
69
70
  : undefined;
70
71
  if (fallback?.kind !== "llm" || !fallbackName) {
71
- return {
72
- kind: "sdk",
73
- platform: "opencode-sdk",
74
- modelMapKey: "opencode-sdk",
75
- values: {
76
- ...values,
77
- timeout: has(values, "timeout") ? values.timeout : DEFAULT_AGENT_TIMEOUT_MS,
78
- },
79
- };
72
+ return { kind: "sdk", platform: "opencode-sdk", modelMapKey: "opencode-sdk", values };
80
73
  }
81
74
  const inherited = engineModelAndInference(fallback);
82
75
  return {
@@ -114,7 +107,10 @@ function runnerDefaults(runner) {
114
107
  const platform = runner.profile.platform ?? runner.profile.name;
115
108
  const fallback = runner.kind === "sdk" ? runner.fallbackConnection : undefined;
116
109
  const model = runner.profile.model ?? fallback?.model;
117
- const inference = fallback ? inferenceOf(fallback) : undefined;
110
+ const fallbackInference = fallback ? inferenceOf(fallback) : undefined;
111
+ const inference = fallbackInference !== undefined || runner.profile.inference !== undefined
112
+ ? { ...fallbackInference, ...runner.profile.inference }
113
+ : undefined;
118
114
  return {
119
115
  kind: runner.kind,
120
116
  platform,
@@ -187,8 +183,19 @@ function requestedToolNames(tools) {
187
183
  return undefined;
188
184
  return Object.keys(policy).filter((tool) => policy[tool] === true);
189
185
  }
190
- /** Assets may only narrow the host's `execution.allowedTools`; without a config nothing is allowed. */
191
- function authorizeTools(tools, config) {
186
+ /**
187
+ * Assets may only narrow the host's `execution.allowedTools`; without a config
188
+ * nothing is allowed. The model-work policy is akm's own and always allowed:
189
+ * it confines an engine more tightly than leaving tools unset does.
190
+ */
191
+ function authorizeTools(tools, config, modelWork) {
192
+ if (modelWork) {
193
+ return {
194
+ status: "allowed",
195
+ reason: "The model-work tool policy is akm's own.",
196
+ policy: { id: MODEL_WORK_POLICY_ID },
197
+ };
198
+ }
192
199
  if (!hasToolSelection(tools))
193
200
  return { status: "not-required" };
194
201
  if (!config) {
@@ -239,14 +246,16 @@ function applyRequest(base, request) {
239
246
  ...timeout,
240
247
  };
241
248
  }
242
- const { model: _model, workspace: _workspace, ...profile } = base.profile;
249
+ const { model: _model, workspace: _workspace, inference: ownInference, ...profile } = base.profile;
243
250
  const workspace = request.runtime.workspace;
251
+ const inference = Object.hasOwn(request, "inference") ? request.inference : ownInference;
244
252
  const next = {
245
253
  ...base,
246
254
  profile: {
247
255
  ...profile,
248
256
  ...(model !== undefined ? { model } : {}),
249
257
  ...(typeof workspace === "string" ? { workspace } : {}),
258
+ ...(inference ? { inference } : {}),
250
259
  },
251
260
  ...timeout,
252
261
  };
@@ -260,14 +269,29 @@ function applyRequest(base, request) {
260
269
  }
261
270
  return next;
262
271
  }
263
- function mergeInference(current, next, source, provenance) {
272
+ /**
273
+ * Reasoning effort has one word in a request: `reasoningEffort`, which is what
274
+ * engines, opencode and the LLM request body call it. `effort` is the same
275
+ * setting as a `models.json` alias or an asset's `effort:` frontmatter spells
276
+ * it, and becomes `reasoningEffort` here, in the one place every layer's
277
+ * inference is merged, so the nearest layer wins whichever word it used. When
278
+ * one inference object has both, `reasoningEffort` wins.
279
+ */
280
+ function withReasoningEffort(inference) {
281
+ if (!Object.hasOwn(inference, "effort"))
282
+ return inference;
283
+ const { effort, ...rest } = inference;
284
+ return Object.hasOwn(rest, "reasoningEffort") ? rest : { ...rest, reasoningEffort: effort };
285
+ }
286
+ function mergeInference(current, layerInference, source, provenance) {
264
287
  provenance["/inference"] = source;
265
- if (next === null) {
288
+ if (layerInference === null) {
266
289
  for (const key of Object.keys(provenance))
267
290
  if (key.startsWith("/inference/"))
268
291
  delete provenance[key];
269
292
  return null;
270
293
  }
294
+ const next = withReasoningEffort(layerInference);
271
295
  for (const key of Object.keys(next)) {
272
296
  provenance[`/inference/${key.replaceAll("~", "~0").replaceAll("/", "~1")}`] = source;
273
297
  }
@@ -369,13 +393,14 @@ export function resolveExecution(input) {
369
393
  return layer;
370
394
  };
371
395
  const schemaLayer = select("outputSchema", "outputSchema");
372
- const toolsLayer = select("tools", "tools");
396
+ const modelWork = input.modelWork === true;
397
+ const toolsLayer = modelWork ? undefined : select("tools", "tools");
373
398
  const timeoutLayer = select("timeout", "runtime.timeoutMs");
374
399
  const workspaceLayer = select("workspace", "runtime.workspace");
375
400
  const environmentLayer = select("environment", "runtime.environment");
376
401
  const settingsLayer = select("runtime", "runtime.settings");
377
402
  const tools = toolsLayer ? cloneToolSelection(toolsLayer.values.tools ?? null, "tools") : undefined;
378
- const authorization = authorizeTools(tools, input.runner ? undefined : input.config);
403
+ const authorization = authorizeTools(tools, input.runner ? undefined : input.config, modelWork);
379
404
  provenance.authorization = {
380
405
  layer: typeof authorization.policy?.id === "string" ? authorization.policy.id : "not-required",
381
406
  kind: "authorization",
@@ -427,8 +452,9 @@ function buildLlm(request, runner, options) {
427
452
  skip(`inference.${key}`);
428
453
  }
429
454
  const chatOptions = {};
455
+ // Sent unless the engine opts out; chatCompletion drops it once if the provider rejects it.
430
456
  if (request.outputSchema) {
431
- if (runner.connection.supportsJsonSchema === true)
457
+ if (runner.connection.supportsJsonSchema !== false)
432
458
  chatOptions.responseSchema = request.outputSchema;
433
459
  else
434
460
  skip("outputSchema");
@@ -4,4 +4,4 @@
4
4
  export { DEFAULT_AGENT_TIMEOUT_MS } from "./config.js";
5
5
  export { _setAgentDetectForTests, defaultWhich, detectAgentCliProfiles, pickDefaultAgentProfile } from "./detect.js";
6
6
  export { BUILTIN_AGENT_PROFILE_NAMES, getBuiltinAgentProfile, listBuiltinAgentProfiles, } from "./profiles.js";
7
- export { buildProposePrompt, buildReflectPrompt, buildSchemaRepairPrompt, extractDraftConfidence, parseAgentProposalPayload, } from "./prompts.js";
7
+ export { buildProposePrompt, buildReflectPrompt, parseAgentProposalPayload } from "./prompts.js";
@@ -8,7 +8,7 @@
8
8
  * coding-agent CLI. Named engines lower canonical harness metadata into this
9
9
  * intentionally small internal shape. The wrapper is in `./spawn.ts`.
10
10
  */
11
- import { COMMON_SPAWN_ENV_PASSTHROUGH } from "../../core/spawn-env.js";
11
+ import { COMMON_SPAWN_ENV_PASSTHROUGH, XDG_BASE_DIR_ENV_PASSTHROUGH } from "../../core/spawn-env.js";
12
12
  // AKM_EVENT_SOURCE carries usage-event provenance (improve/task) so that akm
13
13
  // invocations a spawned agent makes are recorded as machine traffic, not user
14
14
  // demand (DRIFT-6). Without it in the passthrough whitelist, buildChildEnv drops
@@ -30,7 +30,7 @@ const BUILTINS = {
30
30
  bin: "opencode",
31
31
  args: ["run"],
32
32
  stdio: "interactive",
33
- envPassthrough: [...COMMON_PASSTHROUGH, "OPENCODE_API_KEY", "OPENCODE_CONFIG"],
33
+ envPassthrough: [...COMMON_PASSTHROUGH, "OPENCODE_API_KEY", "OPENCODE_CONFIG", ...XDG_BASE_DIR_ENV_PASSTHROUGH],
34
34
  parseOutput: "text",
35
35
  },
36
36
  claude: {
@@ -4,15 +4,14 @@
4
4
  /**
5
5
  * Shared prompt builders for proposal-producing agent commands (#226).
6
6
  *
7
- * `akm reflect` and `akm propose` both shell out to the configured agent CLI
8
- * (via {@link runAgent}) and ask it for a structured proposal payload. The
9
- * prompts are intentionally similar and share their construction here. Agent
10
- * and SDK stdout keeps the JSON/file-write contracts; direct LLM reflect adds
11
- * native-schema and framed-markdown contracts. Keeping the prompt builders in
12
- * `src/integrations/agent/` rather than `src/llm/` is deliberate: these are
13
- * shell-out prompts targeting an agent CLI, not in-tree LLM API calls.
7
+ * `akm reflect` and `akm proposal new` both ask the configured engine for a
8
+ * structured proposal payload. The prompts are intentionally similar and share
9
+ * their construction here. `proposal new` asks every engine kind for the JSON
10
+ * object below, with {@link PROPOSAL_JSON_SCHEMA} as the request's output
11
+ * schema. Reflect asks every engine kind for its own contract, JSON Schema or
12
+ * framed markdown ({@link ReflectOutputMode}).
14
13
  *
15
- * The stdout output an agent must produce is a strict JSON object:
14
+ * The output an engine must produce for `proposal new` is a strict JSON object:
16
15
  *
17
16
  * ```json
18
17
  * {
@@ -32,6 +31,7 @@ import reflectOutputRepair from "../../assets/prompts/reflect-output-repair.md"
32
31
  import { placementTypes } from "../../core/asset/asset-placement.js";
33
32
  import { parseRefInput } from "../../core/asset/resolve-ref.js";
34
33
  import { authoringRulesForType, DESCRIPTION_MAX_CHARS, DESCRIPTION_MIN_CHARS, requiresDescription, } from "../../core/authoring-rules.js";
34
+ import { isRecord } from "../../core/common.js";
35
35
  import { parseEmbeddedJsonResponse, stripCodeFences, stripThinkBlocks } from "../../core/parse.js";
36
36
  /**
37
37
  * Per-asset-type frontmatter / authoring hints surfaced in the prompt so
@@ -77,10 +77,9 @@ export const REFLECT_CONTENT_CAP = 12_000;
77
77
  */
78
78
  export const REFLECT_TRUNCATION_MARKER = "... [truncated — focus on the visible portion]";
79
79
  /**
80
- * Common envelope every prompt asks the agent to honour when NO draft file
81
- * path is available. The wrapper code uses `JSON.parse(stdout)` to extract
82
- * the payload — anything outside the JSON object will be treated as a parse
83
- * error.
80
+ * The JSON envelope `proposal new` asks the engine to honour. The wrapper code
81
+ * uses `JSON.parse(stdout)` to extract the payload — anything outside the JSON
82
+ * object will be treated as a parse error.
84
83
  */
85
84
  const RESPONSE_CONTRACT_JSON = [
86
85
  "Respond ONLY with a single JSON object. No prose before or after.",
@@ -98,48 +97,25 @@ const RESPONSE_CONTRACT_JSON = [
98
97
  "Reviewers and the triage judge read this score when adjudicating the proposal queue. Overclaiming erodes trust in your proposals; underclaiming buries good ones. Be honest.",
99
98
  ].join("\n");
100
99
  /**
101
- * Response contract used when a draft file path is available. Instructs the
102
- * agent to write the improved asset content directly to the file using its
103
- * native file-editing tools — no stdout JSON parsing required.
100
+ * The JSON Schema of {@link RESPONSE_CONTRACT_JSON}'s object, the output
101
+ * schema `proposal new` sends with its request: an LLM engine gets it as
102
+ * `response_format`, codex as `--output-schema`, every agent engine as the
103
+ * shared schema instruction. It is in the strict form those native channels
104
+ * accept (every property required, no others), so it leaves out the optional
105
+ * `frontmatter`, which `content` already carries. The reply is held only to
106
+ * what {@link validateProposalPayload} needs.
104
107
  */
105
- function fileWriteContract(draftFilePath) {
106
- return [
107
- `Write the complete improved asset content to: ${draftFilePath}`,
108
- "Use your file-editing tools to create or overwrite that file.",
109
- "Do NOT output JSON to stdout. Do NOT print the file contents. Just write the file.",
110
- `Never include the text "${REFLECT_TRUNCATION_MARKER}" or any other content from outside the provided asset content in the file you write.`,
111
- "When done, output a single line on stdout: DRAFT_WRITTEN confidence=<0.0-1.0>",
112
- "`confidence` is REQUIRED and must be your honest self-rated [0, 1] score for this proposal:",
113
- " • 0.90+ — fixes a real defect or adds load-bearing missing content; reviewer would clearly accept.",
114
- " • 0.70–0.89 — clear improvement, but a reviewer might prefer different framing.",
115
- " • 0.50–0.69 — marginal / judgment call.",
116
- " • Below 0.50 — not confident; prefer not writing changes at all.",
117
- "Reviewers and the triage judge read this score during adjudication. Overclaim → trust erodes; underclaim → good changes buried.",
118
- ].join("\n");
119
- }
120
- /**
121
- * Extract a confidence score from a `DRAFT_WRITTEN confidence=<n>` line emitted
122
- * by an agent following {@link fileWriteContract}. Tolerates trailing prose,
123
- * surrounding log lines, and missing/invalid confidence (returns `undefined`
124
- * so callers can keep the proposal without a score).
125
- *
126
- * Matched forms (case-insensitive, anywhere in stdout):
127
- * - `DRAFT_WRITTEN confidence=0.85`
128
- * - `DRAFT_WRITTEN confidence=0.85 ...trailing`
129
- * - `DRAFT_WRITTEN` (no confidence — returns `undefined`)
130
- */
131
- export function extractDraftConfidence(stdout) {
132
- if (!stdout)
133
- return undefined;
134
- const match = stdout.match(/\bDRAFT_WRITTEN\b[^\S\r\n]+confidence=([0-9]*\.?[0-9]+)/i);
135
- if (!match)
136
- return undefined;
137
- const value = Number.parseFloat(match[1] ?? "");
138
- if (!Number.isFinite(value) || value < 0 || value > 1)
139
- return undefined;
140
- return value;
141
- }
142
- export function reflectLlmResponseContract(mode, targetScoped) {
108
+ export const PROPOSAL_JSON_SCHEMA = {
109
+ type: "object",
110
+ required: ["ref", "content", "confidence"],
111
+ additionalProperties: false,
112
+ properties: {
113
+ ref: { type: "string", description: "The new asset's ref as a subdir-qualified conceptId." },
114
+ content: { type: "string", description: "The full file contents that will be written if accepted." },
115
+ confidence: { type: "number", minimum: 0, maximum: 1, description: "Self-rated confidence in [0, 1]." },
116
+ },
117
+ };
118
+ export function reflectResponseContract(mode, targetScoped) {
143
119
  if (mode === "json_schema") {
144
120
  return reflectLlmSchemaContract
145
121
  .replace("{{FIELD_RULE}}", targetScoped
@@ -155,14 +131,7 @@ export function reflectLlmResponseContract(mode, targetScoped) {
155
131
  .trim();
156
132
  }
157
133
  export function buildReflectOutputRepairPrompt(mode, targetScoped) {
158
- return reflectOutputRepair.replace("{{OUTPUT_CONTRACT}}", reflectLlmResponseContract(mode, targetScoped)).trim();
159
- }
160
- function reflectResponseContract(input) {
161
- if (input.draftFilePath)
162
- return fileWriteContract(input.draftFilePath);
163
- if (input.outputMode)
164
- return reflectLlmResponseContract(input.outputMode, input.ref !== undefined);
165
- return RESPONSE_CONTRACT_JSON;
134
+ return reflectOutputRepair.replace("{{OUTPUT_CONTRACT}}", reflectResponseContract(mode, targetScoped)).trim();
166
135
  }
167
136
  /**
168
137
  * Whether the source asset content has a non-empty `description:` key in its
@@ -372,7 +341,8 @@ export function buildReflectPrompt(input) {
372
341
  // Embed concrete counts only when the gate will actually fire (source >= 200 chars).
373
342
  const showCharBounds = sourceBodyLen >= 200;
374
343
  const minChars = Math.max(Math.round(0.5 * sourceBodyLen), 150);
375
- const maxChars = Math.min(Math.max(Math.round(2.5 * sourceBodyLen), 2500), 25000);
344
+ // A source already past the 25000 cap may not grow, and need not shrink to the cap.
345
+ const maxChars = Math.max(Math.min(Math.max(Math.round(2.5 * sourceBodyLen), 2500), 25000), sourceBodyLen);
376
346
  sections.push([
377
347
  "## Content preservation rules (MUST follow)",
378
348
  "1. PRESERVE ALL concrete content: code blocks, fenced snippets, CLI commands, numbered/bulleted checklists, tables, YAML/JSON examples, file paths, configuration keys, environment variable names, and CSS/HTML selectors. These are load-bearing — do NOT replace them with prose summaries.",
@@ -381,22 +351,18 @@ export function buildReflectPrompt(input) {
381
351
  ? `3. DO NOT shrink the asset. Your body must be at least ${minChars} characters (source body is ${sourceBodyLen} chars; floor is 50%). If you genuinely need to remove a major section, explain why in a comment line at the top of the body (e.g. \`<!-- removed obsolete section X because ... -->\`).`
382
352
  : "3. DO NOT shrink the asset dramatically. The improved body must be at least 50% of the source body length. If you genuinely need to remove a major section, explain why in a comment line at the top of the body (e.g. `<!-- removed obsolete section X because ... -->`).",
383
353
  showCharBounds
384
- ? `4. DO NOT pad the asset with speculative material. Your body must be at most ${maxChars} characters (source body is ${sourceBodyLen} chars; ceiling is 250%). Do not add invented sections, hypothetical examples, or padding prose.`
354
+ ? `4. DO NOT pad the asset with speculative material. Your body must be at most ${maxChars} characters (source body is ${sourceBodyLen} chars; ceiling is ${maxChars === sourceBodyLen ? "100%" : "250%"}). Do not add invented sections, hypothetical examples, or padding prose.`
385
355
  : "4. DO NOT pad the asset with speculative material. The improved body must be at most 250% of the source body length unless the feedback explicitly requests added sections.",
386
356
  "5. Improve clarity of surrounding prose, fix structural issues, add missing required frontmatter fields. Do NOT rewrite a runbook into an essay.",
387
357
  ].join("\n"));
388
358
  }
389
- if (!input.draftFilePath && !input.outputMode && input.ref) {
390
- // Reinforce that the `ref` field is mandatory and must exactly match the target.
391
- // Small models frequently omit `ref` from the response JSON, causing parse errors.
392
- sections.push(`IMPORTANT: The JSON "ref" field is REQUIRED. It MUST be exactly: "${input.ref}"`);
393
- }
394
- sections.push(reflectResponseContract(input));
359
+ sections.push(reflectResponseContract(input.outputMode ?? "json_schema", input.ref !== undefined));
395
360
  return { prompt: sections.join("\n\n") };
396
361
  }
397
362
  /**
398
- * Build the prompt for `akm propose <type> <name> --task ...`. Asks the
399
- * agent to author a brand-new asset of the given type fulfilling `task`.
363
+ * Build the prompt for `akm proposal new <type> <name> --task ...`. Asks the
364
+ * engine to author a brand-new asset of the given type fulfilling `task`, and
365
+ * to return it as the JSON object of {@link RESPONSE_CONTRACT_JSON}.
400
366
  */
401
367
  export function buildProposePrompt(input) {
402
368
  const sections = [];
@@ -420,44 +386,7 @@ export function buildProposePrompt(input) {
420
386
  }
421
387
  }
422
388
  sections.push("Produce a single proposal that, if accepted, would land as the asset described above.");
423
- sections.push(input.draftFilePath ? fileWriteContract(input.draftFilePath) : RESPONSE_CONTRACT_JSON);
424
- return sections.join("\n\n");
425
- }
426
- /**
427
- * Build the prompt for the schema repair pass in `akm improve`. Asks the
428
- * agent to add the minimal required frontmatter to an asset that failed
429
- * validation — without rewriting the body.
430
- */
431
- export function buildSchemaRepairPrompt(input) {
432
- const sections = [];
433
- sections.push(`This ${input.type} asset failed schema validation with the error: "${input.reason}". ` +
434
- `Your task is to fix the schema issue by adding or correcting the missing/invalid field(s) ` +
435
- `while preserving all existing content.`);
436
- sections.push(`Target ref: ${input.ref}`);
437
- sections.push(`Schema requirements for ${input.type} assets: ${hintForType(input.type)}`);
438
- if (input.standardsContext?.trim()) {
439
- sections.push("Standards to follow (the rulebook for this target):");
440
- sections.push(input.standardsContext.trim());
441
- }
442
- {
443
- const authoringRules = authoringRulesForType(input.type);
444
- if (authoringRules) {
445
- sections.push(authoringRules);
446
- }
447
- }
448
- const CONTENT_CAP = 3000;
449
- const body = input.assetContent.trimEnd();
450
- const truncated = body.length > CONTENT_CAP;
451
- sections.push("Current asset content (first 3000 chars — sufficient to generate missing frontmatter):");
452
- sections.push("```");
453
- sections.push(truncated ? `${body.slice(0, CONTENT_CAP)}\n... [truncated]` : body);
454
- sections.push("```");
455
- sections.push("Produce the minimal fix: add ONLY the missing required frontmatter field(s). " +
456
- "Do not rewrite the body unless it is empty. " +
457
- "If `description` is missing, generate a concise one-sentence description from the content. " +
458
- "If `when_to_use` is missing, generate a one-line trigger sentence. " +
459
- "Preserve all existing frontmatter keys and the full body verbatim.");
460
- sections.push(input.draftFilePath ? fileWriteContract(input.draftFilePath) : RESPONSE_CONTRACT_JSON);
389
+ sections.push(RESPONSE_CONTRACT_JSON);
461
390
  return sections.join("\n\n");
462
391
  }
463
392
  /**
@@ -487,19 +416,31 @@ export function parseAgentProposalPayload(stdout) {
487
416
  throw directErr;
488
417
  parsed = embedded;
489
418
  }
419
+ const verdict = validateProposalPayload(parsed);
420
+ if (!verdict.ok)
421
+ throw new Error(verdict.errors.join("; "));
422
+ return verdict.value;
423
+ }
424
+ /**
425
+ * The proposal in a parsed reply, or why the reply is not one: `ref` and
426
+ * `content` must be non-empty strings. A malformed optional field is dropped,
427
+ * never refused.
428
+ */
429
+ export function validateProposalPayload(parsed) {
430
+ if (!isRecord(parsed))
431
+ return { ok: false, errors: ["agent response is not a JSON object"] };
490
432
  if (typeof parsed.ref !== "string" || !parsed.ref.trim()) {
491
- throw new Error('agent response missing required string field "ref"');
433
+ return { ok: false, errors: ['agent response missing required string field "ref"'] };
492
434
  }
493
435
  if (typeof parsed.content !== "string" || !parsed.content.trim()) {
494
- throw new Error('agent response missing required string field "content"');
436
+ return { ok: false, errors: ['agent response missing required string field "content"'] };
495
437
  }
496
438
  const out = {
497
439
  ref: parsed.ref.trim(),
498
440
  content: parsed.content,
499
441
  };
500
- if (parsed.frontmatter && typeof parsed.frontmatter === "object" && !Array.isArray(parsed.frontmatter)) {
442
+ if (isRecord(parsed.frontmatter))
501
443
  out.frontmatter = parsed.frontmatter;
502
- }
503
444
  // Phase 6A: extract optional `confidence` (number in [0, 1]). Clamp gently
504
445
  // rather than reject — a model that returns 1.0 or 0 with extra precision
505
446
  // (e.g. 1.0000001) should still surface a usable score. Anything that isn't
@@ -509,5 +450,5 @@ export function parseAgentProposalPayload(stdout) {
509
450
  const clamped = Math.max(0, Math.min(1, parsed.confidence));
510
451
  out.confidence = clamped;
511
452
  }
512
- return out;
453
+ return { ok: true, value: out };
513
454
  }
@@ -2,6 +2,9 @@
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  import { ConfigError } from "../../core/errors.js";
5
+ import { withSchemaInstruction } from "../../core/structured.js";
6
+ import { MODEL_WORK_POLICY_ID } from "../../execution/source.js";
7
+ import { HARNESS_MODEL_WORK_IDS } from "../harnesses/ids.js";
5
8
  import { composeConversationFallbackPrompt } from "./conversation-fallback.js";
6
9
  import { composePersonaFallbackPrompt } from "./persona-fallback.js";
7
10
  /** A tool selection that actually names tools (not omitted, null, or empty). */
@@ -47,6 +50,7 @@ export function extensionFields(request) {
47
50
  export function createAgentRequestLowerer(options) {
48
51
  const supportedInference = new Set(options.inference ?? []);
49
52
  return (_profile, request) => {
53
+ const modelWork = request.authorization.policy?.id === MODEL_WORK_POLICY_ID;
50
54
  const notices = [];
51
55
  const skip = (field) => {
52
56
  notices.push(untranslated(options.adapter, field));
@@ -78,24 +82,33 @@ export function createAgentRequestLowerer(options) {
78
82
  }
79
83
  dispatch.agent = request.agent;
80
84
  }
85
+ // Every agent transport gets the schema as the one instruction; a harness
86
+ // with a native channel (codex --output-schema) also reads dispatch.schema.
87
+ if (request.outputSchema) {
88
+ prompt = withSchemaInstruction(prompt, request.outputSchema);
89
+ dispatch.schema = request.outputSchema;
90
+ }
81
91
  dispatch.prompt = prompt;
82
92
  if (request.model)
83
93
  dispatch.model = request.model.resolved;
84
94
  if (Object.hasOwn(request, "inference")) {
85
95
  dispatch.inference = request.inference ?? null;
86
96
  for (const key of Object.keys(request.inference ?? {}).sort()) {
87
- if (!supportedInference.has(key))
97
+ if (!(modelWork && supportedInference.has(key)))
88
98
  skip(`inference.${key}`);
89
99
  }
90
- if (typeof request.inference?.effort === "string")
91
- dispatch.effort = request.inference.effort;
92
100
  }
93
- if (request.outputSchema) {
94
- if (!options.outputSchema)
95
- skip("outputSchema");
96
- dispatch.schema = request.outputSchema;
101
+ if (modelWork) {
102
+ if (!HARNESS_MODEL_WORK_IDS.has(options.adapter)) {
103
+ throw new ConfigError(`The ${options.adapter} transport cannot enforce the model-work tool policy.`, "INVALID_CONFIG_FILE");
104
+ }
105
+ // The builder selects its own confined agent or flags; another agent would replace them.
106
+ if (dispatch.agent) {
107
+ throw new ConfigError(`The ${options.adapter} transport cannot run native agent ${JSON.stringify(dispatch.agent)} under the model-work tool policy.`, "INVALID_CONFIG_FILE");
108
+ }
109
+ dispatch.modelWork = true;
97
110
  }
98
- if (request.tools !== undefined) {
111
+ else if (request.tools !== undefined) {
99
112
  // An explicit empty selection still reaches the builder (e.g. an empty allowlist).
100
113
  if (hasToolSelection(request.tools) && !translatesTools(options.tools, request.tools)) {
101
114
  throw new ConfigError(`The ${options.adapter} transport cannot enforce the resolved tool policy.`, "INVALID_CONFIG_FILE");
@@ -8,11 +8,18 @@
8
8
  * credential and passthrough value that could reach the child is redacted
9
9
  * from the result.
10
10
  */
11
+ import fs from "node:fs";
12
+ import os from "node:os";
13
+ import path from "node:path";
11
14
  import { assertNever } from "../../core/assert.js";
12
15
  import { UsageError } from "../../core/errors.js";
13
16
  import { collectSensitiveValues, isEnvPassthroughValueSafeToExpose, redactSensitiveText, redactSensitiveValue, } from "../../core/redaction.js";
17
+ import { MODEL_WORK_POLICY_ID } from "../../execution/source.js";
14
18
  import { chatCompletion, LlmCallError } from "../../llm/client.js";
19
+ import { emitLlmUsage } from "../../llm/usage-telemetry.js";
20
+ import { getHarness } from "../harnesses/index.js";
15
21
  import { closeServer as disposeOpencodeSdkServers, runOpencodeSdk } from "../harnesses/opencode-sdk/sdk-runner.js";
22
+ import { modelFromArgs } from "./builder-shared.js";
16
23
  import { lookupApiKeyFileValue, lookupApiKeySecretRefValue, lookupCredentialFromEnv, resolveEngine, } from "./engine-resolution.js";
17
24
  import { materializeLlmRunnerConnection, materializeSdkFallbackConnection } from "./runner.js";
18
25
  import { runAgent } from "./spawn.js";
@@ -74,6 +81,42 @@ function llmFailureReason(error) {
74
81
  return "spawn_failed";
75
82
  }
76
83
  }
84
+ const USAGE_ERROR_CODES = {
85
+ timeout: "timeout",
86
+ aborted: "aborted",
87
+ parse_error: "parse_error",
88
+ llm_rate_limit: "rate_limited",
89
+ };
90
+ /**
91
+ * The model an agent or SDK dispatch ran: the request's, else the one an agent
92
+ * CLI's own `args` select, which its command carries when the request names
93
+ * none. An SDK server never sees `args`, so an SDK engine with no model named
94
+ * has none to report: opencode picks it.
95
+ */
96
+ function dispatchedModel({ request, runner }) {
97
+ return request.model?.resolved ?? (runner.kind === "agent" ? modelFromArgs(runner.profile.args) : undefined);
98
+ }
99
+ /**
100
+ * One usage record for an agent or SDK dispatch, through the same sink and
101
+ * ambient `withLlmStage` attribution as the LLM transport's per-HTTP-attempt
102
+ * records: the model it ran, and tokens when the runner reported them.
103
+ */
104
+ function recordDispatchUsage(execution, result) {
105
+ const { inputTokens, outputTokens, reasoningTokens } = result.usage ?? {};
106
+ const reported = [inputTokens, outputTokens, reasoningTokens].filter((count) => count !== undefined);
107
+ const model = dispatchedModel(execution);
108
+ emitLlmUsage({
109
+ outcome: result.ok ? "success" : "error",
110
+ modelSource: "configured",
111
+ ...(model ? { model } : {}),
112
+ durationMs: result.durationMs,
113
+ ...(inputTokens !== undefined ? { promptTokens: inputTokens } : {}),
114
+ ...(outputTokens !== undefined ? { completionTokens: outputTokens } : {}),
115
+ ...(reasoningTokens !== undefined ? { reasoningTokens } : {}),
116
+ ...(reported.length > 0 ? { totalTokens: reported.reduce((sum, count) => sum + count, 0) } : {}),
117
+ ...(result.ok ? {} : { errorCode: (result.reason && USAGE_ERROR_CODES[result.reason]) ?? "unknown_error" }),
118
+ });
119
+ }
77
120
  async function dispatchRunner(runner, prompt, opts, seams, llm) {
78
121
  const envSource = opts.envSource ?? process.env;
79
122
  const secrets = collectDispatchSensitiveValues(runner, opts, envSource);
@@ -103,9 +146,53 @@ async function dispatchRunner(runner, prompt, opts, seams, llm) {
103
146
  }
104
147
  return redactResult(result, collectSensitiveValues(secrets));
105
148
  }
106
- /** Run a built execution. Credentials are read here, once per call, and never returned. */
149
+ /**
150
+ * The scratch working directory for one model-work dispatch on an agent or SDK
151
+ * engine, so the edit the model-work tool policy grants never reaches the
152
+ * stash.
153
+ */
154
+ function createModelWorkDirectory() {
155
+ return fs.mkdtempSync(path.join(os.tmpdir(), "akm-model-work-"));
156
+ }
157
+ /**
158
+ * Run a built execution. Credentials are read here, once per call, and never
159
+ * returned. Model work on an agent or SDK engine runs in a scratch working
160
+ * directory that akm creates for the dispatch and removes after it, and must
161
+ * end with an answer: an agent that stops with none (opencode at its step
162
+ * limit, for one) has failed with `parse_error`.
163
+ */
107
164
  export async function runExecution(execution, options = {}) {
108
- const opts = { ...execution.options };
165
+ const scratch = execution.runner.kind !== "llm" && execution.request.authorization.policy?.id === MODEL_WORK_POLICY_ID
166
+ ? createModelWorkDirectory()
167
+ : undefined;
168
+ try {
169
+ return await runBuiltExecution(execution, options, scratch);
170
+ }
171
+ finally {
172
+ if (scratch)
173
+ fs.rmSync(scratch, { recursive: true, force: true });
174
+ }
175
+ }
176
+ /** A successful reply's text, unwrapped from its harness's framing (claude's `--output-format json` result envelope, for one). */
177
+ export function unwrapHarnessReply(runner, result) {
178
+ const extractor = runner.kind === "agent" ? getHarness(runner.profile.platform ?? runner.profile.name)?.resultExtractor : undefined;
179
+ return extractor ? extractor(result) : { text: result.stdout };
180
+ }
181
+ /** A model-work reply's answer: its harness's framing stripped, and no answer is a `parse_error`. */
182
+ function modelWorkAnswer(runner, result) {
183
+ const extracted = unwrapHarnessReply(runner, result);
184
+ const answer = {
185
+ ...result,
186
+ stdout: extracted.text,
187
+ ...(extracted.sessionId ? { sessionId: extracted.sessionId } : {}),
188
+ };
189
+ if (extracted.text.trim() !== "")
190
+ return answer;
191
+ return { ...answer, ok: false, reason: "parse_error", error: `Engine "${runner.engine}" returned no answer.` };
192
+ }
193
+ /** `scratch` is the model-work working directory, set only for model work on an agent or SDK engine. */
194
+ async function runBuiltExecution(execution, options, scratch) {
195
+ const opts = { ...execution.options, ...(scratch ? { cwd: scratch } : {}) };
109
196
  const operational = options.runOptions ?? {};
110
197
  for (const key of OPERATIONAL_OPTIONS) {
111
198
  if (operational[key] !== undefined)
@@ -140,7 +227,13 @@ export async function runExecution(execution, options = {}) {
140
227
  };
141
228
  }
142
229
  };
143
- return dispatchRunner(execution.runner, execution.prompt, opts, options, llm);
230
+ let result = await dispatchRunner(execution.runner, execution.prompt, opts, options, llm);
231
+ if (scratch !== undefined && result.ok)
232
+ result = modelWorkAnswer(execution.runner, result);
233
+ // The LLM transport records each HTTP attempt itself.
234
+ if (execution.runner.kind !== "llm")
235
+ recordDispatchUsage(execution, result);
236
+ return result;
144
237
  }
145
238
  /** `akm agent`: launch an agent engine interactively, with no prompt of its own. */
146
239
  export async function executeInteractiveAgentInvocation(input, seams = {}) {
@@ -67,6 +67,12 @@ export function decodeFrozenRunnerSpec(value) {
67
67
  export function runnerIsLlm(runner) {
68
68
  return runner.kind === "llm";
69
69
  }
70
- export function runnerSupportsFileWrite(runner) {
71
- return runner.kind !== "llm";
70
+ /**
71
+ * The LLM connection behind a runner, whatever its kind: an llm runner's own,
72
+ * an sdk runner's provider fallback, none for an agent CLI.
73
+ */
74
+ export function runnerLlmConnection(runner) {
75
+ if (runner.kind === "llm")
76
+ return runner.connection;
77
+ return runner.kind === "sdk" ? runner.fallbackConnection : undefined;
72
78
  }