akm-cli 0.9.23 → 0.9.25-alpha.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/CHANGELOG.md +339 -0
  2. package/dist/commands/health/checks.js +10 -11
  3. package/dist/commands/improve/consolidate/pair-pass.js +3 -2
  4. package/dist/commands/improve/consolidate.js +3 -2
  5. package/dist/commands/improve/execution.js +4 -10
  6. package/dist/commands/improve/extract-prompt.js +4 -4
  7. package/dist/commands/improve/extract.js +10 -13
  8. package/dist/commands/improve/improve-cli.js +32 -33
  9. package/dist/commands/improve/improve-strategies.js +49 -43
  10. package/dist/commands/improve/improve-usage-report.js +8 -17
  11. package/dist/commands/improve/preparation.js +3 -1
  12. package/dist/commands/improve/reflect.js +132 -177
  13. package/dist/commands/improve/stage.js +69 -18
  14. package/dist/commands/proposal/drain.js +19 -7
  15. package/dist/commands/proposal/proposal-cli.js +1 -5
  16. package/dist/commands/proposal/propose-cli.js +2 -2
  17. package/dist/commands/proposal/propose.js +80 -83
  18. package/dist/commands/remember.js +3 -3
  19. package/dist/commands/sources/schema-repair.js +1 -1
  20. package/dist/core/config/config-schema.js +37 -60
  21. package/dist/core/config/engine-semantics.js +15 -11
  22. package/dist/core/config/schema/engines.js +33 -15
  23. package/dist/core/config/schema/improve-processes.js +2 -2
  24. package/dist/core/improve-result.js +3 -3
  25. package/dist/core/structured.js +10 -0
  26. package/dist/execution/source.js +14 -0
  27. package/dist/indexer/passes/memory-inference.js +2 -1
  28. package/dist/integrations/agent/builder-shared.js +15 -0
  29. package/dist/integrations/agent/config.js +2 -0
  30. package/dist/integrations/agent/engine-resolution.js +16 -31
  31. package/dist/integrations/agent/execution.js +46 -21
  32. package/dist/integrations/agent/index.js +1 -1
  33. package/dist/integrations/agent/model-map.js +16 -15
  34. package/dist/integrations/agent/prompts.js +53 -76
  35. package/dist/integrations/agent/request-lowering.js +20 -9
  36. package/dist/integrations/agent/runner-dispatch.js +103 -4
  37. package/dist/integrations/agent/runner.js +8 -2
  38. package/dist/integrations/harnesses/aider/agent-builder.js +11 -23
  39. package/dist/integrations/harnesses/aider/index.js +0 -5
  40. package/dist/integrations/harnesses/amazonq/agent-builder.js +10 -21
  41. package/dist/integrations/harnesses/amazonq/index.js +0 -5
  42. package/dist/integrations/harnesses/claude/agent-builder.js +61 -38
  43. package/dist/integrations/harnesses/claude/index.js +0 -14
  44. package/dist/integrations/harnesses/codex/agent-builder.js +4 -4
  45. package/dist/integrations/harnesses/codex/index.js +0 -4
  46. package/dist/integrations/harnesses/copilot/agent-builder.js +4 -13
  47. package/dist/integrations/harnesses/copilot/index.js +2 -7
  48. package/dist/integrations/harnesses/gemini/agent-builder.js +4 -13
  49. package/dist/integrations/harnesses/gemini/index.js +0 -5
  50. package/dist/integrations/harnesses/ids.js +18 -10
  51. package/dist/integrations/harnesses/opencode/agent-builder.js +52 -10
  52. package/dist/integrations/harnesses/opencode/index.js +0 -8
  53. package/dist/integrations/harnesses/opencode/model-config.js +80 -0
  54. package/dist/integrations/harnesses/opencode/model-work-agent.js +80 -0
  55. package/dist/integrations/harnesses/opencode-sdk/harness.js +6 -7
  56. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +168 -24
  57. package/dist/integrations/harnesses/openhands/agent-builder.js +8 -21
  58. package/dist/integrations/harnesses/openhands/index.js +0 -5
  59. package/dist/integrations/harnesses/pi/agent-builder.js +8 -21
  60. package/dist/integrations/harnesses/pi/index.js +0 -5
  61. package/dist/llm/client.js +5 -0
  62. package/dist/llm/feature-gate.js +10 -4
  63. package/dist/llm/index-passes.js +3 -6
  64. package/dist/llm/memory-infer.js +6 -5
  65. package/dist/llm/structured-call.js +33 -11
  66. package/dist/scripts/akm-migrate-node.js +298 -204
  67. package/dist/scripts/akm-migrate.js +298 -204
  68. package/dist/workflows/exec/step-work.js +6 -5
  69. package/dist/workflows/freeze/step-values.js +1 -1
  70. package/docs/reference/cli.md +34 -14
  71. package/docs/reference/configuration.md +171 -12
  72. package/docs/reference/workflow-schema.md +13 -9
  73. package/package.json +1 -1
  74. package/schemas/akm-config.json +36 -0
@@ -220,32 +220,33 @@ function mergeProfiles(base, overlay) {
220
220
  * copied verbatim from the engine's own config value; it must already be
221
221
  * meaningful for the model-map column's platform (akm does not translate
222
222
  * between an engine's connection and an agent platform's own provider
223
- * registry). Only `kind: "llm"` engines contribute inference defaults — an
224
- * agent-kind engine's schema carries no temperature/thinking fields.
223
+ * registry). An agent-kind engine contributes only the inference fields its
224
+ * platform translates (config validation rejects the rest); an `llm` engine
225
+ * also contributes `supportsJsonSchema` and `extraParams`.
225
226
  */
226
227
  export function engineModelAndInference(engine) {
227
228
  const out = {};
228
229
  if (Object.hasOwn(engine, "model") && engine.model !== undefined)
229
230
  out.model = engine.model;
231
+ const inference = {};
232
+ if (Object.hasOwn(engine, "temperature"))
233
+ inference.temperature = engine.temperature;
234
+ if (Object.hasOwn(engine, "maxTokens"))
235
+ inference.maxTokens = engine.maxTokens;
230
236
  if (engine.kind === "llm") {
231
- const inference = {};
232
- if (Object.hasOwn(engine, "temperature"))
233
- inference.temperature = engine.temperature;
234
- if (Object.hasOwn(engine, "maxTokens"))
235
- inference.maxTokens = engine.maxTokens;
236
237
  if (Object.hasOwn(engine, "supportsJsonSchema"))
237
238
  inference.supportsJsonSchema = engine.supportsJsonSchema;
238
239
  if (Object.hasOwn(engine, "extraParams"))
239
240
  inference.extraParams = engine.extraParams;
240
- if (Object.hasOwn(engine, "contextLength"))
241
- inference.contextLength = engine.contextLength;
242
- if (Object.hasOwn(engine, "enableThinking"))
243
- inference.enableThinking = engine.enableThinking;
244
- if (Object.hasOwn(engine, "reasoningEffort"))
245
- inference.reasoningEffort = engine.reasoningEffort;
246
- if (Object.keys(inference).length > 0)
247
- out.inference = inference;
248
241
  }
242
+ if (Object.hasOwn(engine, "contextLength"))
243
+ inference.contextLength = engine.contextLength;
244
+ if (Object.hasOwn(engine, "enableThinking"))
245
+ inference.enableThinking = engine.enableThinking;
246
+ if (Object.hasOwn(engine, "reasoningEffort"))
247
+ inference.reasoningEffort = engine.reasoningEffort;
248
+ if (Object.keys(inference).length > 0)
249
+ out.inference = inference;
249
250
  return Object.freeze(out);
250
251
  }
251
252
  /** Overlay user fields over installed fields, per (alias, column), without resolving `engine` indirection. */
@@ -4,15 +4,14 @@
4
4
  /**
5
5
  * Shared prompt builders for proposal-producing agent commands (#226).
6
6
  *
7
- * `akm reflect` and `akm propose` both shell out to the configured agent CLI
8
- * (via {@link runAgent}) and ask it for a structured proposal payload. The
9
- * prompts are intentionally similar and share their construction here. Agent
10
- * and SDK stdout keeps the JSON/file-write contracts; direct LLM reflect adds
11
- * native-schema and framed-markdown contracts. Keeping the prompt builders in
12
- * `src/integrations/agent/` rather than `src/llm/` is deliberate: these are
13
- * shell-out prompts targeting an agent CLI, not in-tree LLM API calls.
7
+ * `akm reflect` and `akm proposal new` both ask the configured engine for a
8
+ * structured proposal payload. The prompts are intentionally similar and share
9
+ * their construction here. `proposal new` asks every engine kind for the JSON
10
+ * object below, with {@link PROPOSAL_JSON_SCHEMA} as the request's output
11
+ * schema. Reflect asks every engine kind for its own contract, JSON Schema or
12
+ * framed markdown ({@link ReflectOutputMode}).
14
13
  *
15
- * The stdout output an agent must produce is a strict JSON object:
14
+ * The output an engine must produce for `proposal new` is a strict JSON object:
16
15
  *
17
16
  * ```json
18
17
  * {
@@ -32,6 +31,7 @@ import reflectOutputRepair from "../../assets/prompts/reflect-output-repair.md"
32
31
  import { placementTypes } from "../../core/asset/asset-placement.js";
33
32
  import { parseRefInput } from "../../core/asset/resolve-ref.js";
34
33
  import { authoringRulesForType, DESCRIPTION_MAX_CHARS, DESCRIPTION_MIN_CHARS, requiresDescription, } from "../../core/authoring-rules.js";
34
+ import { isRecord } from "../../core/common.js";
35
35
  import { parseEmbeddedJsonResponse, stripCodeFences, stripThinkBlocks } from "../../core/parse.js";
36
36
  /**
37
37
  * Per-asset-type frontmatter / authoring hints surfaced in the prompt so
@@ -77,10 +77,9 @@ export const REFLECT_CONTENT_CAP = 12_000;
77
77
  */
78
78
  export const REFLECT_TRUNCATION_MARKER = "... [truncated — focus on the visible portion]";
79
79
  /**
80
- * Common envelope every prompt asks the agent to honour when NO draft file
81
- * path is available. The wrapper code uses `JSON.parse(stdout)` to extract
82
- * the payload — anything outside the JSON object will be treated as a parse
83
- * error.
80
+ * The JSON envelope `proposal new` asks the engine to honour. The wrapper code
81
+ * uses `JSON.parse(stdout)` to extract the payload — anything outside the JSON
82
+ * object will be treated as a parse error.
84
83
  */
85
84
  const RESPONSE_CONTRACT_JSON = [
86
85
  "Respond ONLY with a single JSON object. No prose before or after.",
@@ -98,48 +97,25 @@ const RESPONSE_CONTRACT_JSON = [
98
97
  "Reviewers and the triage judge read this score when adjudicating the proposal queue. Overclaiming erodes trust in your proposals; underclaiming buries good ones. Be honest.",
99
98
  ].join("\n");
100
99
  /**
101
- * Response contract used when a draft file path is available. Instructs the
102
- * agent to write the improved asset content directly to the file using its
103
- * native file-editing tools — no stdout JSON parsing required.
100
+ * The JSON Schema of {@link RESPONSE_CONTRACT_JSON}'s object, the output
101
+ * schema `proposal new` sends with its request: an LLM engine gets it as
102
+ * `response_format`, codex as `--output-schema`, every agent engine as the
103
+ * shared schema instruction. It is in the strict form those native channels
104
+ * accept (every property required, no others), so it leaves out the optional
105
+ * `frontmatter`, which `content` already carries. The reply is held only to
106
+ * what {@link validateProposalPayload} needs.
104
107
  */
105
- function fileWriteContract(draftFilePath) {
106
- return [
107
- `Write the complete improved asset content to: ${draftFilePath}`,
108
- "Use your file-editing tools to create or overwrite that file.",
109
- "Do NOT output JSON to stdout. Do NOT print the file contents. Just write the file.",
110
- `Never include the text "${REFLECT_TRUNCATION_MARKER}" or any other content from outside the provided asset content in the file you write.`,
111
- "When done, output a single line on stdout: DRAFT_WRITTEN confidence=<0.0-1.0>",
112
- "`confidence` is REQUIRED and must be your honest self-rated [0, 1] score for this proposal:",
113
- " • 0.90+ — fixes a real defect or adds load-bearing missing content; reviewer would clearly accept.",
114
- " • 0.70–0.89 — clear improvement, but a reviewer might prefer different framing.",
115
- " • 0.50–0.69 — marginal / judgment call.",
116
- " • Below 0.50 — not confident; prefer not writing changes at all.",
117
- "Reviewers and the triage judge read this score during adjudication. Overclaim → trust erodes; underclaim → good changes buried.",
118
- ].join("\n");
119
- }
120
- /**
121
- * Extract a confidence score from a `DRAFT_WRITTEN confidence=<n>` line emitted
122
- * by an agent following {@link fileWriteContract}. Tolerates trailing prose,
123
- * surrounding log lines, and missing/invalid confidence (returns `undefined`
124
- * so callers can keep the proposal without a score).
125
- *
126
- * Matched forms (case-insensitive, anywhere in stdout):
127
- * - `DRAFT_WRITTEN confidence=0.85`
128
- * - `DRAFT_WRITTEN confidence=0.85 ...trailing`
129
- * - `DRAFT_WRITTEN` (no confidence — returns `undefined`)
130
- */
131
- export function extractDraftConfidence(stdout) {
132
- if (!stdout)
133
- return undefined;
134
- const match = stdout.match(/\bDRAFT_WRITTEN\b[^\S\r\n]+confidence=([0-9]*\.?[0-9]+)/i);
135
- if (!match)
136
- return undefined;
137
- const value = Number.parseFloat(match[1] ?? "");
138
- if (!Number.isFinite(value) || value < 0 || value > 1)
139
- return undefined;
140
- return value;
141
- }
142
- export function reflectLlmResponseContract(mode, targetScoped) {
108
+ export const PROPOSAL_JSON_SCHEMA = {
109
+ type: "object",
110
+ required: ["ref", "content", "confidence"],
111
+ additionalProperties: false,
112
+ properties: {
113
+ ref: { type: "string", description: "The new asset's ref as a subdir-qualified conceptId." },
114
+ content: { type: "string", description: "The full file contents that will be written if accepted." },
115
+ confidence: { type: "number", minimum: 0, maximum: 1, description: "Self-rated confidence in [0, 1]." },
116
+ },
117
+ };
118
+ export function reflectResponseContract(mode, targetScoped) {
143
119
  if (mode === "json_schema") {
144
120
  return reflectLlmSchemaContract
145
121
  .replace("{{FIELD_RULE}}", targetScoped
@@ -155,14 +131,7 @@ export function reflectLlmResponseContract(mode, targetScoped) {
155
131
  .trim();
156
132
  }
157
133
  export function buildReflectOutputRepairPrompt(mode, targetScoped) {
158
- return reflectOutputRepair.replace("{{OUTPUT_CONTRACT}}", reflectLlmResponseContract(mode, targetScoped)).trim();
159
- }
160
- function reflectResponseContract(input) {
161
- if (input.draftFilePath)
162
- return fileWriteContract(input.draftFilePath);
163
- if (input.outputMode)
164
- return reflectLlmResponseContract(input.outputMode, input.ref !== undefined);
165
- return RESPONSE_CONTRACT_JSON;
134
+ return reflectOutputRepair.replace("{{OUTPUT_CONTRACT}}", reflectResponseContract(mode, targetScoped)).trim();
166
135
  }
167
136
  /**
168
137
  * Whether the source asset content has a non-empty `description:` key in its
@@ -386,17 +355,13 @@ export function buildReflectPrompt(input) {
386
355
  "5. Improve clarity of surrounding prose, fix structural issues, add missing required frontmatter fields. Do NOT rewrite a runbook into an essay.",
387
356
  ].join("\n"));
388
357
  }
389
- if (!input.draftFilePath && !input.outputMode && input.ref) {
390
- // Reinforce that the `ref` field is mandatory and must exactly match the target.
391
- // Small models frequently omit `ref` from the response JSON, causing parse errors.
392
- sections.push(`IMPORTANT: The JSON "ref" field is REQUIRED. It MUST be exactly: "${input.ref}"`);
393
- }
394
- sections.push(reflectResponseContract(input));
358
+ sections.push(reflectResponseContract(input.outputMode ?? "json_schema", input.ref !== undefined));
395
359
  return { prompt: sections.join("\n\n") };
396
360
  }
397
361
  /**
398
- * Build the prompt for `akm propose <type> <name> --task ...`. Asks the
399
- * agent to author a brand-new asset of the given type fulfilling `task`.
362
+ * Build the prompt for `akm proposal new <type> <name> --task ...`. Asks the
363
+ * engine to author a brand-new asset of the given type fulfilling `task`, and
364
+ * to return it as the JSON object of {@link RESPONSE_CONTRACT_JSON}.
400
365
  */
401
366
  export function buildProposePrompt(input) {
402
367
  const sections = [];
@@ -420,7 +385,7 @@ export function buildProposePrompt(input) {
420
385
  }
421
386
  }
422
387
  sections.push("Produce a single proposal that, if accepted, would land as the asset described above.");
423
- sections.push(input.draftFilePath ? fileWriteContract(input.draftFilePath) : RESPONSE_CONTRACT_JSON);
388
+ sections.push(RESPONSE_CONTRACT_JSON);
424
389
  return sections.join("\n\n");
425
390
  }
426
391
  /**
@@ -457,7 +422,7 @@ export function buildSchemaRepairPrompt(input) {
457
422
  "If `description` is missing, generate a concise one-sentence description from the content. " +
458
423
  "If `when_to_use` is missing, generate a one-line trigger sentence. " +
459
424
  "Preserve all existing frontmatter keys and the full body verbatim.");
460
- sections.push(input.draftFilePath ? fileWriteContract(input.draftFilePath) : RESPONSE_CONTRACT_JSON);
425
+ sections.push(RESPONSE_CONTRACT_JSON);
461
426
  return sections.join("\n\n");
462
427
  }
463
428
  /**
@@ -487,19 +452,31 @@ export function parseAgentProposalPayload(stdout) {
487
452
  throw directErr;
488
453
  parsed = embedded;
489
454
  }
455
+ const verdict = validateProposalPayload(parsed);
456
+ if (!verdict.ok)
457
+ throw new Error(verdict.errors.join("; "));
458
+ return verdict.value;
459
+ }
460
+ /**
461
+ * The proposal in a parsed reply, or why the reply is not one: `ref` and
462
+ * `content` must be non-empty strings. A malformed optional field is dropped,
463
+ * never refused.
464
+ */
465
+ export function validateProposalPayload(parsed) {
466
+ if (!isRecord(parsed))
467
+ return { ok: false, errors: ["agent response is not a JSON object"] };
490
468
  if (typeof parsed.ref !== "string" || !parsed.ref.trim()) {
491
- throw new Error('agent response missing required string field "ref"');
469
+ return { ok: false, errors: ['agent response missing required string field "ref"'] };
492
470
  }
493
471
  if (typeof parsed.content !== "string" || !parsed.content.trim()) {
494
- throw new Error('agent response missing required string field "content"');
472
+ return { ok: false, errors: ['agent response missing required string field "content"'] };
495
473
  }
496
474
  const out = {
497
475
  ref: parsed.ref.trim(),
498
476
  content: parsed.content,
499
477
  };
500
- if (parsed.frontmatter && typeof parsed.frontmatter === "object" && !Array.isArray(parsed.frontmatter)) {
478
+ if (isRecord(parsed.frontmatter))
501
479
  out.frontmatter = parsed.frontmatter;
502
- }
503
480
  // Phase 6A: extract optional `confidence` (number in [0, 1]). Clamp gently
504
481
  // rather than reject — a model that returns 1.0 or 0 with extra precision
505
482
  // (e.g. 1.0000001) should still surface a usable score. Anything that isn't
@@ -509,5 +486,5 @@ export function parseAgentProposalPayload(stdout) {
509
486
  const clamped = Math.max(0, Math.min(1, parsed.confidence));
510
487
  out.confidence = clamped;
511
488
  }
512
- return out;
489
+ return { ok: true, value: out };
513
490
  }
@@ -2,6 +2,8 @@
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  import { ConfigError } from "../../core/errors.js";
5
+ import { withSchemaInstruction } from "../../core/structured.js";
6
+ import { isModelWorkTools } from "../../execution/source.js";
5
7
  import { composeConversationFallbackPrompt } from "./conversation-fallback.js";
6
8
  import { composePersonaFallbackPrompt } from "./persona-fallback.js";
7
9
  /** A tool selection that actually names tools (not omitted, null, or empty). */
@@ -45,8 +47,8 @@ export function extensionFields(request) {
45
47
  * composing into the prompt what the harness has no channel for.
46
48
  */
47
49
  export function createAgentRequestLowerer(options) {
48
- const supportedInference = new Set(options.inference ?? []);
49
- return (_profile, request) => {
50
+ return (profile, request) => {
51
+ const supportedInference = new Set(typeof options.inference === "function" ? options.inference(profile, request) : (options.inference ?? []));
50
52
  const notices = [];
51
53
  const skip = (field) => {
52
54
  notices.push(untranslated(options.adapter, field));
@@ -78,6 +80,12 @@ export function createAgentRequestLowerer(options) {
78
80
  }
79
81
  dispatch.agent = request.agent;
80
82
  }
83
+ // Every agent transport gets the schema as the one instruction; a harness
84
+ // with a native channel (codex --output-schema) also reads dispatch.schema.
85
+ if (request.outputSchema) {
86
+ prompt = withSchemaInstruction(prompt, request.outputSchema);
87
+ dispatch.schema = request.outputSchema;
88
+ }
81
89
  dispatch.prompt = prompt;
82
90
  if (request.model)
83
91
  dispatch.model = request.model.resolved;
@@ -87,15 +95,18 @@ export function createAgentRequestLowerer(options) {
87
95
  if (!supportedInference.has(key))
88
96
  skip(`inference.${key}`);
89
97
  }
90
- if (typeof request.inference?.effort === "string")
91
- dispatch.effort = request.inference.effort;
92
98
  }
93
- if (request.outputSchema) {
94
- if (!options.outputSchema)
95
- skip("outputSchema");
96
- dispatch.schema = request.outputSchema;
99
+ if (isModelWorkTools(request.tools)) {
100
+ if (!options.modelWorkTools) {
101
+ throw new ConfigError(`The ${options.adapter} transport cannot enforce the model-work tool policy.`, "INVALID_CONFIG_FILE");
102
+ }
103
+ // The builder selects its own confined agent or flags; another agent would replace them.
104
+ if (dispatch.agent) {
105
+ throw new ConfigError(`The ${options.adapter} transport cannot run native agent ${JSON.stringify(dispatch.agent)} under the model-work tool policy.`, "INVALID_CONFIG_FILE");
106
+ }
107
+ dispatch.tools = request.tools;
97
108
  }
98
- if (request.tools !== undefined) {
109
+ else if (request.tools !== undefined) {
99
110
  // An explicit empty selection still reaches the builder (e.g. an empty allowlist).
100
111
  if (hasToolSelection(request.tools) && !translatesTools(options.tools, request.tools)) {
101
112
  throw new ConfigError(`The ${options.adapter} transport cannot enforce the resolved tool policy.`, "INVALID_CONFIG_FILE");
@@ -8,10 +8,16 @@
8
8
  * credential and passthrough value that could reach the child is redacted
9
9
  * from the result.
10
10
  */
11
+ import fs from "node:fs";
12
+ import os from "node:os";
13
+ import path from "node:path";
11
14
  import { assertNever } from "../../core/assert.js";
12
- import { UsageError } from "../../core/errors.js";
15
+ import { ConfigError, UsageError } from "../../core/errors.js";
13
16
  import { collectSensitiveValues, isEnvPassthroughValueSafeToExpose, redactSensitiveText, redactSensitiveValue, } from "../../core/redaction.js";
17
+ import { isModelWorkTools } from "../../execution/source.js";
14
18
  import { chatCompletion, LlmCallError } from "../../llm/client.js";
19
+ import { emitLlmUsage } from "../../llm/usage-telemetry.js";
20
+ import { getHarness } from "../harnesses/index.js";
15
21
  import { closeServer as disposeOpencodeSdkServers, runOpencodeSdk } from "../harnesses/opencode-sdk/sdk-runner.js";
16
22
  import { lookupApiKeyFileValue, lookupApiKeySecretRefValue, lookupCredentialFromEnv, resolveEngine, } from "./engine-resolution.js";
17
23
  import { materializeLlmRunnerConnection, materializeSdkFallbackConnection } from "./runner.js";
@@ -74,6 +80,32 @@ function llmFailureReason(error) {
74
80
  return "spawn_failed";
75
81
  }
76
82
  }
83
+ const USAGE_ERROR_CODES = {
84
+ timeout: "timeout",
85
+ aborted: "aborted",
86
+ parse_error: "parse_error",
87
+ llm_rate_limit: "rate_limited",
88
+ };
89
+ /**
90
+ * One usage record for an agent or SDK dispatch, through the same sink and
91
+ * ambient `withLlmStage` attribution as the LLM transport's per-HTTP-attempt
92
+ * records: the request's model, and tokens when the runner reported them.
93
+ */
94
+ function recordDispatchUsage(execution, result) {
95
+ const { inputTokens, outputTokens, reasoningTokens } = result.usage ?? {};
96
+ const reported = [inputTokens, outputTokens, reasoningTokens].filter((count) => count !== undefined);
97
+ emitLlmUsage({
98
+ outcome: result.ok ? "success" : "error",
99
+ modelSource: "configured",
100
+ ...(execution.request.model?.resolved ? { model: execution.request.model.resolved } : {}),
101
+ durationMs: result.durationMs,
102
+ ...(inputTokens !== undefined ? { promptTokens: inputTokens } : {}),
103
+ ...(outputTokens !== undefined ? { completionTokens: outputTokens } : {}),
104
+ ...(reasoningTokens !== undefined ? { reasoningTokens } : {}),
105
+ ...(reported.length > 0 ? { totalTokens: reported.reduce((sum, count) => sum + count, 0) } : {}),
106
+ ...(result.ok ? {} : { errorCode: (result.reason && USAGE_ERROR_CODES[result.reason]) ?? "unknown_error" }),
107
+ });
108
+ }
77
109
  async function dispatchRunner(runner, prompt, opts, seams, llm) {
78
110
  const envSource = opts.envSource ?? process.env;
79
111
  const secrets = collectDispatchSensitiveValues(runner, opts, envSource);
@@ -103,9 +135,70 @@ async function dispatchRunner(runner, prompt, opts, seams, llm) {
103
135
  }
104
136
  return redactResult(result, collectSensitiveValues(secrets));
105
137
  }
106
- /** Run a built execution. Credentials are read here, once per call, and never returned. */
138
+ /** The git repository a directory is inside, if any: a `.git` file, or a `.git` directory with a HEAD. */
139
+ function enclosingGitRepository(dir) {
140
+ for (let current = fs.realpathSync(dir);; current = path.dirname(current)) {
141
+ const marker = path.join(current, ".git");
142
+ if (fs.existsSync(path.join(marker, "HEAD")) || (fs.existsSync(marker) && fs.statSync(marker).isFile())) {
143
+ return current;
144
+ }
145
+ if (path.dirname(current) === current)
146
+ return undefined;
147
+ }
148
+ }
149
+ /**
150
+ * The scratch working directory for one model-work dispatch on an agent or SDK
151
+ * engine, so the edit the model-work tool policy grants never reaches the
152
+ * stash. opencode counts a whole git repository as inside its working
153
+ * directory, so a scratch directory inside one is refused.
154
+ */
155
+ function createModelWorkDirectory() {
156
+ const dir = fs.mkdtempSync(path.join(os.tmpdir(), "akm-model-work-"));
157
+ const repository = enclosingGitRepository(dir);
158
+ if (repository === undefined)
159
+ return dir;
160
+ fs.rmSync(dir, { recursive: true, force: true });
161
+ throw new ConfigError(`Model work runs an agent in a scratch directory outside any git repository, but ${dir} is inside the repository at ${repository}.`, "INVALID_CONFIG_FILE", "Point TMPDIR at a directory outside any git repository.");
162
+ }
163
+ /**
164
+ * Run a built execution. Credentials are read here, once per call, and never
165
+ * returned. Model work on an agent or SDK engine runs in a scratch working
166
+ * directory that akm creates for the dispatch and removes after it, and must
167
+ * end with an answer: an agent that stops with none (opencode at its step
168
+ * limit, for one) has failed with `parse_error`.
169
+ */
107
170
  export async function runExecution(execution, options = {}) {
108
- const opts = { ...execution.options };
171
+ const scratch = execution.runner.kind !== "llm" && isModelWorkTools(execution.request.tools)
172
+ ? createModelWorkDirectory()
173
+ : undefined;
174
+ try {
175
+ return await runBuiltExecution(execution, options, scratch);
176
+ }
177
+ finally {
178
+ if (scratch)
179
+ fs.rmSync(scratch, { recursive: true, force: true });
180
+ }
181
+ }
182
+ /**
183
+ * A model-work reply's answer: the harness's result extractor strips its
184
+ * framing (claude's `--output-format json` result envelope, for one), as a
185
+ * workflow unit's does, and no answer is a `parse_error`.
186
+ */
187
+ function modelWorkAnswer(runner, result) {
188
+ const extractor = runner.kind === "agent" ? getHarness(runner.profile.platform ?? runner.profile.name)?.resultExtractor : undefined;
189
+ const extracted = extractor ? extractor(result) : { text: result.stdout };
190
+ const answer = {
191
+ ...result,
192
+ stdout: extracted.text,
193
+ ...(extracted.sessionId ? { sessionId: extracted.sessionId } : {}),
194
+ };
195
+ if (extracted.text.trim() !== "")
196
+ return answer;
197
+ return { ...answer, ok: false, reason: "parse_error", error: `Engine "${runner.engine}" returned no answer.` };
198
+ }
199
+ /** `scratch` is the model-work working directory, set only for model work on an agent or SDK engine. */
200
+ async function runBuiltExecution(execution, options, scratch) {
201
+ const opts = { ...execution.options, ...(scratch ? { cwd: scratch } : {}) };
109
202
  const operational = options.runOptions ?? {};
110
203
  for (const key of OPERATIONAL_OPTIONS) {
111
204
  if (operational[key] !== undefined)
@@ -140,7 +233,13 @@ export async function runExecution(execution, options = {}) {
140
233
  };
141
234
  }
142
235
  };
143
- return dispatchRunner(execution.runner, execution.prompt, opts, options, llm);
236
+ let result = await dispatchRunner(execution.runner, execution.prompt, opts, options, llm);
237
+ if (scratch !== undefined && result.ok)
238
+ result = modelWorkAnswer(execution.runner, result);
239
+ // The LLM transport records each HTTP attempt itself.
240
+ if (execution.runner.kind !== "llm")
241
+ recordDispatchUsage(execution, result);
242
+ return result;
144
243
  }
145
244
  /** `akm agent`: launch an agent engine interactively, with no prompt of its own. */
146
245
  export async function executeInteractiveAgentInvocation(input, seams = {}) {
@@ -67,6 +67,12 @@ export function decodeFrozenRunnerSpec(value) {
67
67
  export function runnerIsLlm(runner) {
68
68
  return runner.kind === "llm";
69
69
  }
70
- export function runnerSupportsFileWrite(runner) {
71
- return runner.kind !== "llm";
70
+ /**
71
+ * The LLM connection behind a runner, whatever its kind: an llm runner's own,
72
+ * an sdk runner's provider fallback, none for an agent CLI.
73
+ */
74
+ export function runnerLlmConnection(runner) {
75
+ if (runner.kind === "llm")
76
+ return runner.connection;
77
+ return runner.kind === "sdk" ? runner.fallbackConnection : undefined;
72
78
  }
@@ -39,13 +39,11 @@
39
39
  * - **schema** — the matrix places Aider in the "via prompt+validate" tier
40
40
  * with *no* structured output mode at all (plan §"Structured-output
41
41
  * normalization", tier "none"): there is no schema flag and no JSON output
42
- * flag, so the JSON Schema is injected into the message payload using the
43
- * exact directive wording of the engine's prompt assembly
44
- * (`step-work.ts` `buildUnitPrompt`) and the pi builder, so all
45
- * dispatch paths speak one dialect. Downstream, embedded-JSON extraction +
46
- * the engine's shared retry-until-valid loop supply the validation Aider
47
- * lacks. No temp schema file is written — that is Codex's native-schema
48
- * mechanism (`--output-schema`), which Aider does not have.
42
+ * flag, so the JSON Schema reaches it only as the instruction the shared
43
+ * request lowering appends to the prompt. Downstream, embedded-JSON
44
+ * extraction + the engine's shared retry-until-valid loop supply the
45
+ * validation Aider lacks. No temp schema file is written — that is Codex's
46
+ * native-schema mechanism (`--output-schema`), which Aider does not have.
49
47
  * - **tools** — deliberately unconsumed. Aider has no per-tool allowlist
50
48
  * flag; tool-ish behaviour is governed by its own switches (`--yes-always`,
51
49
  * git integration, shell-command confirmation). A restrictive policy is
@@ -57,41 +55,32 @@
57
55
  * durable source of truth; resume works even against a harness with no
58
56
  * session model (plan §"Session, MCP, and identity across harnesses" —
59
57
  * Aider is the plan's named example).
60
- * - **effort** — stays unconsumed (reserved; the shared request contract's
61
- * "no builder consumes it yet" note stays true).
58
+ * - **inference** — not translated: the shared lowering reports each field of
59
+ * the request's inference as untranslated (`inference` in `harnesses/ids.ts`
60
+ * lists none for this harness).
62
61
  *
63
62
  * Registered: `aiderBuilder` is `AiderHarness.agentBuilder` (`./index.ts`),
64
63
  * one of the ten harnesses `HARNESS_REGISTRY` constructs
65
64
  * (`harnesses/index.ts`); `agent/builders.ts` derives `BUILTIN_BUILDERS` from
66
65
  * that registry, so this builder is reachable under the `"aider"` platform
67
- * name without any further wiring. The registry entry also declares
68
- * `structuredOutput: "none"` alongside it (`./index.ts`).
66
+ * name without any further wiring.
69
67
  */
70
68
  import { resolveDispatchModel } from "../../agent/builder-shared.js";
71
69
  import { createAgentRequestLowerer } from "../../agent/request-lowering.js";
72
70
  /** Canonical harness/platform id used for model-alias resolution. */
73
71
  export const AIDER_PLATFORM = "aider";
74
- /**
75
- * Assemble the `--message` payload: optional system text, the task prompt,
76
- * and — when a schema is requested — the same schema directive the workflow
77
- * engine's prompt assembly uses (Aider has no native structured output, so
78
- * the prompt is the only channel; plan §"Structured-output normalization",
79
- * tier "none").
80
- */
72
+ /** Assemble the `--message` payload: optional system text, then the task prompt. */
81
73
  function buildMessagePayload(req) {
82
74
  const sections = [];
83
75
  if (req.systemPrompt)
84
76
  sections.push(req.systemPrompt);
85
77
  sections.push(req.prompt);
86
- if (req.schema) {
87
- sections.push(`Respond with ONLY a JSON value matching this JSON Schema (no prose, no code fences):\n${JSON.stringify(req.schema)}`);
88
- }
89
78
  return sections.join("\n\n");
90
79
  }
91
80
  /**
92
81
  * Aider builder.
93
82
  * Command shape:
94
- * aider [--model <m>] --yes-always --no-pretty --message=<[system\n\n]prompt[\n\nschema directive]>
83
+ * aider [--model <m>] --yes-always --no-pretty --message=<[system\n\n]prompt>
95
84
  */
96
85
  export const aiderBuilder = {
97
86
  platform: AIDER_PLATFORM,
@@ -100,7 +89,6 @@ export const aiderBuilder = {
100
89
  adapter: AIDER_PLATFORM,
101
90
  personaChannel: "prompt",
102
91
  tools: "none",
103
- outputSchema: true,
104
92
  }),
105
93
  build(profile, req) {
106
94
  const args = [...profile.args];
@@ -29,11 +29,6 @@ export class AiderHarness extends BaseHarness {
29
29
  agentBuilder = aiderBuilder;
30
30
  resultExtractor = aiderResultExtractor;
31
31
  // ── Workflow-engine descriptor (plan §"Capability matrix", P2) ────────────
32
- // akm spawns the `aider` CLI locally per unit ⇒ local-runner.
33
- pattern = "local-runner";
34
- // No structured-output mode at all (the matrix's "none — parse output"):
35
- // akm injects the schema into the prompt and extracts embedded JSON.
36
- structuredOutput = "none";
37
32
  // No flag-shaped resume: Aider persists context in chat-history files
38
33
  // (`.aider.chat.history.md`), not session ids — the plan's named example of
39
34
  // a harness with no session model. akm's `workflow_run_units` remains the
@@ -35,9 +35,8 @@
35
35
  * blank line.
36
36
  * - **schema** — the matrix places Q in the NO-structured-output tier
37
37
  * ("via prompt+validate": *(none documented)* — there is no `--json` or
38
- * `--output-format` to ask for). The JSON Schema is therefore passed
39
- * through the prompt: a directive matching the engine's wording
40
- * (`step-work.ts` `buildUnitPrompt`) is appended to the payload.
38
+ * `--output-format` to ask for). The JSON Schema therefore reaches it only
39
+ * as the instruction the shared request lowering appends to the prompt.
41
40
  * Stdout stays plain text; `./result-extractor.ts` strips terminal framing
42
41
  * and the engine's shared embedded-JSON parse + retry-until-valid loop does
43
42
  * the rest. No schema temp file is written — that seam is codex-only
@@ -51,16 +50,15 @@
51
50
  * falling back to `--trust-all-tools` (never silently widen a restriction)
52
51
  * — Q then refuses untrusted tool actions in non-interactive mode, which is
53
52
  * the conservative failure mode.
54
- * - **effort** — stays unconsumed (reserved; the shared request contract's
55
- * "no builder consumes it yet" note stays true).
53
+ * - **inference** — not translated: the shared lowering reports each field of
54
+ * the request's inference as untranslated (`inference` in `harnesses/ids.ts`
55
+ * lists none for this harness).
56
56
  *
57
57
  * Registered: `amazonqBuilder` is `AmazonqHarness.agentBuilder`
58
58
  * (`./index.ts`), one of the ten harnesses `HARNESS_REGISTRY` constructs
59
59
  * (`harnesses/index.ts`); `agent/builders.ts` derives `BUILTIN_BUILDERS` from
60
60
  * that registry, so this builder is reachable under the `"amazonq"` platform
61
- * name without any further wiring. The registry-side capability entry —
62
- * pattern `local-runner`, structuredOutput `none` — is declared alongside it
63
- * (`./index.ts`).
61
+ * name without any further wiring.
64
62
  */
65
63
  import { resolveDispatchModel } from "../../agent/builder-shared.js";
66
64
  import { createAgentRequestLowerer } from "../../agent/request-lowering.js";
@@ -84,27 +82,19 @@ function toolPolicyEntries(tools) {
84
82
  }
85
83
  return undefined;
86
84
  }
87
- /**
88
- * Assemble the positional prompt payload: optional system prompt, the task
89
- * prompt, and — when a schema is requested — the same schema directive the
90
- * workflow engine's prompt assembly uses, so both dispatch paths speak one
91
- * dialect.
92
- */
85
+ /** Assemble the positional prompt payload: optional system prompt, then the task prompt. */
93
86
  function buildPromptPayload(req) {
94
87
  const sections = [];
95
88
  if (req.systemPrompt)
96
89
  sections.push(req.systemPrompt);
97
90
  sections.push(req.prompt);
98
- if (req.schema) {
99
- sections.push(`Respond with ONLY a JSON value matching this JSON Schema (no prose, no code fences):\n${JSON.stringify(req.schema)}`);
100
- }
101
91
  return sections.join("\n\n");
102
92
  }
103
93
  /**
104
94
  * Amazon Q Developer CLI builder.
105
95
  * Command shape:
106
96
  * q chat --no-interactive (--trust-all-tools | --trust-tools=<t1,t2>)
107
- * [--model <m>] -- "<systemPrompt?\n\nprompt\n\nschema directive?>"
97
+ * [--model <m>] -- "<systemPrompt?\n\nprompt>"
108
98
  */
109
99
  export const amazonqBuilder = {
110
100
  platform: AMAZONQ_PLATFORM,
@@ -113,7 +103,6 @@ export const amazonqBuilder = {
113
103
  adapter: AMAZONQ_PLATFORM,
114
104
  personaChannel: "prompt",
115
105
  tools: "flat",
116
- outputSchema: true,
117
106
  }),
118
107
  build(profile, req) {
119
108
  // Built-in q profiles would ship `args: []`; headless dispatch is the
@@ -141,8 +130,8 @@ export const amazonqBuilder = {
141
130
  const resolved = resolveDispatchModel(req, profile, AMAZONQ_PLATFORM);
142
131
  args.push("--model", resolved);
143
132
  }
144
- // No system-prompt / schema flags exist on `q chat` — both travel in the
145
- // positional payload, after the end-of-options separator.
133
+ // No system-prompt flag exists on `q chat` — it travels in the positional
134
+ // payload, after the end-of-options separator.
146
135
  args.push("--");
147
136
  args.push(buildPromptPayload(req));
148
137
  return { argv: [profile.bin, ...args] };