akm-cli 0.9.24 → 0.9.25-alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/CHANGELOG.md +173 -0
  2. package/dist/cli.js +1 -1
  3. package/dist/commands/health/checks.js +10 -11
  4. package/dist/commands/improve/consolidate/pair-pass.js +4 -2
  5. package/dist/commands/improve/consolidate.js +10 -4
  6. package/dist/commands/improve/execution.js +4 -11
  7. package/dist/commands/improve/extract-prompt.js +4 -4
  8. package/dist/commands/improve/extract.js +11 -13
  9. package/dist/commands/improve/improve-cli.js +65 -34
  10. package/dist/commands/improve/improve-strategies.js +49 -43
  11. package/dist/commands/improve/improve-usage-report.js +8 -17
  12. package/dist/commands/improve/loop-stages.js +3 -0
  13. package/dist/commands/improve/preparation.js +3 -1
  14. package/dist/commands/improve/reflect-noise.js +125 -0
  15. package/dist/commands/improve/reflect.js +105 -172
  16. package/dist/commands/improve/retrieval-gate.js +7 -2
  17. package/dist/commands/improve/stage.js +67 -24
  18. package/dist/commands/proposal/drain.js +11 -2
  19. package/dist/commands/proposal/proposal-cli.js +1 -5
  20. package/dist/commands/proposal/propose-cli.js +2 -2
  21. package/dist/commands/proposal/propose.js +72 -84
  22. package/dist/commands/proposal/validators/proposal-quality-validators.js +4 -2
  23. package/dist/commands/read/search-cli.js +0 -38
  24. package/dist/commands/remember.js +3 -3
  25. package/dist/commands/sources/schema-repair.js +1 -1
  26. package/dist/core/config/config-schema.js +37 -60
  27. package/dist/core/config/engine-semantics.js +15 -11
  28. package/dist/core/config/schema/improve-processes.js +18 -2
  29. package/dist/core/improve-result.js +3 -3
  30. package/dist/core/redaction.js +4 -0
  31. package/dist/core/spawn-env.js +25 -0
  32. package/dist/core/structured.js +11 -1
  33. package/dist/execution/source.js +10 -0
  34. package/dist/indexer/passes/memory-inference.js +2 -1
  35. package/dist/integrations/agent/builder-shared.js +15 -0
  36. package/dist/integrations/agent/config.js +1 -1
  37. package/dist/integrations/agent/engine-resolution.js +13 -31
  38. package/dist/integrations/agent/execution.js +48 -22
  39. package/dist/integrations/agent/index.js +1 -1
  40. package/dist/integrations/agent/profiles.js +2 -2
  41. package/dist/integrations/agent/prompts.js +55 -114
  42. package/dist/integrations/agent/request-lowering.js +21 -8
  43. package/dist/integrations/agent/runner-dispatch.js +96 -3
  44. package/dist/integrations/agent/runner.js +8 -2
  45. package/dist/integrations/harnesses/aider/agent-builder.js +10 -23
  46. package/dist/integrations/harnesses/aider/index.js +0 -5
  47. package/dist/integrations/harnesses/amazonq/agent-builder.js +9 -21
  48. package/dist/integrations/harnesses/amazonq/index.js +0 -5
  49. package/dist/integrations/harnesses/claude/agent-builder.js +47 -36
  50. package/dist/integrations/harnesses/claude/index.js +0 -14
  51. package/dist/integrations/harnesses/codex/agent-builder.js +3 -4
  52. package/dist/integrations/harnesses/codex/index.js +0 -4
  53. package/dist/integrations/harnesses/copilot/agent-builder.js +4 -13
  54. package/dist/integrations/harnesses/copilot/index.js +2 -7
  55. package/dist/integrations/harnesses/gemini/agent-builder.js +4 -13
  56. package/dist/integrations/harnesses/gemini/index.js +0 -5
  57. package/dist/integrations/harnesses/ids.js +12 -10
  58. package/dist/integrations/harnesses/opencode/agent-builder.js +41 -9
  59. package/dist/integrations/harnesses/opencode/index.js +0 -8
  60. package/dist/integrations/harnesses/opencode/model-config.js +33 -0
  61. package/dist/integrations/harnesses/opencode/model-work-agent.js +115 -0
  62. package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -7
  63. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +114 -28
  64. package/dist/integrations/harnesses/openhands/agent-builder.js +7 -21
  65. package/dist/integrations/harnesses/openhands/index.js +0 -5
  66. package/dist/integrations/harnesses/pi/agent-builder.js +7 -21
  67. package/dist/integrations/harnesses/pi/index.js +0 -5
  68. package/dist/llm/client.js +5 -0
  69. package/dist/llm/feature-gate.js +12 -9
  70. package/dist/llm/index-passes.js +2 -5
  71. package/dist/llm/memory-infer.js +6 -5
  72. package/dist/llm/structured-call.js +33 -11
  73. package/dist/output/shapes/passthrough.js +1 -0
  74. package/dist/scripts/akm-migrate-node.js +298 -228
  75. package/dist/scripts/akm-migrate.js +298 -228
  76. package/dist/workflows/exec/step-work.js +6 -5
  77. package/dist/workflows/exec/unit-dispatch.js +4 -13
  78. package/dist/workflows/freeze/step-values.js +1 -1
  79. package/docs/reference/cli.md +41 -18
  80. package/docs/reference/configuration.md +165 -12
  81. package/docs/reference/data-and-telemetry.md +2 -3
  82. package/docs/reference/workflow-schema.md +10 -9
  83. package/package.json +1 -1
  84. package/schemas/akm-config.json +108 -0
@@ -4,13 +4,14 @@
4
4
  import { getImproveProcessConfig } from "../../core/config/config.js";
5
5
  import { ConfigError } from "../../core/errors.js";
6
6
  import { parseEmbeddedJsonResponse } from "../../core/parse.js";
7
+ import { defaultFeedback } from "../../core/structured.js";
7
8
  import { warn } from "../../core/warn.js";
8
- import { LlmCallError } from "../../llm/client.js";
9
- import { callStructured } from "../../llm/structured-call.js";
9
+ import { runnerLlmConnection } from "../../integrations/agent/runner.js";
10
+ import { callStructured, dispatchFailureReason, dispatchFailureResult, } from "../../llm/structured-call.js";
10
11
  import { currentLlmStage, withLlmStage } from "../../llm/usage-telemetry.js";
11
12
  import { isProceduralRejection } from "../proposal/proposal-types.js";
12
13
  import { createProposal, listProposalsReadOnly, proposalContentHash, recordGateDecision, } from "../proposal/repository.js";
13
- import { resolveImproveLlmExecution } from "./execution.js";
14
+ import { resolveImproveExecution } from "./execution.js";
14
15
  /** Normalize an unknown thrown value to a message. */
15
16
  export function errMessage(e) {
16
17
  return e instanceof Error ? e.message : String(e);
@@ -27,13 +28,13 @@ export function noticeSet(forward) {
27
28
  return { add, list, fields: () => (byKey.size > 0 ? { notices: list() } : {}) };
28
29
  }
29
30
  /**
30
- * A stage's LLM runner: the one the improve plan froze for it (an own
31
+ * A stage's runner: the one the improve plan froze for it (an own
31
32
  * `llmRunner` key, `null` meaning "none"), else the process engine cascade.
32
33
  */
33
34
  export function stageRunner(frozen, config, profile, processName, onNotices) {
34
35
  if (Object.hasOwn(frozen, "llmRunner"))
35
36
  return frozen.llmRunner ?? undefined;
36
- const resolved = resolveImproveLlmExecution({
37
+ const resolved = resolveImproveExecution({
37
38
  config,
38
39
  profile,
39
40
  process: getImproveProcessConfig(processName, profile),
@@ -44,10 +45,38 @@ export function stageRunner(frozen, config, profile, processName, onNotices) {
44
45
  return resolved?.runner;
45
46
  }
46
47
  /**
47
- * One model call. Provider trouble (transport error, timeout, a disabled
48
- * feature) comes back as `{ ok: false }`; only a configuration failure throws.
48
+ * One model call. Provider trouble (transport error, timeout, abort, a
49
+ * disabled feature) comes back as `{ ok: false }`; only a configuration
50
+ * failure throws. A reply to a call with `request.responseSchema` that the
51
+ * stage's own `parse` rejects gets one corrective retry; the last reply comes
52
+ * back either way, and the caller parses it again.
49
53
  */
50
54
  export async function callStage(call) {
55
+ const reply = await callStageOnce(call);
56
+ if (!reply.ok || !call.request?.responseSchema)
57
+ return reply;
58
+ if ((call.parse ?? parseEmbeddedJsonResponse)(reply.raw) !== undefined)
59
+ return reply;
60
+ const feedback = defaultFeedback({ reason: "parse_error", errors: [] });
61
+ const retry = await callStageOnce({ ...call, prompt: `${call.prompt}\n\n${feedback}` });
62
+ // A retry that fails in transport keeps the first reply, which the caller may still accept.
63
+ return retry.ok ? retry : reply;
64
+ }
65
+ /** Timeout and abort come from the dispatch's own reason, whatever the runner's kind. */
66
+ function failureReason(err) {
67
+ const reason = dispatchFailureReason(err);
68
+ return reason === "timeout" || reason === "aborted" ? reason : "error";
69
+ }
70
+ /** A failed call: its reason and message, and the dispatch's own result when it reached a transport. */
71
+ function failedCall(err) {
72
+ const result = dispatchFailureResult(err);
73
+ return { ok: false, reason: failureReason(err), error: errMessage(err), ...(result ? { result } : {}) };
74
+ }
75
+ /**
76
+ * One dispatch with no validation, for a caller that parses and repairs the
77
+ * reply itself (reflect's repair turn, extract's own structured loop).
78
+ */
79
+ export async function callStageOnce(call) {
51
80
  const messages = [
52
81
  ...(call.system ? [{ role: "system", content: call.system }] : []),
53
82
  ...(call.history ?? []),
@@ -63,10 +92,11 @@ export async function callStage(call) {
63
92
  runner: call.runner,
64
93
  messages,
65
94
  ...(call.request ? { request: call.request } : {}),
95
+ ...(call.current ? { current: call.current } : {}),
66
96
  ...(call.onNotices ? { onNotices: call.onNotices } : {}),
67
97
  parse: (r) => r ?? "",
68
98
  onError: (_cls, err) => {
69
- failure = { ok: false, reason: "error", error: errMessage(err) };
99
+ failure = failedCall(err);
70
100
  return undefined;
71
101
  },
72
102
  fallback: undefined,
@@ -85,8 +115,7 @@ export async function callStage(call) {
85
115
  catch (err) {
86
116
  if (err instanceof ConfigError)
87
117
  throw err;
88
- const timedOut = err instanceof LlmCallError && err.code === "timeout";
89
- return { ok: false, reason: timedOut ? "timeout" : "error", error: errMessage(err) };
118
+ return failedCall(err);
90
119
  }
91
120
  }
92
121
  /** Attribute a stage's LLM calls to its process and planned engine (the usage report). */
@@ -162,9 +191,10 @@ export function stageJudgedProposal(stash, proposal, judged, proposalsCtx) {
162
191
  * The judge a quality gate names for itself (#1011): the gate's `engine`,
163
192
  * `model`, `timeoutMs` and `llm` over the process's own settings, as
164
193
  * `processes.triage.judgment` resolves over triage. `undefined` when the gate
165
- * is off or sets none of them, so the caller keeps its own judge. Throws when
166
- * they resolve to no LLM engine, before anything is generated: a judge must be
167
- * one, and the gate never falls back to another.
194
+ * is off or sets none of them, so the caller keeps its own judge. The engine
195
+ * may be of any kind; config validation has required one that confines the
196
+ * model-work tool policy. Throws when they resolve to no engine at all,
197
+ * before anything is generated: the gate never falls back to another judge.
168
198
  */
169
199
  export function resolveQualityGateJudge(config, profile, processName, onNotices) {
170
200
  const process = profile?.processes?.[processName];
@@ -173,7 +203,7 @@ export function resolveQualityGateJudge(config, profile, processName, onNotices)
173
203
  return undefined;
174
204
  if (!["engine", "model", "timeoutMs", "llm"].some((key) => Object.hasOwn(gate, key)))
175
205
  return undefined;
176
- const resolved = resolveImproveLlmExecution({
206
+ const resolved = resolveImproveExecution({
177
207
  config,
178
208
  processName: `${processName}-quality-judge`,
179
209
  ...(profile ? { profile } : {}),
@@ -181,7 +211,7 @@ export function resolveQualityGateJudge(config, profile, processName, onNotices)
181
211
  current: gate,
182
212
  });
183
213
  if (!resolved) {
184
- throw new ConfigError(`The ${processName} quality gate's judge must be an LLM engine. Set processes.${processName}.qualityGate.engine to one.`, "INVALID_CONFIG_FILE");
214
+ throw new ConfigError(`The ${processName} quality gate's judge has no engine. Set processes.${processName}.qualityGate.engine.`, "INVALID_CONFIG_FILE");
185
215
  }
186
216
  onNotices?.(resolved.notices);
187
217
  return resolved.runner;
@@ -232,15 +262,24 @@ function buildChangedRegion(sourceContent, candidateContent) {
232
262
  const added = candidate.slice(prefix, candidate.length - suffix).join("\n");
233
263
  return boundedDocument(`Removed or replaced:\n${removed || "(none)"}\n\nAdded or replacement:\n${added || "(none)"}`);
234
264
  }
235
- /** Judge prompt for an in-place revision. */
236
- export function buildReflectJudgePrompt(candidateContent, sourceContent, feedback) {
265
+ /**
266
+ * What the judge may do with tools when it runs on an agent engine: verify a
267
+ * fact the revision adds or alters, and nothing else. The plain judge's prompt
268
+ * is unchanged (its rubric is tuned and measured without this paragraph).
269
+ */
270
+ function reflectJudgeToolRules(ref) {
271
+ const asset = ref ? `The asset is \`${ref}\`: read it with akm_show, ` : "Read an asset with akm_show ";
272
+ return `Tools: ${asset}or an asset the changed region names, only to verify a fact the revision adds or alters; the text above already shows every change. Do not search, do not read anything else, and do not use a tool to judge structure or wording. One or two reads at most. Before scoring, check three lists: (1) every statement the revision adds: find each in the asset, or as a fact the feedback states about the subject, and score QUALITY 1-2 if any is in neither; a statement is found only when the asset or the feedback says it, in any words: a new step, cause, consequence or detail that merely seems to follow is not found; feedback says what to fix and is not content, so an added statement about how the asset was used, found or verified is unsupported; (2) every fact, caveat and field of the source: find each in the revision, and score PRESERVATION 1-3 if any is missing; (3) every point the feedback makes: find the text it is about changed in the revision, and score NEED 2-3 if any is not; a note that restates the feedback does not address it. A read that finds nothing wrong raises no score above what these lists support. Then reply with the JSON.`;
273
+ }
274
+ /** Judge prompt for an in-place revision. `tools` is set when the judge runs on an agent engine. */
275
+ export function buildReflectJudgePrompt(candidateContent, sourceContent, feedback, tools) {
237
276
  return [
238
277
  "You are evaluating a proposed revision to an existing akm asset.",
239
278
  "",
240
279
  "Score this revision on each criterion from 1 (poor) to 5 (excellent):",
241
- "1. NEED: Does the revision fix a concrete problem in the source? Concrete problems are: something the feedback reports as wrong or missing; a factual error; or broken, garbled, truncated or missing text, including frontmatter fields such as description or when_to_use. Score 4-5 when it fixes one, even a small one. Score 1-2 when the source was already correct and the revision only rewords, restates, reformats, or adds headings, an introduction or a table of contents.",
242
- "2. PRESERVATION: Does it keep every concrete fact, identifier, command, path, number and example from the source, without truncation?",
243
- "3. QUALITY: Is it coherent and accurate, with no claims, steps or details that the source or the feedback does not support?",
280
+ "1. NEED: Does the revision fix a concrete problem in the source? Concrete problems are: something the feedback reports as wrong or missing; a factual error; broken, garbled, truncated or missing text; and a missing or broken title, description or when_to_use field. Compare the source's description with the revision's: a description with a sentence split in its middle by a stray period or line-wrap artifact (as in 'calls. asset writes'), an unbalanced or escaped quote, or a truncated ending is broken, and repairing it is a concrete problem fixed even when the rest of the revision only adds stamps or reformats; adding or removing a trailing period, or rewording a readable description, repairs nothing. A missing title, description or when_to_use is a concrete problem whether or not the feedback mentions it: empty or positive feedback does not mean the source was complete, and the frontmatter is complete only when it has all three. A when_to_use is missing when the frontmatter has none, even if the body has a 'when to use' section; moving or copying that text into the field is the fix. A missing type field or a provenance stamp such as generated or verified is not a concrete problem. Score 4-5 when the revision fixes one, even a small one, whatever else it also reformats. Score 1-2 when the source was already complete and correct and the revision only rewords, restates, reformats, adds a type field or a stamp, or adds headings, an introduction or a table of contents.",
281
+ "2. PRESERVATION: Does it keep every concrete fact, identifier, command, path, number, example, caveat and frontmatter field from the source, without truncation? Check the changed region line by line. Score 1-3 when any of them is dropped or weakened, even when the revision also fixes something. Trimming narrative that states no fact, or removing what the feedback asks to remove or rescope, is not a drop.",
282
+ "3. QUALITY: Is it coherent and accurate, with no claims, steps or details that the source or the feedback does not support? Check each added or changed statement against the source and the feedback: a statement that follows from either counts as supported, and frontmatter stamps such as type, generated, verified or quality, whatever their values, and a restatement of existing content are not claims. Score 1-2 when the revision adds a claim neither supports, turns a draft or proposal into a decision, strengthens a statement beyond the source (a preference into a requirement, a possibility into a fact, a pending fix into a done one), adds a hedge such as 'may be outdated', or a placeholder such as 'TODO' or 'verify'; a fix elsewhere in the revision does not raise this score.",
244
283
  "",
245
284
  "Feedback:",
246
285
  "```",
@@ -262,6 +301,7 @@ export function buildReflectJudgePrompt(candidateContent, sourceContent, feedbac
262
301
  buildChangedRegion(sourceContent, candidateContent),
263
302
  "```",
264
303
  "",
304
+ ...(tools ? [reflectJudgeToolRules(tools.ref), ""] : []),
265
305
  'Return ONLY valid JSON, no prose: {"scores": {"need": <1-5 integer>, "preservation": <1-5 integer>, "quality": <1-5 integer>}, "reason": "<one sentence>"}',
266
306
  ].join("\n");
267
307
  }
@@ -349,13 +389,13 @@ function judgeResponseSchema(keys) {
349
389
  */
350
390
  async function runQualityJudge(feature, config, prompt, keys, chat, options) {
351
391
  const resolved = !options.runnerSelectionFrozen && !options.llmRunner
352
- ? resolveImproveLlmExecution({ config, processName: `${feature}-judge` })
392
+ ? resolveImproveExecution({ config, processName: `${feature}-judge` })
353
393
  : null;
354
394
  if (resolved)
355
395
  options.onNotices?.(resolved.notices);
356
396
  const runner = options.llmRunner ?? resolved?.runner;
357
397
  if (!runner)
358
- return { pass: false, score: -1, reason: "no LLM configured — cannot judge, failing closed" };
398
+ return { pass: false, score: -1, reason: "no engine configured — cannot judge, failing closed" };
359
399
  const outcome = await callStage({
360
400
  feature,
361
401
  runner,
@@ -363,13 +403,14 @@ async function runQualityJudge(feature, config, prompt, keys, chat, options) {
363
403
  prompt,
364
404
  request: {
365
405
  // Off unless the judge's own engine enables thinking (a slower, separate judge engine, #1011).
366
- enableThinking: runner.connection.enableThinking === true,
406
+ enableThinking: runnerLlmConnection(runner)?.enableThinking === true,
367
407
  temperature: 0,
368
408
  responseSchema: judgeResponseSchema(keys),
369
409
  ...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
370
410
  ...(options.signal ? { signal: options.signal } : {}),
371
411
  ...(chat ? { chat } : {}),
372
412
  },
413
+ parse: (raw) => parseJudgeResponse(raw, keys),
373
414
  ...(options.onNotices ? { onNotices: options.onNotices } : {}),
374
415
  });
375
416
  if (!outcome.ok) {
@@ -411,6 +452,8 @@ export function runLessonQualityJudge(config, lessonContent, sourceContent, chat
411
452
  }
412
453
  /** Judge an in-place reflect revision without new-lesson novelty criteria. */
413
454
  export function runReflectQualityJudge(config, candidateContent, sourceContent, feedback, chat, options = {}) {
414
- const prompt = buildReflectJudgePrompt(candidateContent, sourceContent, feedback);
455
+ // A judge on an agent engine gets the tool rules; the runner is the frozen one or none.
456
+ const tools = options.llmRunner && options.llmRunner.kind !== "llm" ? { ref: options.ref } : undefined;
457
+ const prompt = buildReflectJudgePrompt(candidateContent, sourceContent, feedback, tools);
415
458
  return runQualityJudge("proposal_quality_gate", config, prompt, REFLECT_JUDGE_CRITERIA, chat, options);
416
459
  }
@@ -25,6 +25,7 @@ import { ConfigError } from "../../core/errors.js";
25
25
  import { appendEvent } from "../../core/events.js";
26
26
  import { escapeJsonStringControls, stripCodeFences, stripThinkBlocks } from "../../core/parse.js";
27
27
  import { info, warn } from "../../core/warn.js";
28
+ import { DEFAULT_LLM_TIMEOUT_MS } from "../../integrations/agent/config.js";
28
29
  import { buildExecution, resolveExecution } from "../../integrations/agent/execution.js";
29
30
  import { assertRunnerCredentials, runExecution, } from "../../integrations/agent/runner-dispatch.js";
30
31
  import { errMessage, noticeSet } from "../improve/stage.js";
@@ -170,7 +171,14 @@ export function parseJudgmentVerdict(raw) {
170
171
  async function dispatchJudgment(runner, prompt, seams) {
171
172
  let notices = [];
172
173
  try {
173
- const prepared = resolveExecution({ content: prompt, runner });
174
+ // Model work is bounded on every runner kind: one with no timeout of its own gets the default.
175
+ // The judgment runs under the model-work tool policy.
176
+ const prepared = resolveExecution({
177
+ content: prompt,
178
+ runner,
179
+ current: Object.hasOwn(runner, "timeoutMs") ? {} : { timeout: DEFAULT_LLM_TIMEOUT_MS },
180
+ modelWork: true,
181
+ });
174
182
  const lowered = buildExecution(prepared.request, prepared.runner);
175
183
  notices = lowered.notices;
176
184
  const chat = seams.chat;
@@ -330,10 +338,11 @@ export async function drainProposals(opts, promoteFn = akmProposalAccept, reject
330
338
  }
331
339
  }
332
340
  if (opts.judgment && result.deferred.length > 0) {
333
- // Symbolic credentials are checked before any gate, reject or promote.
341
+ // Symbolic credentials, and the runner's model-work tool policy, are checked before any gate, reject or promote.
334
342
  const prepared = resolveExecution({
335
343
  content: "Validate the selected proposal judgment runner before mutation.",
336
344
  runner: opts.judgment,
345
+ modelWork: true,
337
346
  });
338
347
  assertRunnerCredentials(buildExecution(prepared.request, prepared.runner).runner);
339
348
  }
@@ -18,7 +18,7 @@ import { parsePositiveIntFlag } from "../../cli/parse-args.js";
18
18
  import { defineGroupCommand, defineJsonCommand, output } from "../../cli/shared.js";
19
19
  import { resolveStashDir } from "../../core/common.js";
20
20
  import { loadConfig } from "../../core/config/config.js";
21
- import { ConfigError, UsageError } from "../../core/errors.js";
21
+ import { UsageError } from "../../core/errors.js";
22
22
  import { installLlmUsagePersistenceIfAbsent } from "../../llm/usage-persist.js";
23
23
  import { withLlmStage } from "../../llm/usage-telemetry.js";
24
24
  import { resolveImproveExecution } from "../improve/execution.js";
@@ -470,10 +470,6 @@ const proposalDrainCommand = defineJsonCommand({
470
470
  })
471
471
  : null;
472
472
  const judgment = judgmentResolution?.runner ?? null;
473
- const effectiveJudgmentLlm = triageConfig?.judgment?.llm ?? triageConfig?.llm ?? selectedStrategy.config.llm;
474
- if (judgment && judgment.kind !== "llm" && effectiveJudgmentLlm) {
475
- throw new ConfigError(`Triage judgment engine "${judgment.engine ?? "unknown"}" is an agent engine and cannot receive llm overrides.`, "INVALID_CONFIG_FILE");
476
- }
477
473
  // #576: persist + attribute per-call LLM usage for the standalone drain
478
474
  // path. `IfAbsent` keeps an enclosing `akm improve` sink in charge when
479
475
  // drain runs as a sub-step; the disposer clears only a sink we installed.
@@ -26,7 +26,7 @@ const EXIT_GENERAL = EXIT_CODES.GENERAL;
26
26
  export const proposeCommand = defineCommand({
27
27
  meta: {
28
28
  name: "new",
29
- description: "Ask the configured agent CLI to author a brand-new asset and queue it as a proposal",
29
+ description: "Ask the configured engine to author a brand-new asset as JSON and queue it as a proposal",
30
30
  },
31
31
  // Raw defineCommand: declare the global output flags so their space-separated
32
32
  // values are consumed rather than shifting the `type` / `name` positionals.
@@ -48,7 +48,7 @@ export const proposeCommand = defineCommand({
48
48
  task: { type: "string", description: "Task description for the agent (what should the asset do?)" },
49
49
  file: { type: "string", description: "Read the task or prompt text from a UTF-8 file" },
50
50
  engine: { type: "string", description: "Engine to use (defaults to defaults.engine)" },
51
- "timeout-ms": { type: "string", description: "Override the agent CLI timeout in milliseconds" },
51
+ "timeout-ms": { type: "string", description: "Override the engine timeout in milliseconds" },
52
52
  },
53
53
  async run({ args }) {
54
54
  await runWithJsonErrors(async () => {
@@ -2,17 +2,17 @@
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  /**
5
- * `akm propose <type> <name> --task ...` — proposal-producing agent
6
- * command (#226).
5
+ * `akm proposal new <type> <name> --task ...` — proposal-producing command
6
+ * (#226).
7
7
  *
8
- * Mirrors {@link akmReflect} but for fresh authoring. The agent receives a
9
- * task description plus per-asset-type schema hints and is asked to author
10
- * a brand-new asset payload. The output lands ONLY in the proposal queue.
8
+ * Mirrors {@link akmReflect} but for fresh authoring. The engine, of any kind,
9
+ * receives a task description plus per-asset-type schema hints and returns a
10
+ * brand-new asset payload as JSON on stdout. The output lands ONLY in the
11
+ * proposal queue.
11
12
  *
12
13
  * Failures use the same {@link AgentFailureReason} discriminants as
13
14
  * `akm reflect`. `propose_invoked` is emitted at command entry.
14
15
  */
15
- import fs from "node:fs";
16
16
  import { placementTypes, stashDirFor } from "../../core/asset/asset-placement.js";
17
17
  import { parseRefInput } from "../../core/asset/resolve-ref.js";
18
18
  import { resolveStashDir } from "../../core/common.js";
@@ -21,12 +21,13 @@ import { UsageError } from "../../core/errors.js";
21
21
  import { appendEvent } from "../../core/events.js";
22
22
  import { redactSensitiveText } from "../../core/redaction.js";
23
23
  import { resolveStandardsContext } from "../../core/standards/resolve-standards-context.js";
24
+ import { runStructured } from "../../core/structured.js";
24
25
  import { warn } from "../../core/warn.js";
25
26
  import { deriveEntryProvenance } from "../../indexer/installations.js";
26
27
  import { fallbackAnnouncement } from "../../integrations/agent/engine-fallback.js";
27
28
  import { buildExecution, resolveExecution } from "../../integrations/agent/execution.js";
28
- import { buildProposePrompt, parseAgentProposalPayload } from "../../integrations/agent/prompts.js";
29
- import { assertRunnerCredentials, collectDispatchSensitiveValues, runExecution, } from "../../integrations/agent/runner-dispatch.js";
29
+ import { buildProposePrompt, PROPOSAL_JSON_SCHEMA, validateProposalPayload, } from "../../integrations/agent/prompts.js";
30
+ import { assertRunnerCredentials, collectDispatchSensitiveValues, runExecution, unwrapHarnessReply, } from "../../integrations/agent/runner-dispatch.js";
30
31
  import { baseFailureFields, enoentHintMessage, isEnoentFailure } from "../agent/agent-support.js";
31
32
  import { createProposal, resolveProposalQueueTarget, } from "./repository.js";
32
33
  function failureEnvelope(result, type, name, engine, notices, fallbackReason = "non_zero_exit") {
@@ -42,42 +43,63 @@ function failureEnvelope(result, type, name, engine, notices, fallbackReason = "
42
43
  function noticeFields(notices) {
43
44
  return notices.length > 0 ? { notices } : {};
44
45
  }
45
- /** Resolve, lower, and dispatch the already-rendered proposal prompt. */
46
+ const DISPATCH_FAILED = Symbol("proposal-dispatch-failed");
47
+ /**
48
+ * Resolve, lower, and dispatch the already-rendered proposal prompt with the
49
+ * proposal's JSON Schema as its output schema, and capture the reply. A reply
50
+ * that is not a proposal gets one corrective retry.
51
+ */
46
52
  async function dispatchProposalPrompt(prompt, config, options, onDispatchReady) {
47
53
  const current = {
48
54
  ...(options.engine !== undefined ? { engine: options.engine } : {}),
49
55
  ...(options.timeoutMs !== undefined ? { timeout: options.timeoutMs } : {}),
56
+ outputSchema: PROPOSAL_JSON_SCHEMA,
50
57
  };
51
- const prepared = resolveExecution({
52
- content: prompt,
53
- config,
54
- ...(Object.keys(current).length > 0 ? { current } : {}),
55
- });
58
+ const lower = (content) => {
59
+ const prepared = resolveExecution({ content, config, current });
60
+ return { prepared, lowered: buildExecution(prepared.request, prepared.runner) };
61
+ };
62
+ const { prepared, lowered } = lower(prompt);
56
63
  const engineName = prepared.request.engine.name;
57
64
  const announcement = fallbackAnnouncement(prepared.fallbackEngineName, engineName);
58
65
  if (announcement)
59
66
  warn(announcement);
60
- const lowered = buildExecution(prepared.request, prepared.runner);
61
- const interactive = !options.runAgentOptions?.spawn;
62
- const runOptions = {
63
- stdio: interactive ? "interactive" : "captured",
64
- parseOutput: "text",
65
- ...(options.runAgentOptions ?? {}),
66
- };
67
67
  // Validate every required symbolic credential before the entry event opens
68
68
  // durable state. Provider/runtime failures still occur after the event,
69
- // preserving the command-attempt observability contract.
70
- assertRunnerCredentials(lowered.runner, runOptions.envSource);
69
+ // preserving the command-attempt observability contract. The dispatch reads
70
+ // the same caller environment.
71
+ const envSource = options.runAgentOptions?.envSource;
72
+ assertRunnerCredentials(lowered.runner, envSource);
71
73
  onDispatchReady();
72
74
  options.onDispatchReady?.();
73
- const result = await runExecution(lowered, { runOptions });
75
+ // runStructured dispatches at least once, so a returned or DISPATCH_FAILED exchange has a result.
76
+ const results = [];
77
+ let reply;
78
+ try {
79
+ reply = await runStructured({
80
+ dispatch: async (feedback) => {
81
+ const execution = feedback ? lower(`${prompt}\n\n${feedback}`).lowered : lowered;
82
+ const result = await runExecution(execution, { runOptions: options.runAgentOptions ?? {} });
83
+ results.push(result);
84
+ if (!result.ok)
85
+ throw DISPATCH_FAILED;
86
+ return unwrapHarnessReply(execution.runner, result).text;
87
+ },
88
+ validate: validateProposalPayload,
89
+ });
90
+ }
91
+ catch (err) {
92
+ if (err !== DISPATCH_FAILED)
93
+ throw err;
94
+ }
74
95
  return {
75
- result,
96
+ result: results.at(-1),
97
+ ...(reply ? { reply } : {}),
98
+ durationMs: results.reduce((total, result) => total + result.durationMs, 0),
76
99
  engineName,
77
100
  ...(lowered.runner.kind === "llm" ? {} : { engineBin: lowered.runner.profile.bin }),
78
101
  notices: lowered.notices,
79
- sensitiveValues: collectDispatchSensitiveValues(lowered.runner, {}, runOptions.envSource),
80
- interactive,
102
+ sensitiveValues: collectDispatchSensitiveValues(lowered.runner, {}, envSource),
81
103
  };
82
104
  }
83
105
  /**
@@ -128,10 +150,6 @@ export async function akmPropose(options) {
128
150
  const config = options.agentConfig ?? (await import("../../core/config/config.js")).loadConfig();
129
151
  const target = resolveProposalQueueTarget(stash, config);
130
152
  // 2. Build terminal user content.
131
- // Synthesize a temp draft path so opencode can write the asset content
132
- // directly using its file tools rather than returning JSON via stdout.
133
- const draftFilePath = import("node:os").then((os) => import("node:path").then((path) => path.join(os.tmpdir(), `akm-propose-${options.type}-${options.name.replace(/[^a-z0-9_-]/gi, "_")}-${Date.now()}.md`)));
134
- const resolvedDraftPath = await draftFilePath;
135
153
  // Standards "rulebook" for this target — wiki schema (wiki page) or stash
136
154
  // convention/meta facts (non-wiki asset); empty when neither fires.
137
155
  const standardsContext = resolveStandardsContext(`${options.type}:${options.name}`, stash);
@@ -140,14 +158,14 @@ export async function akmPropose(options) {
140
158
  name: options.name,
141
159
  task: options.task,
142
160
  ...(standardsContext.trim() ? { standardsContext } : {}),
143
- draftFilePath: resolvedDraftPath,
144
161
  });
145
- // 3. Preserve the fully-authored prompt as the terminal user content;
146
- // no synthetic persona, conversation turn, schema, or tool selection is
147
- // introduced while it crosses the shared resolved/lowered boundary.
162
+ // 3. Preserve the fully-authored prompt as the terminal user content; the
163
+ // proposal's JSON Schema crosses the shared resolved/lowered boundary as the
164
+ // request's output schema, with no synthetic persona, conversation turn, or
165
+ // tool selection.
148
166
  const dispatch = await dispatchProposalPrompt(prompt, config, options, () => emitProposeInvoked(target.source, options));
149
- const { result, engineName, notices, sensitiveValues } = dispatch;
150
- if (!result.ok) {
167
+ const { result, reply, engineName, notices, sensitiveValues } = dispatch;
168
+ if (!reply) {
151
169
  // B3: ENOENT / not-found gives an actionable hint.
152
170
  if (isEnoentFailure(result)) {
153
171
  return {
@@ -157,54 +175,23 @@ export async function akmPropose(options) {
157
175
  }
158
176
  return failureEnvelope(result, options.type, options.name, engineName, notices);
159
177
  }
160
- // 5. Resolve the proposal content.
161
- // Path A: opencode wrote the draft file — read it directly (no stdout parse).
162
- // Path B: fallback to stdout JSON parse for non-file-writing agents.
163
- let payload;
164
- if (fs.existsSync(resolvedDraftPath)) {
165
- const draftContent = fs.readFileSync(resolvedDraftPath, "utf8");
166
- fs.unlinkSync(resolvedDraftPath);
167
- payload = {
168
- ref: proposeItemRef(target.source, options.type, options.name),
169
- content: draftContent,
178
+ // 5. The proposal the engine returned on stdout, validated.
179
+ if (!reply.ok) {
180
+ return {
181
+ schemaVersion: 2,
182
+ ok: false,
183
+ reason: "parse_error",
184
+ error: `Engine "${engineName}" reply was not valid proposal JSON after ${reply.attempts} attempts: ${reply.errors.join("; ")}`,
185
+ type: options.type,
186
+ name: options.name,
187
+ engine: engineName,
188
+ exitCode: result.exitCode,
189
+ stdout: result.stdout,
190
+ ...(result.stderr ? { stderr: result.stderr } : {}),
191
+ ...noticeFields(notices),
170
192
  };
171
193
  }
172
- else {
173
- // B1: When interactive mode was used and stdout is empty, the agent did not
174
- // write the draft file and stdout was not captured — surface an actionable error.
175
- if (dispatch.interactive && (result.stdout ?? "") === "") {
176
- return {
177
- schemaVersion: 2,
178
- ok: false,
179
- reason: "parse_error",
180
- error: "Agent did not write draft file and stdout was not captured (interactive mode). Check that the agent CLI understood the file-write instruction, or configure a headless profile with stdio: 'captured'.",
181
- type: options.type,
182
- name: options.name,
183
- engine: engineName,
184
- exitCode: result.exitCode,
185
- ...(result.stderr ? { stderr: result.stderr } : {}),
186
- ...noticeFields(notices),
187
- };
188
- }
189
- try {
190
- payload = parseAgentProposalPayload(result.stdout ?? "");
191
- }
192
- catch (err) {
193
- return {
194
- schemaVersion: 2,
195
- ok: false,
196
- reason: "parse_error",
197
- error: err instanceof Error ? err.message : String(err),
198
- type: options.type,
199
- name: options.name,
200
- engine: engineName,
201
- exitCode: result.exitCode,
202
- stdout: result.stdout,
203
- ...(result.stderr ? { stderr: result.stderr } : {}),
204
- ...noticeFields(notices),
205
- };
206
- }
207
- }
194
+ const payload = reply.value;
208
195
  const unsafeContent = generatedContentRejection(payload.content, redactSensitiveText(payload.content, sensitiveValues));
209
196
  if (unsafeContent) {
210
197
  return {
@@ -270,6 +257,7 @@ export async function akmPropose(options) {
270
257
  content: payload.content,
271
258
  ...(payload.frontmatter ? { frontmatter: payload.frontmatter } : {}),
272
259
  },
260
+ ...(payload.confidence !== undefined ? { confidence: payload.confidence } : {}),
273
261
  };
274
262
  const proposal = createProposal(stash, createInput, options.ctx);
275
263
  return {
@@ -278,7 +266,7 @@ export async function akmPropose(options) {
278
266
  proposal,
279
267
  ref: proposal.ref,
280
268
  engine: engineName,
281
- durationMs: result.durationMs,
269
+ durationMs: dispatch.durationMs,
282
270
  ...noticeFields(notices),
283
271
  };
284
272
  }
@@ -9,7 +9,8 @@
9
9
  * growing past min(max(250% of the source, 2500 bytes), 25000 bytes) suggests
10
10
  * speculation. The absolute bounds keep small assets (a p25 source is ~780
11
11
  * bytes) from tripping on one good paragraph, and 25000 (below p99) still
12
- * catches runaway expansion. Sources under 200 bytes are too noisy to judge.
12
+ * catches runaway expansion. Sources under 200 bytes are too noisy to judge. A
13
+ * body that does not grow is never expansion, however long its source.
13
14
  */
14
15
  import { parseFrontmatter } from "../../../core/asset/frontmatter.js";
15
16
  import { parseRefInput } from "../../../core/asset/resolve-ref.js";
@@ -159,7 +160,8 @@ export function checkReflectSize(sourceBody, proposedBody) {
159
160
  return { ok: false, code: "EXCESSIVE_SHRINKAGE", ratio };
160
161
  }
161
162
  const expandCeiling = Math.min(Math.max(REFLECT_EXPAND_RATIO_MAX * sourceLen, REFLECT_ABSOLUTE_CEILING_BYTES), REFLECT_ABSOLUTE_MAX_BYTES);
162
- if (proposedLen > expandCeiling)
163
+ // A body that does not grow is never expansion, even where the cap sits below the source's own length.
164
+ if (proposedLen > sourceLen && proposedLen > expandCeiling)
163
165
  return { ok: false, code: "EXCESSIVE_EXPANSION", ratio };
164
166
  return { ok: true };
165
167
  }