akm-cli 0.9.25-alpha.1 → 0.9.25-alpha.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/CHANGELOG.md +243 -280
  2. package/dist/assets/prompts/reflect-feedback-framing.md +1 -1
  3. package/dist/assets/prompts/reflect-llm-framed-contract.md +2 -9
  4. package/dist/assets/prompts/reflect-llm-schema-contract.md +1 -3
  5. package/dist/assets/prompts/reflect-output-repair.md +1 -1
  6. package/dist/cli.js +1 -1
  7. package/dist/commands/improve/consolidate/pair-pass.js +1 -0
  8. package/dist/commands/improve/consolidate.js +7 -2
  9. package/dist/commands/improve/execution.js +2 -3
  10. package/dist/commands/improve/extract-cli.js +3 -2
  11. package/dist/commands/improve/extract.js +2 -1
  12. package/dist/commands/improve/improve-cli.js +33 -1
  13. package/dist/commands/improve/loop-stages.js +3 -0
  14. package/dist/commands/improve/reflect-noise.js +125 -0
  15. package/dist/commands/improve/reflect.js +150 -333
  16. package/dist/commands/improve/retrieval-gate.js +7 -2
  17. package/dist/commands/improve/session-asset.js +6 -0
  18. package/dist/commands/improve/stage.js +31 -39
  19. package/dist/commands/proposal/drain.js +4 -7
  20. package/dist/commands/proposal/propose.js +2 -11
  21. package/dist/commands/proposal/validators/proposal-quality-validators.js +11 -5
  22. package/dist/commands/proposal/validators/proposal-validators.js +4 -5
  23. package/dist/commands/read/search-cli.js +0 -38
  24. package/dist/core/asset/asset-serialize.js +1 -1
  25. package/dist/core/config/schema/engines.js +15 -33
  26. package/dist/core/config/schema/improve-processes.js +16 -0
  27. package/dist/core/content-safety.js +0 -24
  28. package/dist/core/redaction.js +4 -0
  29. package/dist/core/spawn-env.js +25 -0
  30. package/dist/core/structured.js +1 -1
  31. package/dist/execution/source.js +8 -12
  32. package/dist/integrations/agent/config.js +1 -3
  33. package/dist/integrations/agent/engine-resolution.js +0 -3
  34. package/dist/integrations/agent/execution.js +14 -13
  35. package/dist/integrations/agent/index.js +1 -1
  36. package/dist/integrations/agent/model-map.js +15 -16
  37. package/dist/integrations/agent/profiles.js +2 -2
  38. package/dist/integrations/agent/prompts.js +51 -127
  39. package/dist/integrations/agent/request-lowering.js +9 -7
  40. package/dist/integrations/agent/runner-dispatch.js +25 -31
  41. package/dist/integrations/harnesses/aider/agent-builder.js +1 -2
  42. package/dist/integrations/harnesses/amazonq/agent-builder.js +1 -2
  43. package/dist/integrations/harnesses/claude/agent-builder.js +4 -16
  44. package/dist/integrations/harnesses/codex/agent-builder.js +1 -2
  45. package/dist/integrations/harnesses/codex/index.js +6 -11
  46. package/dist/integrations/harnesses/codex/session-log.js +211 -0
  47. package/dist/integrations/harnesses/ids.js +10 -16
  48. package/dist/integrations/harnesses/opencode/agent-builder.js +14 -24
  49. package/dist/integrations/harnesses/opencode/model-config.js +15 -62
  50. package/dist/integrations/harnesses/opencode/model-work-agent.js +71 -36
  51. package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -6
  52. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +32 -90
  53. package/dist/integrations/harnesses/openhands/agent-builder.js +1 -2
  54. package/dist/integrations/harnesses/pi/agent-builder.js +1 -2
  55. package/dist/integrations/harnesses/types.js +3 -3
  56. package/dist/llm/feature-gate.js +2 -5
  57. package/dist/llm/index-passes.js +2 -2
  58. package/dist/llm/structured-call.js +5 -5
  59. package/dist/output/shapes/passthrough.js +1 -0
  60. package/dist/scripts/akm-migrate-node.js +381 -239
  61. package/dist/scripts/akm-migrate.js +381 -239
  62. package/dist/workflows/exec/unit-dispatch.js +4 -13
  63. package/docs/reference/cli.md +29 -16
  64. package/docs/reference/configuration.md +79 -85
  65. package/docs/reference/data-and-telemetry.md +2 -3
  66. package/docs/reference/workflow-schema.md +6 -9
  67. package/package.json +1 -1
  68. package/schemas/akm-config.json +108 -36
@@ -3,7 +3,5 @@
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  /** Default agent CLI timeout; null means agents run until they finish. */
5
5
  export const DEFAULT_AGENT_TIMEOUT_MS = null;
6
- /** Default hard timeout for direct LLM calls when no engine/use override exists. */
6
+ /** Default hard timeout for a direct LLM call, and the bound on model work on every runner kind, when no engine/use override exists. */
7
7
  export const DEFAULT_LLM_TIMEOUT_MS = 600_000;
8
- /** Default bound on model work (structured calls, judgments) on every runner kind. */
9
- export const DEFAULT_MODEL_WORK_TIMEOUT_MS = 600_000;
@@ -13,7 +13,6 @@ import { collectSensitiveValues } from "../../core/redaction.js";
13
13
  import { resolveSecretFromStore } from "../../sources/snapshot-fetchers/secret-seam.js";
14
14
  import { getHarness } from "../harnesses/index.js";
15
15
  import { DEFAULT_LLM_TIMEOUT_MS } from "./config.js";
16
- import { engineModelAndInference } from "./model-map.js";
17
16
  import { getBuiltinAgentProfile, OPENCODE_SDK_SERVER_BIN } from "./profiles.js";
18
17
  const LLM_CONNECTION_FIELDS = [
19
18
  "provider",
@@ -274,7 +273,6 @@ function lowerAgentEngine(name, engine, config) {
274
273
  const platform = harness.id;
275
274
  const sdk = platform === "opencode-sdk";
276
275
  const builtin = getBuiltinAgentProfile(platform);
277
- const { inference } = engineModelAndInference(engine);
278
276
  const profile = {
279
277
  name,
280
278
  platform,
@@ -287,7 +285,6 @@ function lowerAgentEngine(name, engine, config) {
287
285
  parseOutput: "text",
288
286
  ...(engine.workspace ? { workspace: path.resolve(engine.workspace) } : {}),
289
287
  ...(engine.model ? { model: engine.model } : {}),
290
- ...(inference ? { inference } : {}),
291
288
  };
292
289
  // An engine that sets no timeoutMs leaves it unset, so a caller's own default
293
290
  // (model work's 600 s) can apply; a dispatch with none runs unbounded.
@@ -6,7 +6,7 @@ import { ConfigError } from "../../core/errors.js";
6
6
  import { DURATION_UNITS, parseDuration } from "../../core/time.js";
7
7
  import { EXECUTION_MAX_TIMEOUT_MS } from "../../execution/limits.js";
8
8
  import { createInlineResolvedCommand, createResolvedExecutionRequest, decodeResolvedExecutionRequest, } from "../../execution/resolved-request.js";
9
- import { cloneToolSelection, isModelWorkTools, isPortableExecutionAgentSelector, } from "../../execution/source.js";
9
+ import { cloneToolSelection, isPortableExecutionAgentSelector, MODEL_WORK_POLICY_ID, } from "../../execution/source.js";
10
10
  import { getHarness } from "../harnesses/index.js";
11
11
  import { DEFAULT_LLM_TIMEOUT_MS } from "./config.js";
12
12
  import { FALLBACK_ENGINE_NAME, fallbackEngineConfig, NO_ENGINE_MESSAGE_SUFFIX, NO_ENGINE_REMEDY, } from "./engine-fallback.js";
@@ -78,11 +78,8 @@ function engineDefaults(name, engine, config) {
78
78
  modelMapKey: own.model === undefined ? fallbackName : "opencode-sdk",
79
79
  values: {
80
80
  ...(inherited.model !== undefined ? { model: inherited.model } : {}),
81
+ ...(inherited.inference !== undefined ? { inference: inherited.inference } : {}),
81
82
  ...values,
82
- // The engine's own inference is added over the fallback's, field by field.
83
- ...(inherited.inference !== undefined || own.inference !== undefined
84
- ? { inference: { ...inherited.inference, ...own.inference } }
85
- : {}),
86
83
  timeout: Object.hasOwn(engine, "timeoutMs")
87
84
  ? (engine.timeoutMs ?? null)
88
85
  : Object.hasOwn(fallback, "timeoutMs")
@@ -191,12 +188,16 @@ function requestedToolNames(tools) {
191
188
  * nothing is allowed. The model-work policy is akm's own and always allowed:
192
189
  * it confines an engine more tightly than leaving tools unset does.
193
190
  */
194
- function authorizeTools(tools, config) {
191
+ function authorizeTools(tools, config, modelWork) {
192
+ if (modelWork) {
193
+ return {
194
+ status: "allowed",
195
+ reason: "The model-work tool policy is akm's own.",
196
+ policy: { id: MODEL_WORK_POLICY_ID },
197
+ };
198
+ }
195
199
  if (!hasToolSelection(tools))
196
200
  return { status: "not-required" };
197
- if (isModelWorkTools(tools)) {
198
- return { status: "allowed", reason: "The model-work tool policy is akm's own.", policy: { id: "model-work" } };
199
- }
200
201
  if (!config) {
201
202
  return {
202
203
  status: "denied",
@@ -392,13 +393,14 @@ export function resolveExecution(input) {
392
393
  return layer;
393
394
  };
394
395
  const schemaLayer = select("outputSchema", "outputSchema");
395
- const toolsLayer = select("tools", "tools");
396
+ const modelWork = input.modelWork === true;
397
+ const toolsLayer = modelWork ? undefined : select("tools", "tools");
396
398
  const timeoutLayer = select("timeout", "runtime.timeoutMs");
397
399
  const workspaceLayer = select("workspace", "runtime.workspace");
398
400
  const environmentLayer = select("environment", "runtime.environment");
399
401
  const settingsLayer = select("runtime", "runtime.settings");
400
402
  const tools = toolsLayer ? cloneToolSelection(toolsLayer.values.tools ?? null, "tools") : undefined;
401
- const authorization = authorizeTools(tools, input.runner ? undefined : input.config);
403
+ const authorization = authorizeTools(tools, input.runner ? undefined : input.config, modelWork);
402
404
  provenance.authorization = {
403
405
  layer: typeof authorization.policy?.id === "string" ? authorization.policy.id : "not-required",
404
406
  kind: "authorization",
@@ -438,8 +440,7 @@ function buildLlm(request, runner, options) {
438
440
  if (typeof request.agent === "string" && request.persona === null) {
439
441
  throw new ConfigError(`The direct LLM transport cannot consume native agent selector ${JSON.stringify(request.agent)}.`, "INVALID_CONFIG_FILE");
440
442
  }
441
- // An LLM has no tools, which already meets the model-work policy.
442
- if (hasToolSelection(request.tools) && !isModelWorkTools(request.tools)) {
443
+ if (hasToolSelection(request.tools)) {
443
444
  throw new ConfigError("The direct LLM transport cannot enforce the resolved tool policy.", "INVALID_CONFIG_FILE");
444
445
  }
445
446
  const notices = [...request.notices];
@@ -4,4 +4,4 @@
4
4
  export { DEFAULT_AGENT_TIMEOUT_MS } from "./config.js";
5
5
  export { _setAgentDetectForTests, defaultWhich, detectAgentCliProfiles, pickDefaultAgentProfile } from "./detect.js";
6
6
  export { BUILTIN_AGENT_PROFILE_NAMES, getBuiltinAgentProfile, listBuiltinAgentProfiles, } from "./profiles.js";
7
- export { buildProposePrompt, buildReflectPrompt, buildSchemaRepairPrompt, parseAgentProposalPayload, } from "./prompts.js";
7
+ export { buildProposePrompt, buildReflectPrompt, parseAgentProposalPayload } from "./prompts.js";
@@ -220,33 +220,32 @@ function mergeProfiles(base, overlay) {
220
220
  * copied verbatim from the engine's own config value; it must already be
221
221
  * meaningful for the model-map column's platform (akm does not translate
222
222
  * between an engine's connection and an agent platform's own provider
223
- * registry). An agent-kind engine contributes only the inference fields its
224
- * platform translates (config validation rejects the rest); an `llm` engine
225
- * also contributes `supportsJsonSchema` and `extraParams`.
223
+ * registry). Only `kind: "llm"` engines contribute inference defaults — an
224
+ * agent-kind engine's schema carries no temperature/thinking fields.
226
225
  */
227
226
  export function engineModelAndInference(engine) {
228
227
  const out = {};
229
228
  if (Object.hasOwn(engine, "model") && engine.model !== undefined)
230
229
  out.model = engine.model;
231
- const inference = {};
232
- if (Object.hasOwn(engine, "temperature"))
233
- inference.temperature = engine.temperature;
234
- if (Object.hasOwn(engine, "maxTokens"))
235
- inference.maxTokens = engine.maxTokens;
236
230
  if (engine.kind === "llm") {
231
+ const inference = {};
232
+ if (Object.hasOwn(engine, "temperature"))
233
+ inference.temperature = engine.temperature;
234
+ if (Object.hasOwn(engine, "maxTokens"))
235
+ inference.maxTokens = engine.maxTokens;
237
236
  if (Object.hasOwn(engine, "supportsJsonSchema"))
238
237
  inference.supportsJsonSchema = engine.supportsJsonSchema;
239
238
  if (Object.hasOwn(engine, "extraParams"))
240
239
  inference.extraParams = engine.extraParams;
240
+ if (Object.hasOwn(engine, "contextLength"))
241
+ inference.contextLength = engine.contextLength;
242
+ if (Object.hasOwn(engine, "enableThinking"))
243
+ inference.enableThinking = engine.enableThinking;
244
+ if (Object.hasOwn(engine, "reasoningEffort"))
245
+ inference.reasoningEffort = engine.reasoningEffort;
246
+ if (Object.keys(inference).length > 0)
247
+ out.inference = inference;
241
248
  }
242
- if (Object.hasOwn(engine, "contextLength"))
243
- inference.contextLength = engine.contextLength;
244
- if (Object.hasOwn(engine, "enableThinking"))
245
- inference.enableThinking = engine.enableThinking;
246
- if (Object.hasOwn(engine, "reasoningEffort"))
247
- inference.reasoningEffort = engine.reasoningEffort;
248
- if (Object.keys(inference).length > 0)
249
- out.inference = inference;
250
249
  return Object.freeze(out);
251
250
  }
252
251
  /** Overlay user fields over installed fields, per (alias, column), without resolving `engine` indirection. */
@@ -8,7 +8,7 @@
8
8
  * coding-agent CLI. Named engines lower canonical harness metadata into this
9
9
  * intentionally small internal shape. The wrapper is in `./spawn.ts`.
10
10
  */
11
- import { COMMON_SPAWN_ENV_PASSTHROUGH } from "../../core/spawn-env.js";
11
+ import { COMMON_SPAWN_ENV_PASSTHROUGH, XDG_BASE_DIR_ENV_PASSTHROUGH } from "../../core/spawn-env.js";
12
12
  // AKM_EVENT_SOURCE carries usage-event provenance (improve/task) so that akm
13
13
  // invocations a spawned agent makes are recorded as machine traffic, not user
14
14
  // demand (DRIFT-6). Without it in the passthrough whitelist, buildChildEnv drops
@@ -30,7 +30,7 @@ const BUILTINS = {
30
30
  bin: "opencode",
31
31
  args: ["run"],
32
32
  stdio: "interactive",
33
- envPassthrough: [...COMMON_PASSTHROUGH, "OPENCODE_API_KEY", "OPENCODE_CONFIG"],
33
+ envPassthrough: [...COMMON_PASSTHROUGH, "OPENCODE_API_KEY", "OPENCODE_CONFIG", ...XDG_BASE_DIR_ENV_PASSTHROUGH],
34
34
  parseOutput: "text",
35
35
  },
36
36
  claude: {
@@ -70,10 +70,8 @@ function knownTypeList() {
70
70
  export const REFLECT_CONTENT_CAP = 12_000;
71
71
  /**
72
72
  * Marker appended to truncated asset content when it exceeds the active
73
- * content budget (#952). Exported so `sanitizeReflectPayload` can detect a
74
- * model that echoed this notice back into its rewrite instead of proposing
75
- * real content, and so the output contracts can reference the exact string
76
- * to forbid.
73
+ * content budget (#952). Exported so the proposal validators can refuse a body
74
+ * that still carries it.
77
75
  */
78
76
  export const REFLECT_TRUNCATION_MARKER = "... [truncated — focus on the visible portion]";
79
77
  /**
@@ -119,16 +117,12 @@ export function reflectResponseContract(mode, targetScoped) {
119
117
  if (mode === "json_schema") {
120
118
  return reflectLlmSchemaContract
121
119
  .replace("{{FIELD_RULE}}", targetScoped
122
- ? "The response has exactly the required fields `content`, `confidence`, and `frontmatterPatch`; do not echo `ref` or arbitrary `frontmatter`."
123
- : "The response has exactly the required fields `ref`, `content`, `confidence`, and `frontmatterPatch`; `ref` must identify the selected asset.")
124
- .replaceAll("{{TRUNCATION_MARKER}}", REFLECT_TRUNCATION_MARKER)
120
+ ? "The response has exactly the required fields `confidence` and `frontmatterPatch`; do not echo `ref` or arbitrary `frontmatter`."
121
+ : "The response has exactly the required fields `ref`, `confidence`, and `frontmatterPatch`; `ref` must identify the selected asset.")
125
122
  .trim();
126
123
  }
127
124
  const refLine = targetScoped ? "" : "AKM_REFLECT_REF: <selected asset ref>\n";
128
- return reflectLlmFramedContract
129
- .replace("{{REF_LINE}}", refLine)
130
- .replaceAll("{{TRUNCATION_MARKER}}", REFLECT_TRUNCATION_MARKER)
131
- .trim();
125
+ return reflectLlmFramedContract.replace("{{REF_LINE}}", refLine).trim();
132
126
  }
133
127
  export function buildReflectOutputRepairPrompt(mode, targetScoped) {
134
128
  return reflectOutputRepair.replace("{{OUTPUT_CONTRACT}}", reflectResponseContract(mode, targetScoped)).trim();
@@ -158,23 +152,47 @@ function sourceHasNonEmptyDescription(assetContent) {
158
152
  return value.length > 0;
159
153
  }
160
154
  /**
161
- * Build the prompt for `akm reflect [ref]`. Asks the agent to review an
162
- * existing asset (plus any negative feedback / lint findings) and propose
163
- * an improved version. Returns a {@link ReflectPromptResult} containing the
164
- * prompt string and an optional character ceiling for max-tokens enforcement.
155
+ * The frontmatter problems akm can see for itself, named in the prompt so the
156
+ * model fixes them instead of having to spot them: a description split by a
157
+ * stray period or carrying an escaped quote, no `when_to_use`, no title. A
158
+ * missing description has its own instruction (#636).
159
+ */
160
+ function frontmatterProblems(assetContent) {
161
+ const fm = assetContent?.match(/^---\r?\n([\s\S]*?)\r?\n---\r?\n?/);
162
+ if (!assetContent || !fm)
163
+ return [];
164
+ const block = fm[1] ?? "";
165
+ const problems = [];
166
+ const raw = block
167
+ .match(/^description\s*:(.*(?:\r?\n[ \t]+.*)*)/m)?.[1]
168
+ ?.replace(/\s+/g, " ")
169
+ .trim() ?? "";
170
+ const split = [...raw.matchAll(/[\w`)\]]\. [a-z]/g)].find((m) => !/\b(?:e\.g|i\.e|vs|etc|cf)$/i.test(raw.slice(0, (m.index ?? 0) + 1)));
171
+ if (split) {
172
+ problems.push(`the \`description\` is broken: a stray period splits a sentence ("${raw}"); repair only the break, keeping its wording and every name, number, path and status word in it`);
173
+ }
174
+ else if (raw.includes('\\"')) {
175
+ problems.push("the `description` is broken by an escaped quote; repair only the quoting, keeping the rest of its wording");
176
+ }
177
+ if (!/^when_to_use\s*:\s*(?:\S|\r?\n[ \t]+\S)/m.test(block)) {
178
+ problems.push("there is no `when_to_use`: write one, a single sentence the body supports, saying when to reach for this asset");
179
+ }
180
+ if (!/^title\s*:\s*\S/m.test(block) && !/^#[ \t]+\S/m.test(assetContent.slice(fm[0].length))) {
181
+ problems.push("the body has no level-1 title: give one in `title`");
182
+ }
183
+ return problems;
184
+ }
185
+ /**
186
+ * Build the prompt for `akm reflect [ref]`. Asks the agent to check an
187
+ * existing asset's `description`, `when_to_use` and title against its body
188
+ * (plus any negative feedback / lint findings) and return the fields that
189
+ * need a fix. Returns a {@link ReflectPromptResult} containing the prompt
190
+ * string.
165
191
  */
166
192
  export function buildReflectPrompt(input) {
167
193
  const sections = [];
168
194
  if (input.ref && input.type && input.name) {
169
- // Change 2 — type-conditioned goal framing
170
- const isLesson = input.type === "lesson";
171
- const isSkill = input.type === "skill";
172
- const goalSentence = isLesson
173
- ? `Your task is to distill what usage signals reveal about this ${input.type} asset — when to reach for it, what goes wrong without it, and what real use has revealed that the asset itself does not say. Do not reproduce the source content; your proposal must add information the source does not contain.`
174
- : isSkill
175
- ? "Your task is to review this skill asset, identify what the feedback and related distilled lessons show is broken, missing, unclear, or durable enough to promote into long-term documentation, and produce a single improved proposal. If the strongest evidence points to companion reference material rather than the main SKILL.md, you may instead propose a skill-adjacent knowledge doc such as `knowledge/skills/<skill>/references/<topic>`."
176
- : `Your task is to review this ${input.type} asset, identify what the feedback signals as broken, missing, or unclear, and produce an improved version. Do not reproduce the source content unchanged; your proposal must correct or add something the source lacks.`;
177
- sections.push(goalSentence);
195
+ sections.push(`Your task is to check this ${input.type} asset's \`description\`, \`when_to_use\` and title against its body and the feedback below, and fix any that is missing, broken, or claims something the body does not cover. AKM keeps the body exactly as it is: you change only these fields, and when none needs a change you return null for each.`);
178
196
  sections.push(`Target ref: ${input.ref}`);
179
197
  sections.push(`Asset-type guidance: ${hintForType(input.type)}`);
180
198
  }
@@ -200,12 +218,9 @@ export function buildReflectPrompt(input) {
200
218
  sections.push("Recent feedback / signals:");
201
219
  sections.push("- (no feedback events recorded)");
202
220
  }
203
- else if (input.type === "skill" && input.relatedLessons && input.relatedLessons.length > 0) {
204
- sections.push("No direct feedback events were recorded. Limit substantive changes to what is justified by the related distilled lessons below; do not speculate beyond that evidence.");
205
- }
206
221
  else {
207
- // ref is set but no feedback — explicitly constrain scope to schema compliance
208
- sections.push("No usage feedback recorded. Limit your proposal to schema and structural improvements only: missing required frontmatter fields, unclear `when_to_use`, ambiguous description, or broken formatting. Do not speculate about runtime weaknesses you have not observed.");
222
+ // ref is set but no feedback — explicitly constrain scope to broken or missing fields
223
+ sections.push("No usage feedback recorded. Fix only a missing or broken `description`, `when_to_use` or title; otherwise return null for each.");
209
224
  }
210
225
  if (input.standardsContext?.trim()) {
211
226
  sections.push("Standards to follow (the rulebook for this target):");
@@ -253,7 +268,7 @@ export function buildReflectPrompt(input) {
253
268
  sections.push("```");
254
269
  }
255
270
  else if (input.ref) {
256
- sections.push("(No existing content — propose a fresh asset that fits the ref.)");
271
+ sections.push("(No existing content.)");
257
272
  }
258
273
  else {
259
274
  sections.push("(No existing asset content was supplied.)");
@@ -263,23 +278,10 @@ export function buildReflectPrompt(input) {
263
278
  for (const line of input.schemaHints)
264
279
  sections.push(`- ${line}`);
265
280
  }
266
- if (input.relatedLessons && input.relatedLessons.length > 0) {
267
- sections.push("Related distilled lessons to evaluate for consolidation:");
268
- for (const lesson of input.relatedLessons) {
269
- sections.push(`Lesson ref: ${lesson.ref}`);
270
- sections.push("```");
271
- sections.push(lesson.content.trimEnd());
272
- sections.push("```");
273
- }
274
- sections.push("Evaluate whether these lessons contain strong evidence of factual, repeatable guidance that should be promoted into long-term skill documentation.");
275
- sections.push("Promote only guidance that is durable, generally applicable, and supported by repeated evidence. Do not copy anecdotal details, one-off incidents, or duplicate wording verbatim.");
276
- sections.push("If the guidance belongs in the main skill instructions, update the skill proposal. If it belongs in a companion reference document, return a `knowledge/skills/<skill>/references/<topic>` proposal instead.");
277
- }
278
281
  if (input.rejectedProposals && input.rejectedProposals.length > 0) {
279
282
  const lines = ["## Previously Rejected Proposals"];
280
283
  lines.push("The following proposals for this ref were already reviewed and rejected. " +
281
- "Do NOT reproduce the same content or the same structural shape. " +
282
- "Your new proposal must meaningfully differ from each of these in its approach, framing, or evidence used.");
284
+ "Do not propose the same change again; if no other change is justified, return null for each field.");
283
285
  for (const rp of input.rejectedProposals) {
284
286
  lines.push(`\nRef: ${rp.ref}`);
285
287
  lines.push(`Rejection reason: ${rp.reason}`);
@@ -303,57 +305,16 @@ export function buildReflectPrompt(input) {
303
305
  "The following is your previous draft proposal. " +
304
306
  "Identify specific weaknesses: missing evidence, vague wording, incomplete frontmatter, " +
305
307
  "or claims that duplicate existing content without adding new signal. " +
306
- "Then produce an improved version that addresses those weaknesses. " +
307
- "The revised proposal must be meaningfully better than the draft below — " +
308
- "do not return the same content unchanged.\n\n" +
308
+ "Then produce an improved version that addresses those weaknesses.\n\n" +
309
309
  "Previous draft:\n```\n" +
310
310
  input.priorDraft.trimEnd() +
311
311
  "\n```");
312
312
  }
313
- sections.push("Produce a single proposal that addresses the feedback and respects the asset-type contract. If the proposal's frontmatter is missing `when_to_use`, you MUST generate one — a one-line trigger sentence describing exactly when a user should reach for this asset.");
314
- // Content-preservation safety rails (#reflect-pipeline-fixes).
315
- // These rules counter the observed failure modes where reflect rewrites
316
- // asset content into shorter prose, drops concrete structure, or strips
317
- // load-bearing frontmatter. Loud and explicit so small models follow.
318
- //
319
- // Guard-audit finding 15: this used to also hand back a maxOutputChars
320
- // value so an LLM-path caller could convert it into a hard `max_tokens`
321
- // cap on the API request. llm/client.ts's own doc comment (and
322
- // commands/improve/reflect.ts's recorded history of responses actually
323
- // getting cut off) is explicit that a character-derived max_tokens causes
324
- // silent truncation — a real model's output is measured in tokens, not
325
- // characters, and the ratio between the two varies enough that any fixed
326
- // conversion either truncates legitimate output or provides no real cap at
327
- // all. The size policy below is already enforced twice more (the prompt
328
- // rules the model reads, and the post-processor's own size check), so nothing
329
- // is lost by not adding a THIRD, byte-derived enforcement point that can
330
- // only ever cut a response off early, never usefully re-check it.
331
313
  if (input.ref && input.assetContent?.trim()) {
332
- // Strip frontmatter to get source body length — mirrors checkReflectSize which
333
- // compares body-only lengths. Inline regex avoids importing parseFrontmatter.
334
- const rawContent = input.assetContent.trimEnd();
335
- const fmBodyMatch = rawContent.match(/^---\r?\n[\s\S]*?\r?\n---\r?\n?([\s\S]*)$/);
336
- const sourceBodyLen = (fmBodyMatch ? fmBodyMatch[1] : rawContent).trim().length;
337
- // Compute concrete char bounds matching checkReflectSize constants:
338
- // REFLECT_SIZE_GUARD_MIN_BYTES=200, REFLECT_SHRINK_RATIO_MIN=0.5,
339
- // REFLECT_ABSOLUTE_FLOOR_BYTES=150, REFLECT_EXPAND_RATIO_MAX=2.5,
340
- // REFLECT_ABSOLUTE_CEILING_BYTES=2500, REFLECT_ABSOLUTE_MAX_BYTES=25000.
341
- // Embed concrete counts only when the gate will actually fire (source >= 200 chars).
342
- const showCharBounds = sourceBodyLen >= 200;
343
- const minChars = Math.max(Math.round(0.5 * sourceBodyLen), 150);
344
- const maxChars = Math.min(Math.max(Math.round(2.5 * sourceBodyLen), 2500), 25000);
345
- sections.push([
346
- "## Content preservation rules (MUST follow)",
347
- "1. PRESERVE ALL concrete content: code blocks, fenced snippets, CLI commands, numbered/bulleted checklists, tables, YAML/JSON examples, file paths, configuration keys, environment variable names, and CSS/HTML selectors. These are load-bearing — do NOT replace them with prose summaries.",
348
- "2. PRESERVE the source asset's frontmatter. The post-processor reassembles the final asset from the original frontmatter plus your body. Do NOT emit `---` frontmatter delimiters at the top of `content` — start `content` with the markdown body (e.g. `# Heading` or the first paragraph). If you include frontmatter anyway, identity fields (`name`, `ref`, `id`, `slug`, `type`) will be reset to the original values.",
349
- showCharBounds
350
- ? `3. DO NOT shrink the asset. Your body must be at least ${minChars} characters (source body is ${sourceBodyLen} chars; floor is 50%). If you genuinely need to remove a major section, explain why in a comment line at the top of the body (e.g. \`<!-- removed obsolete section X because ... -->\`).`
351
- : "3. DO NOT shrink the asset dramatically. The improved body must be at least 50% of the source body length. If you genuinely need to remove a major section, explain why in a comment line at the top of the body (e.g. `<!-- removed obsolete section X because ... -->`).",
352
- showCharBounds
353
- ? `4. DO NOT pad the asset with speculative material. Your body must be at most ${maxChars} characters (source body is ${sourceBodyLen} chars; ceiling is 250%). Do not add invented sections, hypothetical examples, or padding prose.`
354
- : "4. DO NOT pad the asset with speculative material. The improved body must be at most 250% of the source body length unless the feedback explicitly requests added sections.",
355
- "5. Improve clarity of surrounding prose, fix structural issues, add missing required frontmatter fields. Do NOT rewrite a runbook into an essay.",
356
- ].join("\n"));
314
+ const problems = frontmatterProblems(input.assetContent);
315
+ sections.push(problems.length > 0
316
+ ? `akm found these problems in the asset; fix each one:\n${problems.map((p) => `- ${p}`).join("\n")}`
317
+ : "akm found no missing or broken field.");
357
318
  }
358
319
  sections.push(reflectResponseContract(input.outputMode ?? "json_schema", input.ref !== undefined));
359
320
  return { prompt: sections.join("\n\n") };
@@ -388,43 +349,6 @@ export function buildProposePrompt(input) {
388
349
  sections.push(RESPONSE_CONTRACT_JSON);
389
350
  return sections.join("\n\n");
390
351
  }
391
- /**
392
- * Build the prompt for the schema repair pass in `akm improve`. Asks the
393
- * agent to add the minimal required frontmatter to an asset that failed
394
- * validation — without rewriting the body.
395
- */
396
- export function buildSchemaRepairPrompt(input) {
397
- const sections = [];
398
- sections.push(`This ${input.type} asset failed schema validation with the error: "${input.reason}". ` +
399
- `Your task is to fix the schema issue by adding or correcting the missing/invalid field(s) ` +
400
- `while preserving all existing content.`);
401
- sections.push(`Target ref: ${input.ref}`);
402
- sections.push(`Schema requirements for ${input.type} assets: ${hintForType(input.type)}`);
403
- if (input.standardsContext?.trim()) {
404
- sections.push("Standards to follow (the rulebook for this target):");
405
- sections.push(input.standardsContext.trim());
406
- }
407
- {
408
- const authoringRules = authoringRulesForType(input.type);
409
- if (authoringRules) {
410
- sections.push(authoringRules);
411
- }
412
- }
413
- const CONTENT_CAP = 3000;
414
- const body = input.assetContent.trimEnd();
415
- const truncated = body.length > CONTENT_CAP;
416
- sections.push("Current asset content (first 3000 chars — sufficient to generate missing frontmatter):");
417
- sections.push("```");
418
- sections.push(truncated ? `${body.slice(0, CONTENT_CAP)}\n... [truncated]` : body);
419
- sections.push("```");
420
- sections.push("Produce the minimal fix: add ONLY the missing required frontmatter field(s). " +
421
- "Do not rewrite the body unless it is empty. " +
422
- "If `description` is missing, generate a concise one-sentence description from the content. " +
423
- "If `when_to_use` is missing, generate a one-line trigger sentence. " +
424
- "Preserve all existing frontmatter keys and the full body verbatim.");
425
- sections.push(RESPONSE_CONTRACT_JSON);
426
- return sections.join("\n\n");
427
- }
428
352
  /**
429
353
  * Parse agent stdout into a proposal payload. The agent contract requires a
430
354
  * single JSON object; anything else is reported as a parse error so callers
@@ -3,7 +3,8 @@
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  import { ConfigError } from "../../core/errors.js";
5
5
  import { withSchemaInstruction } from "../../core/structured.js";
6
- import { isModelWorkTools } from "../../execution/source.js";
6
+ import { MODEL_WORK_POLICY_ID } from "../../execution/source.js";
7
+ import { HARNESS_MODEL_WORK_IDS } from "../harnesses/ids.js";
7
8
  import { composeConversationFallbackPrompt } from "./conversation-fallback.js";
8
9
  import { composePersonaFallbackPrompt } from "./persona-fallback.js";
9
10
  /** A tool selection that actually names tools (not omitted, null, or empty). */
@@ -47,8 +48,9 @@ export function extensionFields(request) {
47
48
  * composing into the prompt what the harness has no channel for.
48
49
  */
49
50
  export function createAgentRequestLowerer(options) {
50
- return (profile, request) => {
51
- const supportedInference = new Set(typeof options.inference === "function" ? options.inference(profile, request) : (options.inference ?? []));
51
+ const supportedInference = new Set(options.inference ?? []);
52
+ return (_profile, request) => {
53
+ const modelWork = request.authorization.policy?.id === MODEL_WORK_POLICY_ID;
52
54
  const notices = [];
53
55
  const skip = (field) => {
54
56
  notices.push(untranslated(options.adapter, field));
@@ -92,19 +94,19 @@ export function createAgentRequestLowerer(options) {
92
94
  if (Object.hasOwn(request, "inference")) {
93
95
  dispatch.inference = request.inference ?? null;
94
96
  for (const key of Object.keys(request.inference ?? {}).sort()) {
95
- if (!supportedInference.has(key))
97
+ if (!(modelWork && supportedInference.has(key)))
96
98
  skip(`inference.${key}`);
97
99
  }
98
100
  }
99
- if (isModelWorkTools(request.tools)) {
100
- if (!options.modelWorkTools) {
101
+ if (modelWork) {
102
+ if (!HARNESS_MODEL_WORK_IDS.has(options.adapter)) {
101
103
  throw new ConfigError(`The ${options.adapter} transport cannot enforce the model-work tool policy.`, "INVALID_CONFIG_FILE");
102
104
  }
103
105
  // The builder selects its own confined agent or flags; another agent would replace them.
104
106
  if (dispatch.agent) {
105
107
  throw new ConfigError(`The ${options.adapter} transport cannot run native agent ${JSON.stringify(dispatch.agent)} under the model-work tool policy.`, "INVALID_CONFIG_FILE");
106
108
  }
107
- dispatch.tools = request.tools;
109
+ dispatch.modelWork = true;
108
110
  }
109
111
  else if (request.tools !== undefined) {
110
112
  // An explicit empty selection still reaches the builder (e.g. an empty allowlist).
@@ -12,13 +12,14 @@ import fs from "node:fs";
12
12
  import os from "node:os";
13
13
  import path from "node:path";
14
14
  import { assertNever } from "../../core/assert.js";
15
- import { ConfigError, UsageError } from "../../core/errors.js";
15
+ import { UsageError } from "../../core/errors.js";
16
16
  import { collectSensitiveValues, isEnvPassthroughValueSafeToExpose, redactSensitiveText, redactSensitiveValue, } from "../../core/redaction.js";
17
- import { isModelWorkTools } from "../../execution/source.js";
17
+ import { MODEL_WORK_POLICY_ID } from "../../execution/source.js";
18
18
  import { chatCompletion, LlmCallError } from "../../llm/client.js";
19
19
  import { emitLlmUsage } from "../../llm/usage-telemetry.js";
20
20
  import { getHarness } from "../harnesses/index.js";
21
21
  import { closeServer as disposeOpencodeSdkServers, runOpencodeSdk } from "../harnesses/opencode-sdk/sdk-runner.js";
22
+ import { modelFromArgs } from "./builder-shared.js";
22
23
  import { lookupApiKeyFileValue, lookupApiKeySecretRefValue, lookupCredentialFromEnv, resolveEngine, } from "./engine-resolution.js";
23
24
  import { materializeLlmRunnerConnection, materializeSdkFallbackConnection } from "./runner.js";
24
25
  import { runAgent } from "./spawn.js";
@@ -86,18 +87,28 @@ const USAGE_ERROR_CODES = {
86
87
  parse_error: "parse_error",
87
88
  llm_rate_limit: "rate_limited",
88
89
  };
90
+ /**
91
+ * The model an agent or SDK dispatch ran: the request's, else the one an agent
92
+ * CLI's own `args` select, which its command carries when the request names
93
+ * none. An SDK server never sees `args`, so an SDK engine with no model named
94
+ * has none to report: opencode picks it.
95
+ */
96
+ function dispatchedModel({ request, runner }) {
97
+ return request.model?.resolved ?? (runner.kind === "agent" ? modelFromArgs(runner.profile.args) : undefined);
98
+ }
89
99
  /**
90
100
  * One usage record for an agent or SDK dispatch, through the same sink and
91
101
  * ambient `withLlmStage` attribution as the LLM transport's per-HTTP-attempt
92
- * records: the request's model, and tokens when the runner reported them.
102
+ * records: the model it ran, and tokens when the runner reported them.
93
103
  */
94
104
  function recordDispatchUsage(execution, result) {
95
105
  const { inputTokens, outputTokens, reasoningTokens } = result.usage ?? {};
96
106
  const reported = [inputTokens, outputTokens, reasoningTokens].filter((count) => count !== undefined);
107
+ const model = dispatchedModel(execution);
97
108
  emitLlmUsage({
98
109
  outcome: result.ok ? "success" : "error",
99
110
  modelSource: "configured",
100
- ...(execution.request.model?.resolved ? { model: execution.request.model.resolved } : {}),
111
+ ...(model ? { model } : {}),
101
112
  durationMs: result.durationMs,
102
113
  ...(inputTokens !== undefined ? { promptTokens: inputTokens } : {}),
103
114
  ...(outputTokens !== undefined ? { completionTokens: outputTokens } : {}),
@@ -135,30 +146,13 @@ async function dispatchRunner(runner, prompt, opts, seams, llm) {
135
146
  }
136
147
  return redactResult(result, collectSensitiveValues(secrets));
137
148
  }
138
- /** The git repository a directory is inside, if any: a `.git` file, or a `.git` directory with a HEAD. */
139
- function enclosingGitRepository(dir) {
140
- for (let current = fs.realpathSync(dir);; current = path.dirname(current)) {
141
- const marker = path.join(current, ".git");
142
- if (fs.existsSync(path.join(marker, "HEAD")) || (fs.existsSync(marker) && fs.statSync(marker).isFile())) {
143
- return current;
144
- }
145
- if (path.dirname(current) === current)
146
- return undefined;
147
- }
148
- }
149
149
  /**
150
150
  * The scratch working directory for one model-work dispatch on an agent or SDK
151
151
  * engine, so the edit the model-work tool policy grants never reaches the
152
- * stash. opencode counts a whole git repository as inside its working
153
- * directory, so a scratch directory inside one is refused.
152
+ * stash.
154
153
  */
155
154
  function createModelWorkDirectory() {
156
- const dir = fs.mkdtempSync(path.join(os.tmpdir(), "akm-model-work-"));
157
- const repository = enclosingGitRepository(dir);
158
- if (repository === undefined)
159
- return dir;
160
- fs.rmSync(dir, { recursive: true, force: true });
161
- throw new ConfigError(`Model work runs an agent in a scratch directory outside any git repository, but ${dir} is inside the repository at ${repository}.`, "INVALID_CONFIG_FILE", "Point TMPDIR at a directory outside any git repository.");
155
+ return fs.mkdtempSync(path.join(os.tmpdir(), "akm-model-work-"));
162
156
  }
163
157
  /**
164
158
  * Run a built execution. Credentials are read here, once per call, and never
@@ -168,7 +162,7 @@ function createModelWorkDirectory() {
168
162
  * limit, for one) has failed with `parse_error`.
169
163
  */
170
164
  export async function runExecution(execution, options = {}) {
171
- const scratch = execution.runner.kind !== "llm" && isModelWorkTools(execution.request.tools)
165
+ const scratch = execution.runner.kind !== "llm" && execution.request.authorization.policy?.id === MODEL_WORK_POLICY_ID
172
166
  ? createModelWorkDirectory()
173
167
  : undefined;
174
168
  try {
@@ -179,14 +173,14 @@ export async function runExecution(execution, options = {}) {
179
173
  fs.rmSync(scratch, { recursive: true, force: true });
180
174
  }
181
175
  }
182
- /**
183
- * A model-work reply's answer: the harness's result extractor strips its
184
- * framing (claude's `--output-format json` result envelope, for one), as a
185
- * workflow unit's does, and no answer is a `parse_error`.
186
- */
187
- function modelWorkAnswer(runner, result) {
176
+ /** A successful reply's text, unwrapped from its harness's framing (claude's `--output-format json` result envelope, for one). */
177
+ export function unwrapHarnessReply(runner, result) {
188
178
  const extractor = runner.kind === "agent" ? getHarness(runner.profile.platform ?? runner.profile.name)?.resultExtractor : undefined;
189
- const extracted = extractor ? extractor(result) : { text: result.stdout };
179
+ return extractor ? extractor(result) : { text: result.stdout };
180
+ }
181
+ /** A model-work reply's answer: its harness's framing stripped, and no answer is a `parse_error`. */
182
+ function modelWorkAnswer(runner, result) {
183
+ const extracted = unwrapHarnessReply(runner, result);
190
184
  const answer = {
191
185
  ...result,
192
186
  stdout: extracted.text,
@@ -56,8 +56,7 @@
56
56
  * session model (plan §"Session, MCP, and identity across harnesses" —
57
57
  * Aider is the plan's named example).
58
58
  * - **inference** — not translated: the shared lowering reports each field of
59
- * the request's inference as untranslated (`inference` in `harnesses/ids.ts`
60
- * lists none for this harness).
59
+ * the request's inference as untranslated.
61
60
  *
62
61
  * Registered: `aiderBuilder` is `AiderHarness.agentBuilder` (`./index.ts`),
63
62
  * one of the ten harnesses `HARNESS_REGISTRY` constructs
@@ -51,8 +51,7 @@
51
51
  * — Q then refuses untrusted tool actions in non-interactive mode, which is
52
52
  * the conservative failure mode.
53
53
  * - **inference** — not translated: the shared lowering reports each field of
54
- * the request's inference as untranslated (`inference` in `harnesses/ids.ts`
55
- * lists none for this harness).
54
+ * the request's inference as untranslated.
56
55
  *
57
56
  * Registered: `amazonqBuilder` is `AmazonqHarness.agentBuilder`
58
57
  * (`./index.ts`), one of the ten harnesses `HARNESS_REGISTRY` constructs