akm-cli 0.9.25-alpha.1 → 0.9.25-alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/CHANGELOG.md +159 -280
  2. package/dist/cli.js +1 -1
  3. package/dist/commands/improve/consolidate/pair-pass.js +1 -0
  4. package/dist/commands/improve/consolidate.js +7 -2
  5. package/dist/commands/improve/execution.js +2 -3
  6. package/dist/commands/improve/extract.js +1 -0
  7. package/dist/commands/improve/improve-cli.js +33 -1
  8. package/dist/commands/improve/loop-stages.js +3 -0
  9. package/dist/commands/improve/reflect-noise.js +125 -0
  10. package/dist/commands/improve/reflect.js +13 -16
  11. package/dist/commands/improve/retrieval-gate.js +7 -2
  12. package/dist/commands/improve/stage.js +31 -39
  13. package/dist/commands/proposal/drain.js +4 -7
  14. package/dist/commands/proposal/propose.js +2 -11
  15. package/dist/commands/proposal/validators/proposal-quality-validators.js +4 -2
  16. package/dist/commands/read/search-cli.js +0 -38
  17. package/dist/core/config/schema/engines.js +15 -33
  18. package/dist/core/config/schema/improve-processes.js +16 -0
  19. package/dist/core/redaction.js +4 -0
  20. package/dist/core/spawn-env.js +25 -0
  21. package/dist/core/structured.js +1 -1
  22. package/dist/execution/source.js +8 -12
  23. package/dist/integrations/agent/config.js +1 -3
  24. package/dist/integrations/agent/engine-resolution.js +0 -3
  25. package/dist/integrations/agent/execution.js +14 -13
  26. package/dist/integrations/agent/index.js +1 -1
  27. package/dist/integrations/agent/model-map.js +15 -16
  28. package/dist/integrations/agent/profiles.js +2 -2
  29. package/dist/integrations/agent/prompts.js +3 -39
  30. package/dist/integrations/agent/request-lowering.js +9 -7
  31. package/dist/integrations/agent/runner-dispatch.js +25 -31
  32. package/dist/integrations/harnesses/aider/agent-builder.js +1 -2
  33. package/dist/integrations/harnesses/amazonq/agent-builder.js +1 -2
  34. package/dist/integrations/harnesses/claude/agent-builder.js +4 -16
  35. package/dist/integrations/harnesses/codex/agent-builder.js +1 -2
  36. package/dist/integrations/harnesses/ids.js +10 -16
  37. package/dist/integrations/harnesses/opencode/agent-builder.js +14 -24
  38. package/dist/integrations/harnesses/opencode/model-config.js +15 -62
  39. package/dist/integrations/harnesses/opencode/model-work-agent.js +71 -36
  40. package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -6
  41. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +32 -90
  42. package/dist/integrations/harnesses/openhands/agent-builder.js +1 -2
  43. package/dist/integrations/harnesses/pi/agent-builder.js +1 -2
  44. package/dist/llm/feature-gate.js +2 -5
  45. package/dist/llm/index-passes.js +2 -2
  46. package/dist/llm/structured-call.js +5 -5
  47. package/dist/output/shapes/passthrough.js +1 -0
  48. package/dist/scripts/akm-migrate-node.js +170 -194
  49. package/dist/scripts/akm-migrate.js +170 -194
  50. package/dist/workflows/exec/unit-dispatch.js +4 -13
  51. package/docs/reference/cli.md +12 -7
  52. package/docs/reference/configuration.md +78 -82
  53. package/docs/reference/data-and-telemetry.md +2 -3
  54. package/docs/reference/workflow-schema.md +6 -9
  55. package/package.json +1 -1
  56. package/schemas/akm-config.json +108 -36
@@ -103,6 +103,21 @@ const fidelityCheckField = z.object({ enabled: z.boolean().optional() }).passthr
103
103
  * byte-identical behaviour). Reflect process only.
104
104
  */
105
105
  const lowValueFilterField = z.object({ enabled: z.boolean().optional() }).passthrough().optional();
106
+ /**
107
+ * The wording lists of reflect's pre-judge defect filter (`findReflectDefect`).
108
+ * Each list is optional: one that is set replaces that rule's default list, and
109
+ * an empty one turns the rule off. Phrases (`placeholders`, `metaCommentary`)
110
+ * match as whole words in any case; `frontmatterKeys` are exact key names.
111
+ * Reflect process only.
112
+ */
113
+ const defectFilterField = z
114
+ .object({
115
+ placeholders: z.array(nonEmptyString).optional(),
116
+ metaCommentary: z.array(nonEmptyString).optional(),
117
+ frontmatterKeys: z.array(nonEmptyString).optional(),
118
+ })
119
+ .passthrough()
120
+ .optional();
106
121
  /**
107
122
  * #626 — extract process: pre-LLM heuristic triage gate. When enabled, a
108
123
  * deterministic scorer decides BEFORE the extraction LLM call whether a
@@ -159,6 +174,7 @@ const REFLECT_PROCESS_FIELDS = {
159
174
  limit: processLimitField,
160
175
  qualityGate: qualityGateField,
161
176
  lowValueFilter: lowValueFilterField,
177
+ defectFilter: defectFilterField,
162
178
  };
163
179
  const DISTILL_PROCESS_FIELDS = {
164
180
  allowedTypes: allowedTypesField,
@@ -19,6 +19,10 @@ const ENV_PASSTHROUGH_REDACTION_POLICY = {
19
19
  OPENCODE_CONFIG: "path",
20
20
  CLAUDE_CONFIG: "path",
21
21
  CODEX_CONFIG: "path",
22
+ XDG_CONFIG_HOME: "path",
23
+ XDG_DATA_HOME: "path",
24
+ XDG_CACHE_HOME: "path",
25
+ XDG_STATE_HOME: "path",
22
26
  AWS_PROFILE: "identifier",
23
27
  AWS_REGION: "identifier",
24
28
  LLM_MODEL: "identifier",
@@ -45,6 +45,31 @@ export const COMMON_SPAWN_ENV_PASSTHROUGH = [
45
45
  "TMPDIR",
46
46
  "AKM_EVENT_SOURCE",
47
47
  ];
48
+ /**
49
+ * The XDG base-directory variables. opencode resolves its config, data, cache
50
+ * and state directories from them, so it must receive them: an akm that runs
51
+ * under a custom `XDG_CONFIG_HOME` otherwise spawns an opencode that reads
52
+ * `$HOME/.config/opencode` and misses the provider config its caller named.
53
+ *
54
+ * Deliberately NOT part of {@link COMMON_SPAWN_ENV_PASSTHROUGH}, which is every
55
+ * harness's baseline, the workflow exec unit's default allowlist (a documented
56
+ * list) and, through profile `envPassthrough`, frozen into workflow plans.
57
+ * codex, gemini and pi keep their own dotdirs under `$HOME`, and handing the
58
+ * names to a shell command would redirect the `git` and `gh` config it reads. A
59
+ * harness that reads them asks for them by name: the opencode profile's list,
60
+ * and the opencode-sdk server's allowlist (`opencodeSdkServerEnvironmentNames`).
61
+ *
62
+ * A name added to a profile's list changes the plans frozen after it (their
63
+ * bytes, so their `plan_hash`); a stored plan keeps the list it was frozen with
64
+ * and still resumes, because nothing gates on that hash. The SDK server's
65
+ * allowlist is not part of a plan, so it takes the names by code.
66
+ */
67
+ export const XDG_BASE_DIR_ENV_PASSTHROUGH = [
68
+ "XDG_CONFIG_HOME",
69
+ "XDG_DATA_HOME",
70
+ "XDG_CACHE_HOME",
71
+ "XDG_STATE_HOME",
72
+ ];
48
73
  /**
49
74
  * The names Windows itself requires of ANY child, whatever the caller's
50
75
  * allowlist says. Applied at build time rather than added to
@@ -34,7 +34,7 @@ import { parseEmbeddedJsonResponse } from "./parse.js";
34
34
  export function withSchemaInstruction(prompt, schema) {
35
35
  return `${prompt}\n\nRespond with ONLY a JSON value matching this JSON Schema (no prose, no code fences):\n${JSON.stringify(schema)}`;
36
36
  }
37
- function defaultFeedback(failure) {
37
+ export function defaultFeedback(failure) {
38
38
  if (failure.reason === "parse_error") {
39
39
  return "Your previous response contained no parseable JSON. Respond with ONLY a JSON value that matches the requested schema — no prose, no code fences.";
40
40
  }
@@ -6,19 +6,15 @@ export const EXECUTION_SOURCE_SCHEMA_VERSION = 1;
6
6
  /** Current internal adapter identifiers are lowercase kebab-case registry keys. */
7
7
  export const EXECUTION_ADAPTER_ID_PATTERN = /^[a-z][a-z0-9]*(?:-[a-z0-9]+)*$/;
8
8
  /**
9
- * The model-work tool policy: unattended model work (improve, the judges,
10
- * index passes, remember) may read, edit only inside the dispatch's own
11
- * scratch working directory, and run `akm search` and `akm show`. The stash
12
- * stays read-only to it. A transport grants what it can confine and refuses
13
- * the policy at build when it can confine nothing; an LLM has no tools.
9
+ * The model-work tool policy: unattended model work (improve, the judges, index
10
+ * passes, remember) may read, edit only inside the dispatch's own scratch
11
+ * working directory, and run `akm search` and `akm show`; the stash stays
12
+ * read-only to it. A transport grants what it can confine and refuses the policy
13
+ * at build when it can confine nothing; an LLM has no tools. A request carries it
14
+ * as `authorization.policy.id`, set only when its caller asks (`modelWork`): no
15
+ * `tools` value names it, so an asset's own `tools:` cannot.
14
16
  */
15
- export const MODEL_WORK_TOOLS = Object.freeze(["read", "edit", "akm search", "akm show"]);
16
- /** Whether a selection names the model-work tool policy. */
17
- export function isModelWorkTools(tools) {
18
- return (Array.isArray(tools) &&
19
- tools.length === MODEL_WORK_TOOLS.length &&
20
- tools.every((tool, index) => tool === MODEL_WORK_TOOLS[index]));
21
- }
17
+ export const MODEL_WORK_POLICY_ID = "model-work";
22
18
  function requireRecord(value, path) {
23
19
  if (value === null || typeof value !== "object" || Array.isArray(value)) {
24
20
  throw new TypeError(`${path} must be an object`);
@@ -3,7 +3,5 @@
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  /** Default agent CLI timeout; null means agents run until they finish. */
5
5
  export const DEFAULT_AGENT_TIMEOUT_MS = null;
6
- /** Default hard timeout for direct LLM calls when no engine/use override exists. */
6
+ /** Default hard timeout for a direct LLM call, and the bound on model work on every runner kind, when no engine/use override exists. */
7
7
  export const DEFAULT_LLM_TIMEOUT_MS = 600_000;
8
- /** Default bound on model work (structured calls, judgments) on every runner kind. */
9
- export const DEFAULT_MODEL_WORK_TIMEOUT_MS = 600_000;
@@ -13,7 +13,6 @@ import { collectSensitiveValues } from "../../core/redaction.js";
13
13
  import { resolveSecretFromStore } from "../../sources/snapshot-fetchers/secret-seam.js";
14
14
  import { getHarness } from "../harnesses/index.js";
15
15
  import { DEFAULT_LLM_TIMEOUT_MS } from "./config.js";
16
- import { engineModelAndInference } from "./model-map.js";
17
16
  import { getBuiltinAgentProfile, OPENCODE_SDK_SERVER_BIN } from "./profiles.js";
18
17
  const LLM_CONNECTION_FIELDS = [
19
18
  "provider",
@@ -274,7 +273,6 @@ function lowerAgentEngine(name, engine, config) {
274
273
  const platform = harness.id;
275
274
  const sdk = platform === "opencode-sdk";
276
275
  const builtin = getBuiltinAgentProfile(platform);
277
- const { inference } = engineModelAndInference(engine);
278
276
  const profile = {
279
277
  name,
280
278
  platform,
@@ -287,7 +285,6 @@ function lowerAgentEngine(name, engine, config) {
287
285
  parseOutput: "text",
288
286
  ...(engine.workspace ? { workspace: path.resolve(engine.workspace) } : {}),
289
287
  ...(engine.model ? { model: engine.model } : {}),
290
- ...(inference ? { inference } : {}),
291
288
  };
292
289
  // An engine that sets no timeoutMs leaves it unset, so a caller's own default
293
290
  // (model work's 600 s) can apply; a dispatch with none runs unbounded.
@@ -6,7 +6,7 @@ import { ConfigError } from "../../core/errors.js";
6
6
  import { DURATION_UNITS, parseDuration } from "../../core/time.js";
7
7
  import { EXECUTION_MAX_TIMEOUT_MS } from "../../execution/limits.js";
8
8
  import { createInlineResolvedCommand, createResolvedExecutionRequest, decodeResolvedExecutionRequest, } from "../../execution/resolved-request.js";
9
- import { cloneToolSelection, isModelWorkTools, isPortableExecutionAgentSelector, } from "../../execution/source.js";
9
+ import { cloneToolSelection, isPortableExecutionAgentSelector, MODEL_WORK_POLICY_ID, } from "../../execution/source.js";
10
10
  import { getHarness } from "../harnesses/index.js";
11
11
  import { DEFAULT_LLM_TIMEOUT_MS } from "./config.js";
12
12
  import { FALLBACK_ENGINE_NAME, fallbackEngineConfig, NO_ENGINE_MESSAGE_SUFFIX, NO_ENGINE_REMEDY, } from "./engine-fallback.js";
@@ -78,11 +78,8 @@ function engineDefaults(name, engine, config) {
78
78
  modelMapKey: own.model === undefined ? fallbackName : "opencode-sdk",
79
79
  values: {
80
80
  ...(inherited.model !== undefined ? { model: inherited.model } : {}),
81
+ ...(inherited.inference !== undefined ? { inference: inherited.inference } : {}),
81
82
  ...values,
82
- // The engine's own inference is added over the fallback's, field by field.
83
- ...(inherited.inference !== undefined || own.inference !== undefined
84
- ? { inference: { ...inherited.inference, ...own.inference } }
85
- : {}),
86
83
  timeout: Object.hasOwn(engine, "timeoutMs")
87
84
  ? (engine.timeoutMs ?? null)
88
85
  : Object.hasOwn(fallback, "timeoutMs")
@@ -191,12 +188,16 @@ function requestedToolNames(tools) {
191
188
  * nothing is allowed. The model-work policy is akm's own and always allowed:
192
189
  * it confines an engine more tightly than leaving tools unset does.
193
190
  */
194
- function authorizeTools(tools, config) {
191
+ function authorizeTools(tools, config, modelWork) {
192
+ if (modelWork) {
193
+ return {
194
+ status: "allowed",
195
+ reason: "The model-work tool policy is akm's own.",
196
+ policy: { id: MODEL_WORK_POLICY_ID },
197
+ };
198
+ }
195
199
  if (!hasToolSelection(tools))
196
200
  return { status: "not-required" };
197
- if (isModelWorkTools(tools)) {
198
- return { status: "allowed", reason: "The model-work tool policy is akm's own.", policy: { id: "model-work" } };
199
- }
200
201
  if (!config) {
201
202
  return {
202
203
  status: "denied",
@@ -392,13 +393,14 @@ export function resolveExecution(input) {
392
393
  return layer;
393
394
  };
394
395
  const schemaLayer = select("outputSchema", "outputSchema");
395
- const toolsLayer = select("tools", "tools");
396
+ const modelWork = input.modelWork === true;
397
+ const toolsLayer = modelWork ? undefined : select("tools", "tools");
396
398
  const timeoutLayer = select("timeout", "runtime.timeoutMs");
397
399
  const workspaceLayer = select("workspace", "runtime.workspace");
398
400
  const environmentLayer = select("environment", "runtime.environment");
399
401
  const settingsLayer = select("runtime", "runtime.settings");
400
402
  const tools = toolsLayer ? cloneToolSelection(toolsLayer.values.tools ?? null, "tools") : undefined;
401
- const authorization = authorizeTools(tools, input.runner ? undefined : input.config);
403
+ const authorization = authorizeTools(tools, input.runner ? undefined : input.config, modelWork);
402
404
  provenance.authorization = {
403
405
  layer: typeof authorization.policy?.id === "string" ? authorization.policy.id : "not-required",
404
406
  kind: "authorization",
@@ -438,8 +440,7 @@ function buildLlm(request, runner, options) {
438
440
  if (typeof request.agent === "string" && request.persona === null) {
439
441
  throw new ConfigError(`The direct LLM transport cannot consume native agent selector ${JSON.stringify(request.agent)}.`, "INVALID_CONFIG_FILE");
440
442
  }
441
- // An LLM has no tools, which already meets the model-work policy.
442
- if (hasToolSelection(request.tools) && !isModelWorkTools(request.tools)) {
443
+ if (hasToolSelection(request.tools)) {
443
444
  throw new ConfigError("The direct LLM transport cannot enforce the resolved tool policy.", "INVALID_CONFIG_FILE");
444
445
  }
445
446
  const notices = [...request.notices];
@@ -4,4 +4,4 @@
4
4
  export { DEFAULT_AGENT_TIMEOUT_MS } from "./config.js";
5
5
  export { _setAgentDetectForTests, defaultWhich, detectAgentCliProfiles, pickDefaultAgentProfile } from "./detect.js";
6
6
  export { BUILTIN_AGENT_PROFILE_NAMES, getBuiltinAgentProfile, listBuiltinAgentProfiles, } from "./profiles.js";
7
- export { buildProposePrompt, buildReflectPrompt, buildSchemaRepairPrompt, parseAgentProposalPayload, } from "./prompts.js";
7
+ export { buildProposePrompt, buildReflectPrompt, parseAgentProposalPayload } from "./prompts.js";
@@ -220,33 +220,32 @@ function mergeProfiles(base, overlay) {
220
220
  * copied verbatim from the engine's own config value; it must already be
221
221
  * meaningful for the model-map column's platform (akm does not translate
222
222
  * between an engine's connection and an agent platform's own provider
223
- * registry). An agent-kind engine contributes only the inference fields its
224
- * platform translates (config validation rejects the rest); an `llm` engine
225
- * also contributes `supportsJsonSchema` and `extraParams`.
223
+ * registry). Only `kind: "llm"` engines contribute inference defaults — an
224
+ * agent-kind engine's schema carries no temperature/thinking fields.
226
225
  */
227
226
  export function engineModelAndInference(engine) {
228
227
  const out = {};
229
228
  if (Object.hasOwn(engine, "model") && engine.model !== undefined)
230
229
  out.model = engine.model;
231
- const inference = {};
232
- if (Object.hasOwn(engine, "temperature"))
233
- inference.temperature = engine.temperature;
234
- if (Object.hasOwn(engine, "maxTokens"))
235
- inference.maxTokens = engine.maxTokens;
236
230
  if (engine.kind === "llm") {
231
+ const inference = {};
232
+ if (Object.hasOwn(engine, "temperature"))
233
+ inference.temperature = engine.temperature;
234
+ if (Object.hasOwn(engine, "maxTokens"))
235
+ inference.maxTokens = engine.maxTokens;
237
236
  if (Object.hasOwn(engine, "supportsJsonSchema"))
238
237
  inference.supportsJsonSchema = engine.supportsJsonSchema;
239
238
  if (Object.hasOwn(engine, "extraParams"))
240
239
  inference.extraParams = engine.extraParams;
240
+ if (Object.hasOwn(engine, "contextLength"))
241
+ inference.contextLength = engine.contextLength;
242
+ if (Object.hasOwn(engine, "enableThinking"))
243
+ inference.enableThinking = engine.enableThinking;
244
+ if (Object.hasOwn(engine, "reasoningEffort"))
245
+ inference.reasoningEffort = engine.reasoningEffort;
246
+ if (Object.keys(inference).length > 0)
247
+ out.inference = inference;
241
248
  }
242
- if (Object.hasOwn(engine, "contextLength"))
243
- inference.contextLength = engine.contextLength;
244
- if (Object.hasOwn(engine, "enableThinking"))
245
- inference.enableThinking = engine.enableThinking;
246
- if (Object.hasOwn(engine, "reasoningEffort"))
247
- inference.reasoningEffort = engine.reasoningEffort;
248
- if (Object.keys(inference).length > 0)
249
- out.inference = inference;
250
249
  return Object.freeze(out);
251
250
  }
252
251
  /** Overlay user fields over installed fields, per (alias, column), without resolving `engine` indirection. */
@@ -8,7 +8,7 @@
8
8
  * coding-agent CLI. Named engines lower canonical harness metadata into this
9
9
  * intentionally small internal shape. The wrapper is in `./spawn.ts`.
10
10
  */
11
- import { COMMON_SPAWN_ENV_PASSTHROUGH } from "../../core/spawn-env.js";
11
+ import { COMMON_SPAWN_ENV_PASSTHROUGH, XDG_BASE_DIR_ENV_PASSTHROUGH } from "../../core/spawn-env.js";
12
12
  // AKM_EVENT_SOURCE carries usage-event provenance (improve/task) so that akm
13
13
  // invocations a spawned agent makes are recorded as machine traffic, not user
14
14
  // demand (DRIFT-6). Without it in the passthrough whitelist, buildChildEnv drops
@@ -30,7 +30,7 @@ const BUILTINS = {
30
30
  bin: "opencode",
31
31
  args: ["run"],
32
32
  stdio: "interactive",
33
- envPassthrough: [...COMMON_PASSTHROUGH, "OPENCODE_API_KEY", "OPENCODE_CONFIG"],
33
+ envPassthrough: [...COMMON_PASSTHROUGH, "OPENCODE_API_KEY", "OPENCODE_CONFIG", ...XDG_BASE_DIR_ENV_PASSTHROUGH],
34
34
  parseOutput: "text",
35
35
  },
36
36
  claude: {
@@ -341,7 +341,8 @@ export function buildReflectPrompt(input) {
341
341
  // Embed concrete counts only when the gate will actually fire (source >= 200 chars).
342
342
  const showCharBounds = sourceBodyLen >= 200;
343
343
  const minChars = Math.max(Math.round(0.5 * sourceBodyLen), 150);
344
- const maxChars = Math.min(Math.max(Math.round(2.5 * sourceBodyLen), 2500), 25000);
344
+ // A source already past the 25000 cap may not grow, and need not shrink to the cap.
345
+ const maxChars = Math.max(Math.min(Math.max(Math.round(2.5 * sourceBodyLen), 2500), 25000), sourceBodyLen);
345
346
  sections.push([
346
347
  "## Content preservation rules (MUST follow)",
347
348
  "1. PRESERVE ALL concrete content: code blocks, fenced snippets, CLI commands, numbered/bulleted checklists, tables, YAML/JSON examples, file paths, configuration keys, environment variable names, and CSS/HTML selectors. These are load-bearing — do NOT replace them with prose summaries.",
@@ -350,7 +351,7 @@ export function buildReflectPrompt(input) {
350
351
  ? `3. DO NOT shrink the asset. Your body must be at least ${minChars} characters (source body is ${sourceBodyLen} chars; floor is 50%). If you genuinely need to remove a major section, explain why in a comment line at the top of the body (e.g. \`<!-- removed obsolete section X because ... -->\`).`
351
352
  : "3. DO NOT shrink the asset dramatically. The improved body must be at least 50% of the source body length. If you genuinely need to remove a major section, explain why in a comment line at the top of the body (e.g. `<!-- removed obsolete section X because ... -->`).",
352
353
  showCharBounds
353
- ? `4. DO NOT pad the asset with speculative material. Your body must be at most ${maxChars} characters (source body is ${sourceBodyLen} chars; ceiling is 250%). Do not add invented sections, hypothetical examples, or padding prose.`
354
+ ? `4. DO NOT pad the asset with speculative material. Your body must be at most ${maxChars} characters (source body is ${sourceBodyLen} chars; ceiling is ${maxChars === sourceBodyLen ? "100%" : "250%"}). Do not add invented sections, hypothetical examples, or padding prose.`
354
355
  : "4. DO NOT pad the asset with speculative material. The improved body must be at most 250% of the source body length unless the feedback explicitly requests added sections.",
355
356
  "5. Improve clarity of surrounding prose, fix structural issues, add missing required frontmatter fields. Do NOT rewrite a runbook into an essay.",
356
357
  ].join("\n"));
@@ -388,43 +389,6 @@ export function buildProposePrompt(input) {
388
389
  sections.push(RESPONSE_CONTRACT_JSON);
389
390
  return sections.join("\n\n");
390
391
  }
391
- /**
392
- * Build the prompt for the schema repair pass in `akm improve`. Asks the
393
- * agent to add the minimal required frontmatter to an asset that failed
394
- * validation — without rewriting the body.
395
- */
396
- export function buildSchemaRepairPrompt(input) {
397
- const sections = [];
398
- sections.push(`This ${input.type} asset failed schema validation with the error: "${input.reason}". ` +
399
- `Your task is to fix the schema issue by adding or correcting the missing/invalid field(s) ` +
400
- `while preserving all existing content.`);
401
- sections.push(`Target ref: ${input.ref}`);
402
- sections.push(`Schema requirements for ${input.type} assets: ${hintForType(input.type)}`);
403
- if (input.standardsContext?.trim()) {
404
- sections.push("Standards to follow (the rulebook for this target):");
405
- sections.push(input.standardsContext.trim());
406
- }
407
- {
408
- const authoringRules = authoringRulesForType(input.type);
409
- if (authoringRules) {
410
- sections.push(authoringRules);
411
- }
412
- }
413
- const CONTENT_CAP = 3000;
414
- const body = input.assetContent.trimEnd();
415
- const truncated = body.length > CONTENT_CAP;
416
- sections.push("Current asset content (first 3000 chars — sufficient to generate missing frontmatter):");
417
- sections.push("```");
418
- sections.push(truncated ? `${body.slice(0, CONTENT_CAP)}\n... [truncated]` : body);
419
- sections.push("```");
420
- sections.push("Produce the minimal fix: add ONLY the missing required frontmatter field(s). " +
421
- "Do not rewrite the body unless it is empty. " +
422
- "If `description` is missing, generate a concise one-sentence description from the content. " +
423
- "If `when_to_use` is missing, generate a one-line trigger sentence. " +
424
- "Preserve all existing frontmatter keys and the full body verbatim.");
425
- sections.push(RESPONSE_CONTRACT_JSON);
426
- return sections.join("\n\n");
427
- }
428
392
  /**
429
393
  * Parse agent stdout into a proposal payload. The agent contract requires a
430
394
  * single JSON object; anything else is reported as a parse error so callers
@@ -3,7 +3,8 @@
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  import { ConfigError } from "../../core/errors.js";
5
5
  import { withSchemaInstruction } from "../../core/structured.js";
6
- import { isModelWorkTools } from "../../execution/source.js";
6
+ import { MODEL_WORK_POLICY_ID } from "../../execution/source.js";
7
+ import { HARNESS_MODEL_WORK_IDS } from "../harnesses/ids.js";
7
8
  import { composeConversationFallbackPrompt } from "./conversation-fallback.js";
8
9
  import { composePersonaFallbackPrompt } from "./persona-fallback.js";
9
10
  /** A tool selection that actually names tools (not omitted, null, or empty). */
@@ -47,8 +48,9 @@ export function extensionFields(request) {
47
48
  * composing into the prompt what the harness has no channel for.
48
49
  */
49
50
  export function createAgentRequestLowerer(options) {
50
- return (profile, request) => {
51
- const supportedInference = new Set(typeof options.inference === "function" ? options.inference(profile, request) : (options.inference ?? []));
51
+ const supportedInference = new Set(options.inference ?? []);
52
+ return (_profile, request) => {
53
+ const modelWork = request.authorization.policy?.id === MODEL_WORK_POLICY_ID;
52
54
  const notices = [];
53
55
  const skip = (field) => {
54
56
  notices.push(untranslated(options.adapter, field));
@@ -92,19 +94,19 @@ export function createAgentRequestLowerer(options) {
92
94
  if (Object.hasOwn(request, "inference")) {
93
95
  dispatch.inference = request.inference ?? null;
94
96
  for (const key of Object.keys(request.inference ?? {}).sort()) {
95
- if (!supportedInference.has(key))
97
+ if (!(modelWork && supportedInference.has(key)))
96
98
  skip(`inference.${key}`);
97
99
  }
98
100
  }
99
- if (isModelWorkTools(request.tools)) {
100
- if (!options.modelWorkTools) {
101
+ if (modelWork) {
102
+ if (!HARNESS_MODEL_WORK_IDS.has(options.adapter)) {
101
103
  throw new ConfigError(`The ${options.adapter} transport cannot enforce the model-work tool policy.`, "INVALID_CONFIG_FILE");
102
104
  }
103
105
  // The builder selects its own confined agent or flags; another agent would replace them.
104
106
  if (dispatch.agent) {
105
107
  throw new ConfigError(`The ${options.adapter} transport cannot run native agent ${JSON.stringify(dispatch.agent)} under the model-work tool policy.`, "INVALID_CONFIG_FILE");
106
108
  }
107
- dispatch.tools = request.tools;
109
+ dispatch.modelWork = true;
108
110
  }
109
111
  else if (request.tools !== undefined) {
110
112
  // An explicit empty selection still reaches the builder (e.g. an empty allowlist).
@@ -12,13 +12,14 @@ import fs from "node:fs";
12
12
  import os from "node:os";
13
13
  import path from "node:path";
14
14
  import { assertNever } from "../../core/assert.js";
15
- import { ConfigError, UsageError } from "../../core/errors.js";
15
+ import { UsageError } from "../../core/errors.js";
16
16
  import { collectSensitiveValues, isEnvPassthroughValueSafeToExpose, redactSensitiveText, redactSensitiveValue, } from "../../core/redaction.js";
17
- import { isModelWorkTools } from "../../execution/source.js";
17
+ import { MODEL_WORK_POLICY_ID } from "../../execution/source.js";
18
18
  import { chatCompletion, LlmCallError } from "../../llm/client.js";
19
19
  import { emitLlmUsage } from "../../llm/usage-telemetry.js";
20
20
  import { getHarness } from "../harnesses/index.js";
21
21
  import { closeServer as disposeOpencodeSdkServers, runOpencodeSdk } from "../harnesses/opencode-sdk/sdk-runner.js";
22
+ import { modelFromArgs } from "./builder-shared.js";
22
23
  import { lookupApiKeyFileValue, lookupApiKeySecretRefValue, lookupCredentialFromEnv, resolveEngine, } from "./engine-resolution.js";
23
24
  import { materializeLlmRunnerConnection, materializeSdkFallbackConnection } from "./runner.js";
24
25
  import { runAgent } from "./spawn.js";
@@ -86,18 +87,28 @@ const USAGE_ERROR_CODES = {
86
87
  parse_error: "parse_error",
87
88
  llm_rate_limit: "rate_limited",
88
89
  };
90
+ /**
91
+ * The model an agent or SDK dispatch ran: the request's, else the one an agent
92
+ * CLI's own `args` select, which its command carries when the request names
93
+ * none. An SDK server never sees `args`, so an SDK engine with no model named
94
+ * has none to report: opencode picks it.
95
+ */
96
+ function dispatchedModel({ request, runner }) {
97
+ return request.model?.resolved ?? (runner.kind === "agent" ? modelFromArgs(runner.profile.args) : undefined);
98
+ }
89
99
  /**
90
100
  * One usage record for an agent or SDK dispatch, through the same sink and
91
101
  * ambient `withLlmStage` attribution as the LLM transport's per-HTTP-attempt
92
- * records: the request's model, and tokens when the runner reported them.
102
+ * records: the model it ran, and tokens when the runner reported them.
93
103
  */
94
104
  function recordDispatchUsage(execution, result) {
95
105
  const { inputTokens, outputTokens, reasoningTokens } = result.usage ?? {};
96
106
  const reported = [inputTokens, outputTokens, reasoningTokens].filter((count) => count !== undefined);
107
+ const model = dispatchedModel(execution);
97
108
  emitLlmUsage({
98
109
  outcome: result.ok ? "success" : "error",
99
110
  modelSource: "configured",
100
- ...(execution.request.model?.resolved ? { model: execution.request.model.resolved } : {}),
111
+ ...(model ? { model } : {}),
101
112
  durationMs: result.durationMs,
102
113
  ...(inputTokens !== undefined ? { promptTokens: inputTokens } : {}),
103
114
  ...(outputTokens !== undefined ? { completionTokens: outputTokens } : {}),
@@ -135,30 +146,13 @@ async function dispatchRunner(runner, prompt, opts, seams, llm) {
135
146
  }
136
147
  return redactResult(result, collectSensitiveValues(secrets));
137
148
  }
138
- /** The git repository a directory is inside, if any: a `.git` file, or a `.git` directory with a HEAD. */
139
- function enclosingGitRepository(dir) {
140
- for (let current = fs.realpathSync(dir);; current = path.dirname(current)) {
141
- const marker = path.join(current, ".git");
142
- if (fs.existsSync(path.join(marker, "HEAD")) || (fs.existsSync(marker) && fs.statSync(marker).isFile())) {
143
- return current;
144
- }
145
- if (path.dirname(current) === current)
146
- return undefined;
147
- }
148
- }
149
149
  /**
150
150
  * The scratch working directory for one model-work dispatch on an agent or SDK
151
151
  * engine, so the edit the model-work tool policy grants never reaches the
152
- * stash. opencode counts a whole git repository as inside its working
153
- * directory, so a scratch directory inside one is refused.
152
+ * stash.
154
153
  */
155
154
  function createModelWorkDirectory() {
156
- const dir = fs.mkdtempSync(path.join(os.tmpdir(), "akm-model-work-"));
157
- const repository = enclosingGitRepository(dir);
158
- if (repository === undefined)
159
- return dir;
160
- fs.rmSync(dir, { recursive: true, force: true });
161
- throw new ConfigError(`Model work runs an agent in a scratch directory outside any git repository, but ${dir} is inside the repository at ${repository}.`, "INVALID_CONFIG_FILE", "Point TMPDIR at a directory outside any git repository.");
155
+ return fs.mkdtempSync(path.join(os.tmpdir(), "akm-model-work-"));
162
156
  }
163
157
  /**
164
158
  * Run a built execution. Credentials are read here, once per call, and never
@@ -168,7 +162,7 @@ function createModelWorkDirectory() {
168
162
  * limit, for one) has failed with `parse_error`.
169
163
  */
170
164
  export async function runExecution(execution, options = {}) {
171
- const scratch = execution.runner.kind !== "llm" && isModelWorkTools(execution.request.tools)
165
+ const scratch = execution.runner.kind !== "llm" && execution.request.authorization.policy?.id === MODEL_WORK_POLICY_ID
172
166
  ? createModelWorkDirectory()
173
167
  : undefined;
174
168
  try {
@@ -179,14 +173,14 @@ export async function runExecution(execution, options = {}) {
179
173
  fs.rmSync(scratch, { recursive: true, force: true });
180
174
  }
181
175
  }
182
- /**
183
- * A model-work reply's answer: the harness's result extractor strips its
184
- * framing (claude's `--output-format json` result envelope, for one), as a
185
- * workflow unit's does, and no answer is a `parse_error`.
186
- */
187
- function modelWorkAnswer(runner, result) {
176
+ /** A successful reply's text, unwrapped from its harness's framing (claude's `--output-format json` result envelope, for one). */
177
+ export function unwrapHarnessReply(runner, result) {
188
178
  const extractor = runner.kind === "agent" ? getHarness(runner.profile.platform ?? runner.profile.name)?.resultExtractor : undefined;
189
- const extracted = extractor ? extractor(result) : { text: result.stdout };
179
+ return extractor ? extractor(result) : { text: result.stdout };
180
+ }
181
+ /** A model-work reply's answer: its harness's framing stripped, and no answer is a `parse_error`. */
182
+ function modelWorkAnswer(runner, result) {
183
+ const extracted = unwrapHarnessReply(runner, result);
190
184
  const answer = {
191
185
  ...result,
192
186
  stdout: extracted.text,
@@ -56,8 +56,7 @@
56
56
  * session model (plan §"Session, MCP, and identity across harnesses" —
57
57
  * Aider is the plan's named example).
58
58
  * - **inference** — not translated: the shared lowering reports each field of
59
- * the request's inference as untranslated (`inference` in `harnesses/ids.ts`
60
- * lists none for this harness).
59
+ * the request's inference as untranslated.
61
60
  *
62
61
  * Registered: `aiderBuilder` is `AiderHarness.agentBuilder` (`./index.ts`),
63
62
  * one of the ten harnesses `HARNESS_REGISTRY` constructs
@@ -51,8 +51,7 @@
51
51
  * — Q then refuses untrusted tool actions in non-interactive mode, which is
52
52
  * the conservative failure mode.
53
53
  * - **inference** — not translated: the shared lowering reports each field of
54
- * the request's inference as untranslated (`inference` in `harnesses/ids.ts`
55
- * lists none for this harness).
54
+ * the request's inference as untranslated.
56
55
  *
57
56
  * Registered: `amazonqBuilder` is `AmazonqHarness.agentBuilder`
58
57
  * (`./index.ts`), one of the ten harnesses `HARNESS_REGISTRY` constructs
@@ -28,7 +28,6 @@
28
28
  *
29
29
  * The builder's `platform` stays `'claude'` (the canonical harness id).
30
30
  */
31
- import { isModelWorkTools } from "../../../execution/source.js";
32
31
  import { modelFromArgs, normalizeTools, resolveDispatchModel, } from "../../agent/builder-shared.js";
33
32
  import { createAgentRequestLowerer } from "../../agent/request-lowering.js";
34
33
  /** The model-work tool policy on Claude Code: read, edit in the working directory, `akm search`, `akm show`. */
@@ -45,17 +44,11 @@ export const MODEL_WORK_CLAUDE_FLAGS = Object.freeze([
45
44
  /**
46
45
  * Claude Code builder.
47
46
  * Command shape:
48
- * claude [--agent <name>] [--system-prompt "..."] [--model <m>] [--effort <level>]
49
- * [--allowedTools <t>] [--output-format json] --print -- "<prompt>"
47
+ * claude [--agent <name>] [--system-prompt "..."] [--model <m>] [--allowedTools <t>]
48
+ * [--output-format json] --print -- "<prompt>"
50
49
  *
51
50
  * --print switches Claude Code to non-interactive captured output mode.
52
51
  *
53
- * `--effort` carries the request's `reasoningEffort` as the harness's own
54
- * level (`low`, `medium`, `high`, `xhigh` or `max` in 2.1.283), passed through
55
- * as given: a value Claude Code rejects fails the dispatch. It is the only
56
- * inference field Claude Code takes on its command line, so the others are
57
- * reported as untranslated.
58
- *
59
52
  * The model-work tool policy lowers to {@link MODEL_WORK_CLAUDE_FLAGS} in place
60
53
  * of the engine's `args` and `--allowedTools`, verified against Claude Code
61
54
  * 2.1.283 and a local stub:
@@ -78,11 +71,9 @@ export const claudeBuilder = {
78
71
  personaChannel: "native",
79
72
  nativeAgentSelector: true,
80
73
  tools: "all",
81
- modelWorkTools: true,
82
- inference: ["reasoningEffort"],
83
74
  }),
84
75
  build(profile, req) {
85
- const modelWork = isModelWorkTools(req.tools);
76
+ const modelWork = req.modelWork === true;
86
77
  const args = modelWork ? [...MODEL_WORK_CLAUDE_FLAGS] : [...profile.args];
87
78
  if (req.agent) {
88
79
  args.push("--agent", req.agent);
@@ -99,10 +90,7 @@ export const claudeBuilder = {
99
90
  if (model)
100
91
  args.push("--model", model);
101
92
  }
102
- const effort = req.inference?.reasoningEffort;
103
- if (typeof effort === "string" && effort.length > 0)
104
- args.push("--effort", effort);
105
- if (req.tools && !modelWork) {
93
+ if (req.tools) {
106
94
  args.push("--allowedTools", normalizeTools(req.tools));
107
95
  }
108
96
  if (req.schema) {
@@ -43,8 +43,7 @@
43
43
  * field yet); {@link codexResumeArgs} exposes the argv prefix for the
44
44
  * integration task that wires session-id reuse from `workflow_run_units`.
45
45
  * - The request's inference is not translated: the shared lowering reports
46
- * each field as untranslated (`inference` in `harnesses/ids.ts` lists none
47
- * for codex). codex would take `reasoningEffort` as
46
+ * each field as untranslated. codex would take `reasoningEffort` as
48
47
  * `-c model_reasoning_effort=<v>`, which is left to the integration task.
49
48
  *
50
49
  * Registered: `codexBuilder` is `CodexHarness.agentBuilder` (`./index.ts`),