akm-cli 0.9.24 → 0.9.25-alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/CHANGELOG.md +173 -0
  2. package/dist/cli.js +1 -1
  3. package/dist/commands/health/checks.js +10 -11
  4. package/dist/commands/improve/consolidate/pair-pass.js +4 -2
  5. package/dist/commands/improve/consolidate.js +10 -4
  6. package/dist/commands/improve/execution.js +4 -11
  7. package/dist/commands/improve/extract-prompt.js +4 -4
  8. package/dist/commands/improve/extract.js +11 -13
  9. package/dist/commands/improve/improve-cli.js +65 -34
  10. package/dist/commands/improve/improve-strategies.js +49 -43
  11. package/dist/commands/improve/improve-usage-report.js +8 -17
  12. package/dist/commands/improve/loop-stages.js +3 -0
  13. package/dist/commands/improve/preparation.js +3 -1
  14. package/dist/commands/improve/reflect-noise.js +125 -0
  15. package/dist/commands/improve/reflect.js +105 -172
  16. package/dist/commands/improve/retrieval-gate.js +7 -2
  17. package/dist/commands/improve/stage.js +67 -24
  18. package/dist/commands/proposal/drain.js +11 -2
  19. package/dist/commands/proposal/proposal-cli.js +1 -5
  20. package/dist/commands/proposal/propose-cli.js +2 -2
  21. package/dist/commands/proposal/propose.js +72 -84
  22. package/dist/commands/proposal/validators/proposal-quality-validators.js +4 -2
  23. package/dist/commands/read/search-cli.js +0 -38
  24. package/dist/commands/remember.js +3 -3
  25. package/dist/commands/sources/schema-repair.js +1 -1
  26. package/dist/core/config/config-schema.js +37 -60
  27. package/dist/core/config/engine-semantics.js +15 -11
  28. package/dist/core/config/schema/improve-processes.js +18 -2
  29. package/dist/core/improve-result.js +3 -3
  30. package/dist/core/redaction.js +4 -0
  31. package/dist/core/spawn-env.js +25 -0
  32. package/dist/core/structured.js +11 -1
  33. package/dist/execution/source.js +10 -0
  34. package/dist/indexer/passes/memory-inference.js +2 -1
  35. package/dist/integrations/agent/builder-shared.js +15 -0
  36. package/dist/integrations/agent/config.js +1 -1
  37. package/dist/integrations/agent/engine-resolution.js +13 -31
  38. package/dist/integrations/agent/execution.js +48 -22
  39. package/dist/integrations/agent/index.js +1 -1
  40. package/dist/integrations/agent/profiles.js +2 -2
  41. package/dist/integrations/agent/prompts.js +55 -114
  42. package/dist/integrations/agent/request-lowering.js +21 -8
  43. package/dist/integrations/agent/runner-dispatch.js +96 -3
  44. package/dist/integrations/agent/runner.js +8 -2
  45. package/dist/integrations/harnesses/aider/agent-builder.js +10 -23
  46. package/dist/integrations/harnesses/aider/index.js +0 -5
  47. package/dist/integrations/harnesses/amazonq/agent-builder.js +9 -21
  48. package/dist/integrations/harnesses/amazonq/index.js +0 -5
  49. package/dist/integrations/harnesses/claude/agent-builder.js +47 -36
  50. package/dist/integrations/harnesses/claude/index.js +0 -14
  51. package/dist/integrations/harnesses/codex/agent-builder.js +3 -4
  52. package/dist/integrations/harnesses/codex/index.js +0 -4
  53. package/dist/integrations/harnesses/copilot/agent-builder.js +4 -13
  54. package/dist/integrations/harnesses/copilot/index.js +2 -7
  55. package/dist/integrations/harnesses/gemini/agent-builder.js +4 -13
  56. package/dist/integrations/harnesses/gemini/index.js +0 -5
  57. package/dist/integrations/harnesses/ids.js +12 -10
  58. package/dist/integrations/harnesses/opencode/agent-builder.js +41 -9
  59. package/dist/integrations/harnesses/opencode/index.js +0 -8
  60. package/dist/integrations/harnesses/opencode/model-config.js +33 -0
  61. package/dist/integrations/harnesses/opencode/model-work-agent.js +115 -0
  62. package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -7
  63. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +114 -28
  64. package/dist/integrations/harnesses/openhands/agent-builder.js +7 -21
  65. package/dist/integrations/harnesses/openhands/index.js +0 -5
  66. package/dist/integrations/harnesses/pi/agent-builder.js +7 -21
  67. package/dist/integrations/harnesses/pi/index.js +0 -5
  68. package/dist/llm/client.js +5 -0
  69. package/dist/llm/feature-gate.js +12 -9
  70. package/dist/llm/index-passes.js +2 -5
  71. package/dist/llm/memory-infer.js +6 -5
  72. package/dist/llm/structured-call.js +33 -11
  73. package/dist/output/shapes/passthrough.js +1 -0
  74. package/dist/scripts/akm-migrate-node.js +298 -228
  75. package/dist/scripts/akm-migrate.js +298 -228
  76. package/dist/workflows/exec/step-work.js +6 -5
  77. package/dist/workflows/exec/unit-dispatch.js +4 -13
  78. package/dist/workflows/freeze/step-values.js +1 -1
  79. package/docs/reference/cli.md +41 -18
  80. package/docs/reference/configuration.md +165 -12
  81. package/docs/reference/data-and-telemetry.md +2 -3
  82. package/docs/reference/workflow-schema.md +10 -9
  83. package/package.json +1 -1
  84. package/schemas/akm-config.json +108 -0
@@ -0,0 +1,33 @@
1
+ // This Source Code Form is subject to the terms of the Mozilla Public
2
+ // License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ /** The inference keys model work's agent carries as options. */
5
+ export const MODEL_WORK_AGENT_INFERENCE = ["temperature", "reasoningEffort", "enableThinking"];
6
+ const isPositiveInteger = (value) => typeof value === "number" && Number.isInteger(value) && value > 0;
7
+ /**
8
+ * Split `inference` into opencode config: `entry` for a model (its options, and
9
+ * its limit), and for model work `agentOptions` for the confined agent in place
10
+ * of the entry's options. A value of the wrong type is left out.
11
+ */
12
+ export function opencodeInferenceConfig(inference, modelWork) {
13
+ const { temperature, reasoningEffort, enableThinking, maxTokens, contextLength } = inference ?? {};
14
+ const options = {};
15
+ if (typeof temperature === "number" && Number.isFinite(temperature))
16
+ options.temperature = temperature;
17
+ if (typeof reasoningEffort === "string" && reasoningEffort.length > 0)
18
+ options.reasoningEffort = reasoningEffort;
19
+ if (typeof enableThinking === "boolean") {
20
+ options.chat_template_kwargs = { enable_thinking: enableThinking };
21
+ options.enable_thinking = enableThinking;
22
+ }
23
+ const hasOptions = Object.keys(options).length > 0;
24
+ return {
25
+ entry: {
26
+ ...(hasOptions && !modelWork ? { options } : {}),
27
+ ...(isPositiveInteger(maxTokens) && isPositiveInteger(contextLength)
28
+ ? { limit: { context: contextLength, output: maxTokens } }
29
+ : {}),
30
+ },
31
+ ...(hasOptions && modelWork ? { agentOptions: options } : {}),
32
+ };
33
+ }
@@ -0,0 +1,115 @@
1
+ // This Source Code Form is subject to the terms of the Mozilla Public
2
+ // License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ /**
5
+ * The opencode agent that runs unattended model work under the model-work
6
+ * tool policy (`MODEL_WORK_POLICY_ID`). The CLI builder injects it through
7
+ * `OPENCODE_CONFIG_CONTENT` and selects it with `--agent`; the SDK runner adds
8
+ * it to its server config and names it in the prompt body.
9
+ *
10
+ * What opencode 1.18.25 confines, checked against a local stub:
11
+ * - read, grep and glob work in the session directory and akm's primary
12
+ * stash, and nowhere else. edit works in the session directory only: the
13
+ * stash is denied, by its path without the leading slash, because opencode
14
+ * matches edit patterns root-relative (to the git root, when the session
15
+ * directory is in a repository). Write is part of the edit permission.
16
+ * - `akm_search` and `akm_show` are the akm-opencode plugin's read tools; its
17
+ * other three are denied. bash is denied: opencode matches a bash rule
18
+ * against the command's words only, so `akm show x > ~/stash/asset.md`
19
+ * would pass an `akm show *` rule and write anywhere.
20
+ * - every other tool is denied, `doom_loop` included (its default, `ask`,
21
+ * would hang a headless server). Each permission opencode knows is named,
22
+ * so a same-named agent in the user's config cannot re-allow one through
23
+ * opencode's config merge, and `*` covers the rest.
24
+ * The same rules also go in the top-level `permission`, so an opencode that
25
+ * falls back to its default agent is confined too.
26
+ *
27
+ * The agent also keeps opencode's coding-assistant defaults out of model work:
28
+ * - its own short `prompt` replaces the provider's coding prompt, which
29
+ * tells the model to search extensively; opencode appends a request's
30
+ * system text after it and never substitutes it;
31
+ * - it sets no `steps`. At its step limit opencode sends a "maximum steps"
32
+ * text as a trailing assistant message, which a qwen chat template (LM
33
+ * Studio, llama-server) renders as the start of the model's reply: LM
34
+ * Studio then returns nothing and llama-server returns that text as the
35
+ * answer. The dispatch timeout bounds a run instead;
36
+ * - automatic compaction is off, so a long run cannot summarize the task
37
+ * away.
38
+ */
39
+ import path from "node:path";
40
+ import { resolveStashDir } from "../../../core/common.js";
41
+ import { getStateDir } from "../../../core/paths.js";
42
+ export const MODEL_WORK_OPENCODE_AGENT = "akm-model-work";
43
+ const MODEL_WORK_PROMPT = "You do one bounded task for akm. Use tools only to check what the task needs, never repeat a tool call, and reply with exactly what the task asks for.";
44
+ function modelWorkPermission(stash) {
45
+ return {
46
+ "*": "deny",
47
+ read: "allow",
48
+ grep: "allow",
49
+ glob: "allow",
50
+ edit: { "*": "allow", ...(stash ? { [`${stash.slice(1)}/*`]: "deny" } : {}) },
51
+ external_directory: {
52
+ ...(stash ? { [`${stash}/*`]: "allow" } : {}),
53
+ "~/.local/share/opencode/tool-output/*": "deny",
54
+ },
55
+ akm_search: "allow",
56
+ akm_show: "allow",
57
+ akm_feedback: "deny",
58
+ akm_remember: "deny",
59
+ akm_curate: "deny",
60
+ bash: "deny",
61
+ doom_loop: "deny",
62
+ list: "deny",
63
+ lsp: "deny",
64
+ question: "deny",
65
+ skill: "deny",
66
+ task: "deny",
67
+ todowrite: "deny",
68
+ webfetch: "deny",
69
+ websearch: "deny",
70
+ };
71
+ }
72
+ /**
73
+ * What an opencode model-work dispatch gives the akm-opencode plugin (it comes from the user's opencode config and gives
74
+ * the model `akm_search` and `akm_show`): its own switches off, and its state in akm's state directory, the same for every
75
+ * dispatch because an SDK server outlives its dispatch. win32 has no /bin/true, so the plugin keeps its CLI there.
76
+ */
77
+ export function modelWorkPluginEnv() {
78
+ return {
79
+ AKM_AUTO_CURATE: "0",
80
+ AKM_AUTO_LEARNING: "0",
81
+ AKM_AUTO_SKILL_PROPOSALS: "0",
82
+ AKM_WRITE_GATE: "off",
83
+ XDG_STATE_HOME: path.join(getStateDir(), "opencode-model-work"),
84
+ ...(process.platform === "win32" ? {} : { AKM_OPENCODE_CLI: "/bin/true" }),
85
+ };
86
+ }
87
+ /**
88
+ * The opencode config fragment that defines and confines the model-work agent.
89
+ * The agent carries the request's inference options (`model-config.ts`), which
90
+ * apply to its calls only: opencode's own calls on the same model, a title for
91
+ * the session, keep the model's defaults.
92
+ */
93
+ export function modelWorkOpencodeConfig(options) {
94
+ let stash;
95
+ try {
96
+ stash = resolveStashDir();
97
+ }
98
+ catch {
99
+ // akm has no stash here, so the agent is given no path into one.
100
+ }
101
+ const permission = modelWorkPermission(stash);
102
+ return {
103
+ permission: { ...permission },
104
+ compaction: { auto: false },
105
+ agent: {
106
+ [MODEL_WORK_OPENCODE_AGENT]: {
107
+ mode: "primary",
108
+ description: "akm unattended model work: read and edit inside its working directory only.",
109
+ prompt: MODEL_WORK_PROMPT,
110
+ ...(options ? { options } : {}),
111
+ permission: { ...permission },
112
+ },
113
+ },
114
+ };
115
+ }
@@ -22,6 +22,7 @@
22
22
  * breaks the cycle.
23
23
  */
24
24
  import { createAgentRequestLowerer } from "../../agent/request-lowering.js";
25
+ import { MODEL_WORK_AGENT_INFERENCE } from "../opencode/model-config.js";
25
26
  import { caps } from "../shared.js";
26
27
  import { BaseHarness } from "../types.js";
27
28
  /**
@@ -33,12 +34,6 @@ export class OpencodeSdkHarness extends BaseHarness {
33
34
  id = "opencode-sdk";
34
35
  displayName = "OpenCode SDK";
35
36
  // ── Workflow-engine descriptor (plan §"Capability matrix", P2) ────────────
36
- // Embedded-SDK dispatch on this machine ⇒ local-runner (the matrix's
37
- // "local (sdk/cli)" row, SDK half).
38
- pattern = "local-runner";
39
- // `session.prompt` returns structured SDK events/messages; akm extracts the
40
- // final message then validates against the node schema ⇒ native-json tier.
41
- structuredOutput = "native-json";
42
37
  executionLowerer = {
43
38
  platform: "opencode-sdk",
44
39
  personaChannel: "native",
@@ -47,7 +42,7 @@ export class OpencodeSdkHarness extends BaseHarness {
47
42
  personaChannel: "native",
48
43
  nativeAgentSelector: true,
49
44
  tools: "sdk",
50
- outputSchema: false,
45
+ inference: MODEL_WORK_AGENT_INFERENCE,
51
46
  }),
52
47
  };
53
48
  // No flag-shaped resume: session reuse is programmatic — the SDK session id is
@@ -92,8 +92,11 @@
92
92
  */
93
93
  import { spawn } from "node:child_process";
94
94
  import { createHash } from "node:crypto";
95
- import { COMMON_SPAWN_ENV_PASSTHROUGH, spawnEnvNamesFor } from "../../../core/spawn-env.js";
95
+ import { isRecord } from "../../../core/common.js";
96
+ import { COMMON_SPAWN_ENV_PASSTHROUGH, spawnEnvNamesFor, XDG_BASE_DIR_ENV_PASSTHROUGH } from "../../../core/spawn-env.js";
96
97
  import { DEFAULT_AGENT_TIMEOUT_MS } from "../../agent/config.js";
98
+ import { opencodeInferenceConfig } from "../opencode/model-config.js";
99
+ import { MODEL_WORK_OPENCODE_AGENT, modelWorkOpencodeConfig, modelWorkPluginEnv } from "../opencode/model-work-agent.js";
97
100
  // Server registry — one server per complete server-material signature. Caller
98
101
  // deadlines race the shared promise independently; they never become startup
99
102
  // configuration inherited by later callers.
@@ -232,21 +235,45 @@ function toolsToSdkAllowlist(tools) {
232
235
  out[n] = true;
233
236
  return out;
234
237
  }
238
+ /** The inference a fallback LLM connection carries, which opencode can take. */
239
+ function fallbackInference(llmConfig) {
240
+ const out = {};
241
+ for (const key of ["temperature", "maxTokens", "contextLength", "enableThinking", "reasoningEffort"]) {
242
+ if (llmConfig?.[key] !== undefined)
243
+ out[key] = llmConfig[key];
244
+ }
245
+ return out;
246
+ }
235
247
  /**
236
248
  * Assemble the OpenCode SDK server config from the profile + LLM fallback.
237
249
  * Pure and exported for tests. `profile.model` is already exact because model
238
- * aliases resolve once before harness lowering.
250
+ * aliases resolve once before harness lowering. A server for model work also
251
+ * defines the confined model-work agent (`../opencode/model-work-agent`).
252
+ *
253
+ * The model the config routes through `akm-custom` is declared in full, with
254
+ * the dispatch's inference (`../opencode/model-config`): the fallback LLM
255
+ * engine's, under `requestInference`, the request's own. For model work the
256
+ * options go on the confined agent instead of the model, which needs no model
257
+ * named: it runs whichever model opencode picks. A model the user's own opencode
258
+ * config provides carries none: set inference there.
239
259
  */
240
- export function buildSdkConfig(profile, llmConfig) {
260
+ export function buildSdkConfig(profile, llmConfig, modelWork = false, requestInference) {
241
261
  const endpoint = llmConfig?.endpoint;
242
262
  const apiKey = llmConfig?.apiKey;
243
263
  const profileModel = profile.model;
244
264
  const model = profileModel ?? llmConfig?.model;
265
+ const inference = requestInference === null ? undefined : { ...fallbackInference(llmConfig), ...requestInference };
266
+ const { entry, agentOptions } = opencodeInferenceConfig(inference, modelWork);
245
267
  const sdkConfig = {};
246
268
  if (model)
247
269
  sdkConfig.model = model;
248
270
  if (endpoint || apiKey) {
249
- // Configure a custom OpenAI-compatible provider
271
+ // The first path segment selects the OpenCode provider. Model IDs may
272
+ // themselves contain slashes, but still belong to this custom endpoint.
273
+ const modelId = model?.startsWith("akm-custom/") ? model.slice("akm-custom/".length) : model;
274
+ // Configure a custom OpenAI-compatible provider. OpenCode registers only
275
+ // the models a custom provider lists, so the routed model is declared
276
+ // (#1015: without it every dispatch failed with ProviderModelNotFoundError).
250
277
  sdkConfig.provider = {
251
278
  "akm-custom": {
252
279
  npm: "@ai-sdk/openai-compatible",
@@ -254,14 +281,13 @@ export function buildSdkConfig(profile, llmConfig) {
254
281
  baseURL: canonicalProviderBase(endpoint) ?? undefined,
255
282
  ...(apiKey ? { apiKey } : {}),
256
283
  },
284
+ ...(modelId ? { models: { [modelId]: entry } } : {}),
257
285
  },
258
286
  };
259
- // The first path segment selects the OpenCode provider. Model IDs may
260
- // themselves contain slashes, but still belong to this custom endpoint.
261
- if (model)
262
- sdkConfig.model = model.startsWith("akm-custom/") ? model : `akm-custom/${model}`;
287
+ if (modelId)
288
+ sdkConfig.model = `akm-custom/${modelId}`;
263
289
  }
264
- return sdkConfig;
290
+ return modelWork ? { ...sdkConfig, ...modelWorkOpencodeConfig(agentOptions) } : sdkConfig;
265
291
  }
266
292
  /** Digest the executable and exact environment received by the child. */
267
293
  function serverRegistryKey(profile, env) {
@@ -270,11 +296,23 @@ function serverRegistryKey(profile, env) {
270
296
  .update(JSON.stringify(canonicalize(material)))
271
297
  .digest("hex");
272
298
  }
273
- /** @internal Exact environment allowlist used to start the OpenCode SDK server. */
299
+ /**
300
+ * @internal Exact environment allowlist used to start the OpenCode SDK server:
301
+ * the common baseline, the XDG base-directory variables opencode resolves its
302
+ * config, data, cache and state from, and the profile's own names. The XDG names
303
+ * are the server's, not the profile's: profile `envPassthrough` is frozen into
304
+ * workflow plans, and an SDK profile's list has always been empty.
305
+ */
274
306
  export function opencodeSdkServerEnvironmentNames(profile) {
275
- return [...new Set([...spawnEnvNamesFor(COMMON_SPAWN_ENV_PASSTHROUGH), ...(profile.envPassthrough ?? [])])];
307
+ return [
308
+ ...new Set([
309
+ ...spawnEnvNamesFor(COMMON_SPAWN_ENV_PASSTHROUGH),
310
+ ...XDG_BASE_DIR_ENV_PASSTHROUGH,
311
+ ...(profile.envPassthrough ?? []),
312
+ ]),
313
+ ];
276
314
  }
277
- function buildServerEnv(profile, config, bindings, envSource) {
315
+ function buildServerEnv(profile, config, bindings, envSource, modelWork) {
278
316
  const env = {};
279
317
  for (const key of opencodeSdkServerEnvironmentNames(profile)) {
280
318
  const value = envSource[key];
@@ -283,6 +321,8 @@ function buildServerEnv(profile, config, bindings, envSource) {
283
321
  }
284
322
  for (const [key, value] of Object.entries(bindings ?? {}))
285
323
  env[key] = value;
324
+ if (modelWork)
325
+ Object.assign(env, modelWorkPluginEnv());
286
326
  env.OPENCODE_CONFIG_CONTENT = JSON.stringify(config);
287
327
  return env;
288
328
  }
@@ -550,11 +590,11 @@ async function startServer(profile, sdkConfig, env, registryKey, startupSignal)
550
590
  * start (the registry stores the in-flight promise). A failed start is
551
591
  * evicted so the next call can retry instead of caching the error forever.
552
592
  */
553
- function getOrStartServer(profile, llmConfig, env, envSource = process.env) {
593
+ function getOrStartServer(profile, llmConfig, env, envSource = process.env, modelWork = false, inference) {
554
594
  if (_testServer)
555
595
  return { promise: Promise.resolve(_testServer), release() { } };
556
- const sdkConfig = buildSdkConfig(profile, llmConfig);
557
- const serverEnv = buildServerEnv(profile, sdkConfig, env, envSource);
596
+ const sdkConfig = buildSdkConfig(profile, llmConfig, modelWork, inference);
597
+ const serverEnv = buildServerEnv(profile, sdkConfig, env, envSource, modelWork);
558
598
  const key = serverRegistryKey(profile, serverEnv);
559
599
  let entry = _servers.get(key);
560
600
  if (!entry) {
@@ -665,6 +705,25 @@ function errorText(err) {
665
705
  function appendStderr(stderr, message) {
666
706
  return stderr ? `${stderr}\n${message}` : message;
667
707
  }
708
+ /**
709
+ * Map an SDK `{ error }` result or a reply's `info.error` onto a failure
710
+ * (#1015). Both are opencode NamedErrors, `{ name, data: { message } }`. An
711
+ * aborted message is `aborted`, a reply cut off at the output limit is
712
+ * `parse_error`, and every other error (auth, API, unknown, an HTTP error
713
+ * body) is `non_zero_exit`.
714
+ */
715
+ function sdkErrorFailure(error) {
716
+ if (!isRecord(error) || typeof error.name !== "string") {
717
+ return { reason: "non_zero_exit", message: typeof error === "string" ? error : JSON.stringify(error) };
718
+ }
719
+ const detail = isRecord(error.data) ? error.data.message : undefined;
720
+ const message = typeof detail === "string" ? `${error.name}: ${detail}` : error.name;
721
+ if (error.name === "MessageAbortedError")
722
+ return { reason: "aborted", message };
723
+ if (error.name === "MessageOutputLengthError")
724
+ return { reason: "parse_error", message };
725
+ return { reason: "non_zero_exit", message };
726
+ }
668
727
  async function deleteSessionBestEffort(client, sessionId, query, setTimeoutFn, clearTimeoutFn) {
669
728
  try {
670
729
  const deleted = await raceSdkOperation(client.session.delete({ path: { id: sessionId }, ...(query ? { query } : {}) }), {
@@ -681,6 +740,10 @@ async function deleteSessionBestEffort(client, sessionId, query, setTimeoutFn, c
681
740
  return `OpenCode session cleanup failed: ${errorText(err)}`;
682
741
  }
683
742
  }
743
+ /** Stop a server-side session that the dispatch has given up on, so it stops calling the model. */
744
+ function abortSessionBestEffort(client, sessionId, query) {
745
+ void client.session.abort?.({ path: { id: sessionId }, ...(query ? { query } : {}) }).catch(() => { });
746
+ }
684
747
  function abortedBeforeSdkStart(profile) {
685
748
  return {
686
749
  ok: false,
@@ -701,12 +764,13 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
701
764
  const clearTimeoutImpl = opts.clearTimeoutFn ?? clearTimeout;
702
765
  if (opts.signal?.aborted)
703
766
  return abortedBeforeSdkStart(profile);
767
+ const modelWork = opts.dispatch?.modelWork === true;
704
768
  let client;
705
769
  if (_testServer) {
706
770
  client = _testServer.client;
707
771
  }
708
772
  else {
709
- const startupHandle = getOrStartServer(profile, llmConfig, opts.env, opts.envSource);
773
+ const startupHandle = getOrStartServer(profile, llmConfig, opts.env, opts.envSource, modelWork, opts.dispatch?.inference);
710
774
  try {
711
775
  const startup = await raceSdkOperation(startupHandle.promise, {
712
776
  timeoutMs: remainingTimeoutMs(),
@@ -830,8 +894,9 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
830
894
  // dispatch request. Both were previously accepted on AgentDispatchRequest but
831
895
  // silently dropped on the SDK path, so SDK-mode dispatch ignored agent-asset
832
896
  // system prompts and tool policies entirely (the CLI path honours both).
897
+ // Model work runs the confined agent its server config defines; its tools are that agent's.
833
898
  const dispatch = opts.dispatch;
834
- const agent = dispatch?.agent;
899
+ const agent = modelWork ? MODEL_WORK_OPENCODE_AGENT : dispatch?.agent;
835
900
  const system = dispatch?.systemPrompt;
836
901
  const tools = toolsToSdkAllowlist(dispatch?.tools);
837
902
  const body = { parts: [{ type: "text", text: prompt }] };
@@ -852,6 +917,10 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
852
917
  void deleteSessionBestEffort(client, sessionId, query, setTimeoutImpl, clearTimeoutImpl);
853
918
  },
854
919
  });
920
+ // A session the dispatch stops early is aborted on the server too.
921
+ if (prompted === SDK_OPERATION_ABORTED || prompted === SDK_OPERATION_TIMED_OUT) {
922
+ abortSessionBestEffort(client, sessionId, query);
923
+ }
855
924
  if (prompted === SDK_OPERATION_ABORTED) {
856
925
  result = {
857
926
  ok: false,
@@ -878,21 +947,38 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
878
947
  }
879
948
  else {
880
949
  const parts = prompted.data?.parts ?? [];
881
- const textPart = parts.find((p) => p.type === "text");
882
- const stdout = textPart?.text ?? "";
950
+ // The last text part is the answer; earlier ones narrate the steps before it.
951
+ const stdout = parts.filter((p) => p.type === "text").at(-1)?.text ?? "";
883
952
  // Token accounting from the AssistantMessage (previously discarded) —
884
953
  // the seam that makes workflow budget.maxTokens meterable on the
885
954
  // default sdk runner.
886
955
  const usage = extractUsage(prompted.data?.info);
887
- result = {
888
- ok: true,
889
- stdout,
890
- stderr: "",
891
- durationMs: Date.now() - start,
892
- exitCode: 0,
893
- sessionId,
894
- ...(usage ? { usage } : {}),
895
- };
956
+ const sdkError = prompted.error ?? prompted.data?.info?.error;
957
+ if (sdkError) {
958
+ const failure = sdkErrorFailure(sdkError);
959
+ result = {
960
+ ok: false,
961
+ stdout,
962
+ stderr: failure.message,
963
+ durationMs: Date.now() - start,
964
+ exitCode: failure.reason === "aborted" ? null : 1,
965
+ reason: failure.reason,
966
+ error: failure.message,
967
+ sessionId,
968
+ ...(usage ? { usage } : {}),
969
+ };
970
+ }
971
+ else {
972
+ result = {
973
+ ok: true,
974
+ stdout,
975
+ stderr: "",
976
+ durationMs: Date.now() - start,
977
+ exitCode: 0,
978
+ sessionId,
979
+ ...(usage ? { usage } : {}),
980
+ };
981
+ }
896
982
  }
897
983
  }
898
984
  catch (err) {
@@ -44,10 +44,8 @@
44
44
  * - **schema** — the matrix places OpenHands in the "via prompt+validate"
45
45
  * tier (plan §"Structured-output normalization", tier "native-json"): no
46
46
  * Codex-style `--output-schema` flag exists, so NO temp schema file is
47
- * written; the JSON Schema is injected into the task payload using the
48
- * exact directive wording of the engine's prompt assembly
49
- * (`step-work.ts` `buildUnitPrompt`) and the pi/aider builders, so
50
- * all dispatch paths speak one dialect. Downstream, the extractor pulls the
47
+ * written; the JSON Schema reaches it as the instruction the shared request
48
+ * lowering appends to the prompt. Downstream, the extractor pulls the
51
49
  * final message out of the JSONL stream and the engine's shared
52
50
  * retry-until-valid loop performs the actual validation.
53
51
  * - **tools** — deliberately unconsumed. OpenHands has no per-tool allowlist
@@ -60,16 +58,14 @@
60
58
  * conversation/session id opportunistically when the stream reveals one;
61
59
  * akm's `workflow_run_units` remains the durable source of truth either way
62
60
  * (plan §"Session, MCP, and identity across harnesses").
63
- * - **effort** — stays unconsumed (reserved; the shared request contract's
64
- * "no builder consumes it yet" note stays true).
61
+ * - **inference** — not translated: the shared lowering reports each field of
62
+ * the request's inference as untranslated.
65
63
  *
66
64
  * Registered: `openhandsBuilder` is `OpenhandsHarness.agentBuilder`
67
65
  * (`./index.ts`), one of the ten harnesses `HARNESS_REGISTRY` constructs
68
66
  * (`harnesses/index.ts`); `agent/builders.ts` derives `BUILTIN_BUILDERS` from
69
67
  * that registry, so this builder is reachable under the `"openhands"`
70
- * platform name without any further wiring. The registry entry also declares
71
- * `pattern: "local-runner"`, `structuredOutput: "native-json"` alongside it
72
- * (`./index.ts`).
68
+ * platform name without any further wiring.
73
69
  */
74
70
  import { resolveDispatchModel } from "../../agent/builder-shared.js";
75
71
  import { createAgentRequestLowerer } from "../../agent/request-lowering.js";
@@ -82,27 +78,18 @@ export const OPENHANDS_PLATFORM = "openhands";
82
78
  * share the one constant.
83
79
  */
84
80
  export const OPENHANDS_MODEL_ENV = "LLM_MODEL";
85
- /**
86
- * Assemble the `--task` payload: optional system text, the task prompt, and —
87
- * when a schema is requested — the same schema directive the workflow
88
- * engine's prompt assembly uses (OpenHands has no native schema flag, so the
89
- * prompt is the schema's only channel; plan §"Structured-output
90
- * normalization").
91
- */
81
+ /** Assemble the `--task` payload: optional system text, then the task prompt. */
92
82
  function buildTaskPayload(req) {
93
83
  const sections = [];
94
84
  if (req.systemPrompt)
95
85
  sections.push(req.systemPrompt);
96
86
  sections.push(req.prompt);
97
- if (req.schema) {
98
- sections.push(`Respond with ONLY a JSON value matching this JSON Schema (no prose, no code fences):\n${JSON.stringify(req.schema)}`);
99
- }
100
87
  return sections.join("\n\n");
101
88
  }
102
89
  /**
103
90
  * OpenHands builder.
104
91
  * Command shape:
105
- * openhands --headless --json --task=<[system\n\n]prompt[\n\nschema directive]>
92
+ * openhands --headless --json --task=<[system\n\n]prompt>
106
93
  * with the resolved model (if any) carried on env as LLM_MODEL.
107
94
  */
108
95
  export const openhandsBuilder = {
@@ -112,7 +99,6 @@ export const openhandsBuilder = {
112
99
  adapter: OPENHANDS_PLATFORM,
113
100
  personaChannel: "prompt",
114
101
  tools: "none",
115
- outputSchema: true,
116
102
  }),
117
103
  build(profile, req) {
118
104
  const args = [...profile.args];
@@ -29,11 +29,6 @@ export class OpenhandsHarness extends BaseHarness {
29
29
  agentBuilder = openhandsBuilder;
30
30
  resultExtractor = openhandsResultExtractor;
31
31
  // ── Workflow-engine descriptor (plan §"Capability matrix", P2) ────────────
32
- // akm spawns `openhands --headless` locally per unit ⇒ local-runner.
33
- pattern = "local-runner";
34
- // `--json` emits a documented JSONL event stream akm parses, then validates
35
- // against the node schema ⇒ native-json tier.
36
- structuredOutput = "native-json";
37
32
  // No flag-shaped resume: per the matrix OpenHands resumes from workspace state, not a
38
33
  // session-id flag. The extractor still captures a conversation id
39
34
  // opportunistically; akm's `workflow_run_units` remains the durable source
@@ -27,10 +27,9 @@
27
27
  * - **systemPrompt** — passed via `--system-prompt` (Pi follows the Claude
28
28
  * Code flag conventions).
29
29
  * - **schema** — the matrix places Pi in the "via prompt+validate" tier (no
30
- * native `--output-schema` equivalent, unlike Codex), so the JSON Schema is
31
- * passed through the prompt: a directive matching the engine's wording
32
- * (`step-work.ts` `buildUnitPrompt`) is appended to the prompt
33
- * payload, and `--mode json` is emitted so stdout is the documented JSONL
30
+ * native `--output-schema` equivalent, unlike Codex), so the JSON Schema
31
+ * reaches it as the instruction the shared request lowering appends to the
32
+ * prompt, and `--mode json` is emitted so stdout is the documented JSONL
34
33
  * event stream that `./result-extractor.ts` normalizes. The engine's shared
35
34
  * retry-until-valid loop performs the actual validation. Without a schema
36
35
  * the argv matches the matrix's bare headless shape (`pi -p "<p>"`) and the
@@ -40,8 +39,8 @@
40
39
  * there is no documented per-tool allowlist flag, and inventing one would
41
40
  * produce a silently broken command. A restrictive policy is therefore
42
41
  * dropped rather than approximated — never silently widened.
43
- * - **effort** — stays unconsumed (reserved; the shared request contract's
44
- * "no builder consumes it yet" note stays true).
42
+ * - **inference** — not translated: the shared lowering reports each field of
43
+ * the request's inference as untranslated.
45
44
  *
46
45
  * Registered: `piBuilder` is `PiHarness.agentBuilder` (`./index.ts`), one of
47
46
  * the ten harnesses `HARNESS_REGISTRY` constructs (`harnesses/index.ts`);
@@ -54,22 +53,10 @@ import { resolveDispatchModel } from "../../agent/builder-shared.js";
54
53
  import { createAgentRequestLowerer } from "../../agent/request-lowering.js";
55
54
  /** Canonical harness/platform id used for model-alias resolution. */
56
55
  export const PI_PLATFORM = "pi";
57
- /**
58
- * Assemble the positional prompt payload: the task prompt and — when a schema
59
- * is requested — the same schema directive the workflow engine's prompt
60
- * assembly uses, so both dispatch paths speak one dialect.
61
- */
62
- function buildPromptPayload(req) {
63
- const sections = [req.prompt];
64
- if (req.schema) {
65
- sections.push(`Respond with ONLY a JSON value matching this JSON Schema (no prose, no code fences):\n${JSON.stringify(req.schema)}`);
66
- }
67
- return sections.join("\n\n");
68
- }
69
56
  /**
70
57
  * Pi builder.
71
58
  * Command shape:
72
- * pi [--system-prompt "..."] [--model <m>] [--mode json] -p -- "<prompt[\n\nschema directive]>"
59
+ * pi [--system-prompt "..."] [--model <m>] [--mode json] -p -- "<prompt>"
73
60
  */
74
61
  export const piBuilder = {
75
62
  platform: PI_PLATFORM,
@@ -78,7 +65,6 @@ export const piBuilder = {
78
65
  adapter: PI_PLATFORM,
79
66
  personaChannel: "native",
80
67
  tools: "none",
81
- outputSchema: true,
82
68
  }),
83
69
  build(profile, req) {
84
70
  const args = [...profile.args];
@@ -97,7 +83,7 @@ export const piBuilder = {
97
83
  // -p = non-interactive print mode; prompt is the trailing positional.
98
84
  args.push("-p");
99
85
  args.push("--");
100
- args.push(buildPromptPayload(req));
86
+ args.push(req.prompt);
101
87
  return { argv: [profile.bin, ...args] };
102
88
  },
103
89
  };
@@ -29,11 +29,6 @@ export class PiHarness extends BaseHarness {
29
29
  agentBuilder = piBuilder;
30
30
  resultExtractor = piResultExtractor;
31
31
  // ── Workflow-engine descriptor (plan §"Capability matrix", P2) ────────────
32
- // akm spawns the `pi` CLI locally per unit ⇒ local-runner.
33
- pattern = "local-runner";
34
- // `--mode json` emits a documented JSONL event stream akm parses, then
35
- // validates against the node schema ⇒ native-json tier.
36
- structuredOutput = "native-json";
37
32
  // Session-id env marker only — the matrix's bare PI_* presence vars must
38
33
  // not stamp identity onto manual runs (see `AkmHarness.identityEnv`).
39
34
  identityEnv = ["PI_SESSION_ID"];
@@ -412,6 +412,11 @@ async function chatCompletionAttemptOnce(config, messages, options, timeoutMs, i
412
412
  catch {
413
413
  throw new LlmCallError(`LLM response was not valid JSON ${config.endpoint}: ${redactSensitiveText(redactErrorBody(rawOkBody), resolvedKey ? [resolvedKey] : [])}`, "parse_error", response.status);
414
414
  }
415
+ // A 2xx can still carry a provider failure. OpenRouter answers a provider
416
+ // that fails after the headers with a body holding only `error`.
417
+ if (json.error !== undefined && json.error !== null && !json.choices?.length) {
418
+ throw new LlmCallError(`LLM provider error (${response.status}) ${config.endpoint}: ${redactSensitiveText(redactErrorBody(rawOkBody), resolvedKey ? [resolvedKey] : [])}`, "provider_error", response.status);
419
+ }
415
420
  const responseModel = typeof json.model === "string" && json.model.trim().length > 0 ? json.model : undefined;
416
421
  terminalFields = {
417
422
  model: responseModel ?? config.model,