akm-cli 0.9.24 → 0.9.25-alpha.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/CHANGELOG.md +294 -0
  2. package/dist/commands/health/checks.js +10 -11
  3. package/dist/commands/improve/consolidate/pair-pass.js +3 -2
  4. package/dist/commands/improve/consolidate.js +3 -2
  5. package/dist/commands/improve/execution.js +4 -10
  6. package/dist/commands/improve/extract-prompt.js +4 -4
  7. package/dist/commands/improve/extract.js +10 -13
  8. package/dist/commands/improve/improve-cli.js +32 -33
  9. package/dist/commands/improve/improve-strategies.js +49 -43
  10. package/dist/commands/improve/improve-usage-report.js +8 -17
  11. package/dist/commands/improve/preparation.js +3 -1
  12. package/dist/commands/improve/reflect.js +108 -172
  13. package/dist/commands/improve/stage.js +69 -18
  14. package/dist/commands/proposal/drain.js +14 -2
  15. package/dist/commands/proposal/proposal-cli.js +1 -5
  16. package/dist/commands/proposal/propose-cli.js +2 -2
  17. package/dist/commands/proposal/propose.js +80 -83
  18. package/dist/commands/remember.js +3 -3
  19. package/dist/commands/sources/schema-repair.js +1 -1
  20. package/dist/core/config/config-schema.js +37 -60
  21. package/dist/core/config/engine-semantics.js +15 -11
  22. package/dist/core/config/schema/engines.js +33 -15
  23. package/dist/core/config/schema/improve-processes.js +2 -2
  24. package/dist/core/improve-result.js +3 -3
  25. package/dist/core/structured.js +10 -0
  26. package/dist/execution/source.js +14 -0
  27. package/dist/indexer/passes/memory-inference.js +2 -1
  28. package/dist/integrations/agent/builder-shared.js +15 -0
  29. package/dist/integrations/agent/config.js +2 -0
  30. package/dist/integrations/agent/engine-resolution.js +16 -31
  31. package/dist/integrations/agent/execution.js +46 -21
  32. package/dist/integrations/agent/index.js +1 -1
  33. package/dist/integrations/agent/model-map.js +16 -15
  34. package/dist/integrations/agent/prompts.js +53 -76
  35. package/dist/integrations/agent/request-lowering.js +20 -9
  36. package/dist/integrations/agent/runner-dispatch.js +103 -4
  37. package/dist/integrations/agent/runner.js +8 -2
  38. package/dist/integrations/harnesses/aider/agent-builder.js +11 -23
  39. package/dist/integrations/harnesses/aider/index.js +0 -5
  40. package/dist/integrations/harnesses/amazonq/agent-builder.js +10 -21
  41. package/dist/integrations/harnesses/amazonq/index.js +0 -5
  42. package/dist/integrations/harnesses/claude/agent-builder.js +61 -38
  43. package/dist/integrations/harnesses/claude/index.js +0 -14
  44. package/dist/integrations/harnesses/codex/agent-builder.js +4 -4
  45. package/dist/integrations/harnesses/codex/index.js +0 -4
  46. package/dist/integrations/harnesses/copilot/agent-builder.js +4 -13
  47. package/dist/integrations/harnesses/copilot/index.js +2 -7
  48. package/dist/integrations/harnesses/gemini/agent-builder.js +4 -13
  49. package/dist/integrations/harnesses/gemini/index.js +0 -5
  50. package/dist/integrations/harnesses/ids.js +18 -10
  51. package/dist/integrations/harnesses/opencode/agent-builder.js +52 -10
  52. package/dist/integrations/harnesses/opencode/index.js +0 -8
  53. package/dist/integrations/harnesses/opencode/model-config.js +80 -0
  54. package/dist/integrations/harnesses/opencode/model-work-agent.js +80 -0
  55. package/dist/integrations/harnesses/opencode-sdk/harness.js +6 -7
  56. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +168 -24
  57. package/dist/integrations/harnesses/openhands/agent-builder.js +8 -21
  58. package/dist/integrations/harnesses/openhands/index.js +0 -5
  59. package/dist/integrations/harnesses/pi/agent-builder.js +8 -21
  60. package/dist/integrations/harnesses/pi/index.js +0 -5
  61. package/dist/llm/client.js +5 -0
  62. package/dist/llm/feature-gate.js +10 -4
  63. package/dist/llm/index-passes.js +3 -6
  64. package/dist/llm/memory-infer.js +6 -5
  65. package/dist/llm/structured-call.js +33 -11
  66. package/dist/scripts/akm-migrate-node.js +298 -204
  67. package/dist/scripts/akm-migrate.js +298 -204
  68. package/dist/workflows/exec/step-work.js +6 -5
  69. package/dist/workflows/freeze/step-values.js +1 -1
  70. package/docs/reference/cli.md +29 -11
  71. package/docs/reference/configuration.md +168 -11
  72. package/docs/reference/workflow-schema.md +13 -9
  73. package/package.json +1 -1
  74. package/schemas/akm-config.json +36 -0
@@ -0,0 +1,80 @@
1
+ // This Source Code Form is subject to the terms of the Mozilla Public
2
+ // License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ /**
5
+ * The opencode agent that runs unattended model work under the model-work
6
+ * tool policy (`MODEL_WORK_TOOLS`). The CLI builder injects it through
7
+ * `OPENCODE_CONFIG_CONTENT` and selects it with `--agent`; the SDK runner adds
8
+ * it to its server config and names it in the prompt body.
9
+ *
10
+ * What opencode 1.18.25 confines, checked against a local stub:
11
+ * - read and edit stay inside the session directory (`external_directory`
12
+ * is denied). Write is part of opencode's edit permission, so it is
13
+ * confined the same way and cannot be denied on its own.
14
+ * - bash is denied, so `akm search` and `akm show` are not granted: opencode
15
+ * matches a bash rule against the command's words only, so
16
+ * `akm show x > ~/stash/asset.md` would pass an `akm show *` rule and
17
+ * write anywhere.
18
+ * - every other tool is denied, `doom_loop` included (its default, `ask`,
19
+ * would hang a headless server). Each permission opencode knows is named,
20
+ * so a same-named agent in the user's config cannot re-allow one through
21
+ * opencode's config merge, and `*` covers the rest.
22
+ * The same rules also go in the top-level `permission`, so an opencode that
23
+ * falls back to its default agent is confined too.
24
+ *
25
+ * The agent also keeps opencode's coding-assistant defaults out of model work:
26
+ * - its own short `prompt` replaces the provider's coding prompt, which
27
+ * tells the model to search extensively; opencode appends a request's
28
+ * system text after it and never substitutes it;
29
+ * - `steps` bounds the agentic loop. opencode only asks the model to stop
30
+ * at the limit, so the SDK runner also aborts the session a little past
31
+ * it (`MODEL_WORK_STEP_GRACE`); on the CLI the dispatch timeout bounds it;
32
+ * - automatic compaction is off, so a long run cannot summarize the task
33
+ * away.
34
+ */
35
+ export const MODEL_WORK_OPENCODE_AGENT = "akm-model-work";
36
+ /** The agentic iterations a model-work run may take: a judge answers in one, a generator in a few. */
37
+ export const MODEL_WORK_STEPS = 8;
38
+ /** Steps past {@link MODEL_WORK_STEPS} after which the SDK runner aborts the session. */
39
+ export const MODEL_WORK_STEP_GRACE = 2;
40
+ const MODEL_WORK_PROMPT = "You do one bounded task for akm. Use tools only to check what the task needs, never repeat a tool call, and reply with exactly what the task asks for.";
41
+ const MODEL_WORK_PERMISSION = {
42
+ "*": "deny",
43
+ read: "allow",
44
+ edit: "allow",
45
+ external_directory: "deny",
46
+ bash: "deny",
47
+ doom_loop: "deny",
48
+ glob: "deny",
49
+ grep: "deny",
50
+ list: "deny",
51
+ lsp: "deny",
52
+ question: "deny",
53
+ skill: "deny",
54
+ task: "deny",
55
+ todowrite: "deny",
56
+ webfetch: "deny",
57
+ websearch: "deny",
58
+ };
59
+ /**
60
+ * The opencode config fragment that defines and confines the model-work agent.
61
+ * The agent carries the request's inference options (`model-config.ts`), which
62
+ * apply to its calls only: opencode's own calls on the same model, a title for
63
+ * the session, keep the model's defaults.
64
+ */
65
+ export function modelWorkOpencodeConfig(options) {
66
+ return {
67
+ permission: { ...MODEL_WORK_PERMISSION },
68
+ compaction: { auto: false },
69
+ agent: {
70
+ [MODEL_WORK_OPENCODE_AGENT]: {
71
+ mode: "primary",
72
+ description: "akm unattended model work: read and edit inside its working directory only.",
73
+ prompt: MODEL_WORK_PROMPT,
74
+ steps: MODEL_WORK_STEPS,
75
+ ...(options ? { options } : {}),
76
+ permission: { ...MODEL_WORK_PERMISSION },
77
+ },
78
+ },
79
+ };
80
+ }
@@ -21,7 +21,9 @@
21
21
  * Importing the descriptor from this leaf keeps the registry a config-leaf and
22
22
  * breaks the cycle.
23
23
  */
24
+ import { isModelWorkTools } from "../../../execution/source.js";
24
25
  import { createAgentRequestLowerer } from "../../agent/request-lowering.js";
26
+ import { opencodeCarriedKeys } from "../opencode/model-config.js";
25
27
  import { caps } from "../shared.js";
26
28
  import { BaseHarness } from "../types.js";
27
29
  /**
@@ -33,12 +35,6 @@ export class OpencodeSdkHarness extends BaseHarness {
33
35
  id = "opencode-sdk";
34
36
  displayName = "OpenCode SDK";
35
37
  // ── Workflow-engine descriptor (plan §"Capability matrix", P2) ────────────
36
- // Embedded-SDK dispatch on this machine ⇒ local-runner (the matrix's
37
- // "local (sdk/cli)" row, SDK half).
38
- pattern = "local-runner";
39
- // `session.prompt` returns structured SDK events/messages; akm extracts the
40
- // final message then validates against the node schema ⇒ native-json tier.
41
- structuredOutput = "native-json";
42
38
  executionLowerer = {
43
39
  platform: "opencode-sdk",
44
40
  personaChannel: "native",
@@ -47,7 +43,10 @@ export class OpencodeSdkHarness extends BaseHarness {
47
43
  personaChannel: "native",
48
44
  nativeAgentSelector: true,
49
45
  tools: "sdk",
50
- outputSchema: false,
46
+ modelWorkTools: true,
47
+ // The server config gives the routed model its inference entry, so the request must name a model, except
48
+ // that model work's agent carries its options whichever model opencode picks.
49
+ inference: (_profile, request) => opencodeCarriedKeys(Boolean(request.model?.resolved), request.inference, isModelWorkTools(request.tools)),
51
50
  }),
52
51
  };
53
52
  // No flag-shaped resume: session reuse is programmatic — the SDK session id is
@@ -92,8 +92,12 @@
92
92
  */
93
93
  import { spawn } from "node:child_process";
94
94
  import { createHash } from "node:crypto";
95
+ import { isRecord } from "../../../core/common.js";
95
96
  import { COMMON_SPAWN_ENV_PASSTHROUGH, spawnEnvNamesFor } from "../../../core/spawn-env.js";
97
+ import { isModelWorkTools } from "../../../execution/source.js";
96
98
  import { DEFAULT_AGENT_TIMEOUT_MS } from "../../agent/config.js";
99
+ import { opencodeInferenceConfig, opencodeModelConfig } from "../opencode/model-config.js";
100
+ import { MODEL_WORK_OPENCODE_AGENT, MODEL_WORK_STEP_GRACE, MODEL_WORK_STEPS, modelWorkOpencodeConfig, } from "../opencode/model-work-agent.js";
97
101
  // Server registry — one server per complete server-material signature. Caller
98
102
  // deadlines race the shared promise independently; they never become startup
99
103
  // configuration inherited by later callers.
@@ -232,21 +236,45 @@ function toolsToSdkAllowlist(tools) {
232
236
  out[n] = true;
233
237
  return out;
234
238
  }
239
+ /** The inference a fallback LLM connection carries, which opencode can take. */
240
+ function fallbackInference(llmConfig) {
241
+ const out = {};
242
+ for (const key of ["temperature", "maxTokens", "contextLength", "enableThinking", "reasoningEffort"]) {
243
+ if (llmConfig?.[key] !== undefined)
244
+ out[key] = llmConfig[key];
245
+ }
246
+ return out;
247
+ }
235
248
  /**
236
249
  * Assemble the OpenCode SDK server config from the profile + LLM fallback.
237
250
  * Pure and exported for tests. `profile.model` is already exact because model
238
- * aliases resolve once before harness lowering.
251
+ * aliases resolve once before harness lowering. A server for model work also
252
+ * defines the confined model-work agent (`../opencode/model-work-agent`).
253
+ *
254
+ * The routed model carries the dispatch's inference (`../opencode/model-config`):
255
+ * the fallback LLM engine's, under `requestInference`, the request's own. A
256
+ * model the config routes through `akm-custom` is declared in full; any other
257
+ * model's entry merges over the user's own opencode config for it. For model
258
+ * work the options go on the confined agent instead of the model, which needs
259
+ * no model named: it runs whichever model opencode picks.
239
260
  */
240
- export function buildSdkConfig(profile, llmConfig) {
261
+ export function buildSdkConfig(profile, llmConfig, modelWork = false, requestInference) {
241
262
  const endpoint = llmConfig?.endpoint;
242
263
  const apiKey = llmConfig?.apiKey;
243
264
  const profileModel = profile.model;
244
265
  const model = profileModel ?? llmConfig?.model;
266
+ const inference = requestInference === null ? undefined : { ...fallbackInference(llmConfig), ...requestInference };
267
+ const { entry, agentOptions } = opencodeInferenceConfig(inference, modelWork);
245
268
  const sdkConfig = {};
246
269
  if (model)
247
270
  sdkConfig.model = model;
248
271
  if (endpoint || apiKey) {
249
- // Configure a custom OpenAI-compatible provider
272
+ // The first path segment selects the OpenCode provider. Model IDs may
273
+ // themselves contain slashes, but still belong to this custom endpoint.
274
+ const modelId = model?.startsWith("akm-custom/") ? model.slice("akm-custom/".length) : model;
275
+ // Configure a custom OpenAI-compatible provider. OpenCode registers only
276
+ // the models a custom provider lists, so the routed model is declared
277
+ // (#1015: without it every dispatch failed with ProviderModelNotFoundError).
250
278
  sdkConfig.provider = {
251
279
  "akm-custom": {
252
280
  npm: "@ai-sdk/openai-compatible",
@@ -254,14 +282,18 @@ export function buildSdkConfig(profile, llmConfig) {
254
282
  baseURL: canonicalProviderBase(endpoint) ?? undefined,
255
283
  ...(apiKey ? { apiKey } : {}),
256
284
  },
285
+ ...(modelId ? { models: { [modelId]: entry } } : {}),
257
286
  },
258
287
  };
259
- // The first path segment selects the OpenCode provider. Model IDs may
260
- // themselves contain slashes, but still belong to this custom endpoint.
261
- if (model)
262
- sdkConfig.model = model.startsWith("akm-custom/") ? model : `akm-custom/${model}`;
288
+ if (modelId)
289
+ sdkConfig.model = `akm-custom/${modelId}`;
290
+ }
291
+ else {
292
+ const modelConfig = opencodeModelConfig(model, entry);
293
+ if (modelConfig)
294
+ Object.assign(sdkConfig, modelConfig);
263
295
  }
264
- return sdkConfig;
296
+ return modelWork ? { ...sdkConfig, ...modelWorkOpencodeConfig(agentOptions) } : sdkConfig;
265
297
  }
266
298
  /** Digest the executable and exact environment received by the child. */
267
299
  function serverRegistryKey(profile, env) {
@@ -550,10 +582,10 @@ async function startServer(profile, sdkConfig, env, registryKey, startupSignal)
550
582
  * start (the registry stores the in-flight promise). A failed start is
551
583
  * evicted so the next call can retry instead of caching the error forever.
552
584
  */
553
- function getOrStartServer(profile, llmConfig, env, envSource = process.env) {
585
+ function getOrStartServer(profile, llmConfig, env, envSource = process.env, modelWork = false, inference) {
554
586
  if (_testServer)
555
587
  return { promise: Promise.resolve(_testServer), release() { } };
556
- const sdkConfig = buildSdkConfig(profile, llmConfig);
588
+ const sdkConfig = buildSdkConfig(profile, llmConfig, modelWork, inference);
557
589
  const serverEnv = buildServerEnv(profile, sdkConfig, env, envSource);
558
590
  const key = serverRegistryKey(profile, serverEnv);
559
591
  let entry = _servers.get(key);
@@ -665,6 +697,25 @@ function errorText(err) {
665
697
  function appendStderr(stderr, message) {
666
698
  return stderr ? `${stderr}\n${message}` : message;
667
699
  }
700
+ /**
701
+ * Map an SDK `{ error }` result or a reply's `info.error` onto a failure
702
+ * (#1015). Both are opencode NamedErrors, `{ name, data: { message } }`. An
703
+ * aborted message is `aborted`, a reply cut off at the output limit is
704
+ * `parse_error`, and every other error (auth, API, unknown, an HTTP error
705
+ * body) is `non_zero_exit`.
706
+ */
707
+ function sdkErrorFailure(error) {
708
+ if (!isRecord(error) || typeof error.name !== "string") {
709
+ return { reason: "non_zero_exit", message: typeof error === "string" ? error : JSON.stringify(error) };
710
+ }
711
+ const detail = isRecord(error.data) ? error.data.message : undefined;
712
+ const message = typeof detail === "string" ? `${error.name}: ${detail}` : error.name;
713
+ if (error.name === "MessageAbortedError")
714
+ return { reason: "aborted", message };
715
+ if (error.name === "MessageOutputLengthError")
716
+ return { reason: "parse_error", message };
717
+ return { reason: "non_zero_exit", message };
718
+ }
668
719
  async function deleteSessionBestEffort(client, sessionId, query, setTimeoutFn, clearTimeoutFn) {
669
720
  try {
670
721
  const deleted = await raceSdkOperation(client.session.delete({ path: { id: sessionId }, ...(query ? { query } : {}) }), {
@@ -681,6 +732,55 @@ async function deleteSessionBestEffort(client, sessionId, query, setTimeoutFn, c
681
732
  return `OpenCode session cleanup failed: ${errorText(err)}`;
682
733
  }
683
734
  }
735
+ /** Stop a server-side session that the dispatch has given up on, so it stops calling the model. */
736
+ function abortSessionBestEffort(client, sessionId, query) {
737
+ void client.session.abort?.({ path: { id: sessionId }, ...(query ? { query } : {}) }).catch(() => { });
738
+ }
739
+ /** How often a model-work session's messages are polled for its step count. */
740
+ const MODEL_WORK_STEP_POLL_MS = 1_000;
741
+ /**
742
+ * Count a model-work session's steps, its `step-start` parts, once a second
743
+ * and abort the session once they pass `MODEL_WORK_STEPS` +
744
+ * `MODEL_WORK_STEP_GRACE`: opencode 1.18.25 only asks the model to stop at the
745
+ * agent's `steps`. Polled rather than read from the server's event stream,
746
+ * because the SDK's stream cannot be closed mid-read without an unhandled
747
+ * AbortError (its abort handler drops the promise `reader.cancel()` returns),
748
+ * which the CLI turns into a crash. Best effort: the dispatch timeout still
749
+ * bounds the session.
750
+ */
751
+ function watchModelWorkSteps(client, sessionId, query, timers) {
752
+ let steps = 0;
753
+ let timer;
754
+ let stopped = false;
755
+ const poll = async () => {
756
+ const listed = await client.session.messages?.({ path: { id: sessionId }, ...(query ? { query } : {}) });
757
+ if (stopped)
758
+ return;
759
+ steps = (listed?.data ?? [])
760
+ .flatMap((message) => message.parts ?? [])
761
+ .filter((p) => p.type === "step-start").length;
762
+ if (steps > MODEL_WORK_STEPS + MODEL_WORK_STEP_GRACE)
763
+ abortSessionBestEffort(client, sessionId, query);
764
+ else
765
+ schedule();
766
+ };
767
+ const schedule = () => {
768
+ if (stopped || !client.session.messages)
769
+ return;
770
+ timer = timers.setTimeoutFn(() => void poll().catch(() => schedule()), MODEL_WORK_STEP_POLL_MS);
771
+ if (typeof timer !== "number")
772
+ timer.unref?.();
773
+ };
774
+ schedule();
775
+ return {
776
+ steps: () => steps,
777
+ stop: () => {
778
+ stopped = true;
779
+ if (timer !== undefined)
780
+ timers.clearTimeoutFn(timer);
781
+ },
782
+ };
783
+ }
684
784
  function abortedBeforeSdkStart(profile) {
685
785
  return {
686
786
  ok: false,
@@ -701,12 +801,13 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
701
801
  const clearTimeoutImpl = opts.clearTimeoutFn ?? clearTimeout;
702
802
  if (opts.signal?.aborted)
703
803
  return abortedBeforeSdkStart(profile);
804
+ const modelWork = isModelWorkTools(opts.dispatch?.tools);
704
805
  let client;
705
806
  if (_testServer) {
706
807
  client = _testServer.client;
707
808
  }
708
809
  else {
709
- const startupHandle = getOrStartServer(profile, llmConfig, opts.env, opts.envSource);
810
+ const startupHandle = getOrStartServer(profile, llmConfig, opts.env, opts.envSource, modelWork, opts.dispatch?.inference);
710
811
  try {
711
812
  const startup = await raceSdkOperation(startupHandle.promise, {
712
813
  timeoutMs: remainingTimeoutMs(),
@@ -830,10 +931,11 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
830
931
  // dispatch request. Both were previously accepted on AgentDispatchRequest but
831
932
  // silently dropped on the SDK path, so SDK-mode dispatch ignored agent-asset
832
933
  // system prompts and tool policies entirely (the CLI path honours both).
934
+ // Model work runs the confined agent its server config defines; its tools are that agent's.
833
935
  const dispatch = opts.dispatch;
834
- const agent = dispatch?.agent;
936
+ const agent = modelWork ? MODEL_WORK_OPENCODE_AGENT : dispatch?.agent;
835
937
  const system = dispatch?.systemPrompt;
836
- const tools = toolsToSdkAllowlist(dispatch?.tools);
938
+ const tools = modelWork ? undefined : toolsToSdkAllowlist(dispatch?.tools);
837
939
  const body = { parts: [{ type: "text", text: prompt }] };
838
940
  if (agent)
839
941
  body.agent = agent;
@@ -842,6 +944,9 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
842
944
  if (tools)
843
945
  body.tools = tools;
844
946
  let result;
947
+ const steps = modelWork
948
+ ? watchModelWorkSteps(client, sessionId, query, { setTimeoutFn: setTimeoutImpl, clearTimeoutFn: clearTimeoutImpl })
949
+ : undefined;
845
950
  try {
846
951
  const prompted = await raceSdkOperation(client.session.prompt({ path: { id: sessionId }, body, ...(query ? { query } : {}) }), {
847
952
  timeoutMs: remainingTimeoutMs(),
@@ -852,6 +957,10 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
852
957
  void deleteSessionBestEffort(client, sessionId, query, setTimeoutImpl, clearTimeoutImpl);
853
958
  },
854
959
  });
960
+ // A model-work session the dispatch stops early is aborted on the server too.
961
+ if (modelWork && (prompted === SDK_OPERATION_ABORTED || prompted === SDK_OPERATION_TIMED_OUT)) {
962
+ abortSessionBestEffort(client, sessionId, query);
963
+ }
855
964
  if (prompted === SDK_OPERATION_ABORTED) {
856
965
  result = {
857
966
  ok: false,
@@ -878,21 +987,53 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
878
987
  }
879
988
  else {
880
989
  const parts = prompted.data?.parts ?? [];
881
- const textPart = parts.find((p) => p.type === "text");
882
- const stdout = textPart?.text ?? "";
990
+ // The last text part is the answer; earlier ones narrate the steps before it.
991
+ const stdout = parts.filter((p) => p.type === "text").at(-1)?.text ?? "";
883
992
  // Token accounting from the AssistantMessage (previously discarded) —
884
993
  // the seam that makes workflow budget.maxTokens meterable on the
885
994
  // default sdk runner.
886
995
  const usage = extractUsage(prompted.data?.info);
887
- result = {
888
- ok: true,
889
- stdout,
890
- stderr: "",
891
- durationMs: Date.now() - start,
892
- exitCode: 0,
893
- sessionId,
894
- ...(usage ? { usage } : {}),
895
- };
996
+ const sdkError = prompted.error ?? prompted.data?.info?.error;
997
+ const overSteps = steps !== undefined && steps.steps() > MODEL_WORK_STEPS + MODEL_WORK_STEP_GRACE;
998
+ if (overSteps) {
999
+ const message = `opencode-sdk agent "${profile.name}" ran past the ${MODEL_WORK_STEPS}-step limit for model work; akm aborted the session.`;
1000
+ result = {
1001
+ ok: false,
1002
+ stdout,
1003
+ stderr: message,
1004
+ durationMs: Date.now() - start,
1005
+ exitCode: 1,
1006
+ reason: "parse_error",
1007
+ error: message,
1008
+ sessionId,
1009
+ ...(usage ? { usage } : {}),
1010
+ };
1011
+ }
1012
+ else if (sdkError) {
1013
+ const failure = sdkErrorFailure(sdkError);
1014
+ result = {
1015
+ ok: false,
1016
+ stdout,
1017
+ stderr: failure.message,
1018
+ durationMs: Date.now() - start,
1019
+ exitCode: failure.reason === "aborted" ? null : 1,
1020
+ reason: failure.reason,
1021
+ error: failure.message,
1022
+ sessionId,
1023
+ ...(usage ? { usage } : {}),
1024
+ };
1025
+ }
1026
+ else {
1027
+ result = {
1028
+ ok: true,
1029
+ stdout,
1030
+ stderr: "",
1031
+ durationMs: Date.now() - start,
1032
+ exitCode: 0,
1033
+ sessionId,
1034
+ ...(usage ? { usage } : {}),
1035
+ };
1036
+ }
896
1037
  }
897
1038
  }
898
1039
  catch (err) {
@@ -907,6 +1048,9 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
907
1048
  sessionId,
908
1049
  };
909
1050
  }
1051
+ finally {
1052
+ steps?.stop();
1053
+ }
910
1054
  // Clean up session to prevent disk accumulation in ~/.local/share/opencode/.
911
1055
  // Failures are non-fatal to the agent result but must not be invisible.
912
1056
  const cleanupWarning = await deleteSessionBestEffort(client, sessionId, query, setTimeoutImpl, clearTimeoutImpl);
@@ -44,10 +44,8 @@
44
44
  * - **schema** — the matrix places OpenHands in the "via prompt+validate"
45
45
  * tier (plan §"Structured-output normalization", tier "native-json"): no
46
46
  * Codex-style `--output-schema` flag exists, so NO temp schema file is
47
- * written; the JSON Schema is injected into the task payload using the
48
- * exact directive wording of the engine's prompt assembly
49
- * (`step-work.ts` `buildUnitPrompt`) and the pi/aider builders, so
50
- * all dispatch paths speak one dialect. Downstream, the extractor pulls the
47
+ * written; the JSON Schema reaches it as the instruction the shared request
48
+ * lowering appends to the prompt. Downstream, the extractor pulls the
51
49
  * final message out of the JSONL stream and the engine's shared
52
50
  * retry-until-valid loop performs the actual validation.
53
51
  * - **tools** — deliberately unconsumed. OpenHands has no per-tool allowlist
@@ -60,16 +58,15 @@
60
58
  * conversation/session id opportunistically when the stream reveals one;
61
59
  * akm's `workflow_run_units` remains the durable source of truth either way
62
60
  * (plan §"Session, MCP, and identity across harnesses").
63
- * - **effort** — stays unconsumed (reserved; the shared request contract's
64
- * "no builder consumes it yet" note stays true).
61
+ * - **inference** — not translated: the shared lowering reports each field of
62
+ * the request's inference as untranslated (`inference` in `harnesses/ids.ts`
63
+ * lists none for this harness).
65
64
  *
66
65
  * Registered: `openhandsBuilder` is `OpenhandsHarness.agentBuilder`
67
66
  * (`./index.ts`), one of the ten harnesses `HARNESS_REGISTRY` constructs
68
67
  * (`harnesses/index.ts`); `agent/builders.ts` derives `BUILTIN_BUILDERS` from
69
68
  * that registry, so this builder is reachable under the `"openhands"`
70
- * platform name without any further wiring. The registry entry also declares
71
- * `pattern: "local-runner"`, `structuredOutput: "native-json"` alongside it
72
- * (`./index.ts`).
69
+ * platform name without any further wiring.
73
70
  */
74
71
  import { resolveDispatchModel } from "../../agent/builder-shared.js";
75
72
  import { createAgentRequestLowerer } from "../../agent/request-lowering.js";
@@ -82,27 +79,18 @@ export const OPENHANDS_PLATFORM = "openhands";
82
79
  * share the one constant.
83
80
  */
84
81
  export const OPENHANDS_MODEL_ENV = "LLM_MODEL";
85
- /**
86
- * Assemble the `--task` payload: optional system text, the task prompt, and —
87
- * when a schema is requested — the same schema directive the workflow
88
- * engine's prompt assembly uses (OpenHands has no native schema flag, so the
89
- * prompt is the schema's only channel; plan §"Structured-output
90
- * normalization").
91
- */
82
+ /** Assemble the `--task` payload: optional system text, then the task prompt. */
92
83
  function buildTaskPayload(req) {
93
84
  const sections = [];
94
85
  if (req.systemPrompt)
95
86
  sections.push(req.systemPrompt);
96
87
  sections.push(req.prompt);
97
- if (req.schema) {
98
- sections.push(`Respond with ONLY a JSON value matching this JSON Schema (no prose, no code fences):\n${JSON.stringify(req.schema)}`);
99
- }
100
88
  return sections.join("\n\n");
101
89
  }
102
90
  /**
103
91
  * OpenHands builder.
104
92
  * Command shape:
105
- * openhands --headless --json --task=<[system\n\n]prompt[\n\nschema directive]>
93
+ * openhands --headless --json --task=<[system\n\n]prompt>
106
94
  * with the resolved model (if any) carried on env as LLM_MODEL.
107
95
  */
108
96
  export const openhandsBuilder = {
@@ -112,7 +100,6 @@ export const openhandsBuilder = {
112
100
  adapter: OPENHANDS_PLATFORM,
113
101
  personaChannel: "prompt",
114
102
  tools: "none",
115
- outputSchema: true,
116
103
  }),
117
104
  build(profile, req) {
118
105
  const args = [...profile.args];
@@ -29,11 +29,6 @@ export class OpenhandsHarness extends BaseHarness {
29
29
  agentBuilder = openhandsBuilder;
30
30
  resultExtractor = openhandsResultExtractor;
31
31
  // ── Workflow-engine descriptor (plan §"Capability matrix", P2) ────────────
32
- // akm spawns `openhands --headless` locally per unit ⇒ local-runner.
33
- pattern = "local-runner";
34
- // `--json` emits a documented JSONL event stream akm parses, then validates
35
- // against the node schema ⇒ native-json tier.
36
- structuredOutput = "native-json";
37
32
  // No flag-shaped resume: per the matrix OpenHands resumes from workspace state, not a
38
33
  // session-id flag. The extractor still captures a conversation id
39
34
  // opportunistically; akm's `workflow_run_units` remains the durable source
@@ -27,10 +27,9 @@
27
27
  * - **systemPrompt** — passed via `--system-prompt` (Pi follows the Claude
28
28
  * Code flag conventions).
29
29
  * - **schema** — the matrix places Pi in the "via prompt+validate" tier (no
30
- * native `--output-schema` equivalent, unlike Codex), so the JSON Schema is
31
- * passed through the prompt: a directive matching the engine's wording
32
- * (`step-work.ts` `buildUnitPrompt`) is appended to the prompt
33
- * payload, and `--mode json` is emitted so stdout is the documented JSONL
30
+ * native `--output-schema` equivalent, unlike Codex), so the JSON Schema
31
+ * reaches it as the instruction the shared request lowering appends to the
32
+ * prompt, and `--mode json` is emitted so stdout is the documented JSONL
34
33
  * event stream that `./result-extractor.ts` normalizes. The engine's shared
35
34
  * retry-until-valid loop performs the actual validation. Without a schema
36
35
  * the argv matches the matrix's bare headless shape (`pi -p "<p>"`) and the
@@ -40,8 +39,9 @@
40
39
  * there is no documented per-tool allowlist flag, and inventing one would
41
40
  * produce a silently broken command. A restrictive policy is therefore
42
41
  * dropped rather than approximated — never silently widened.
43
- * - **effort** — stays unconsumed (reserved; the shared request contract's
44
- * "no builder consumes it yet" note stays true).
42
+ * - **inference** — not translated: the shared lowering reports each field of
43
+ * the request's inference as untranslated (`inference` in `harnesses/ids.ts`
44
+ * lists none for this harness).
45
45
  *
46
46
  * Registered: `piBuilder` is `PiHarness.agentBuilder` (`./index.ts`), one of
47
47
  * the ten harnesses `HARNESS_REGISTRY` constructs (`harnesses/index.ts`);
@@ -54,22 +54,10 @@ import { resolveDispatchModel } from "../../agent/builder-shared.js";
54
54
  import { createAgentRequestLowerer } from "../../agent/request-lowering.js";
55
55
  /** Canonical harness/platform id used for model-alias resolution. */
56
56
  export const PI_PLATFORM = "pi";
57
- /**
58
- * Assemble the positional prompt payload: the task prompt and — when a schema
59
- * is requested — the same schema directive the workflow engine's prompt
60
- * assembly uses, so both dispatch paths speak one dialect.
61
- */
62
- function buildPromptPayload(req) {
63
- const sections = [req.prompt];
64
- if (req.schema) {
65
- sections.push(`Respond with ONLY a JSON value matching this JSON Schema (no prose, no code fences):\n${JSON.stringify(req.schema)}`);
66
- }
67
- return sections.join("\n\n");
68
- }
69
57
  /**
70
58
  * Pi builder.
71
59
  * Command shape:
72
- * pi [--system-prompt "..."] [--model <m>] [--mode json] -p -- "<prompt[\n\nschema directive]>"
60
+ * pi [--system-prompt "..."] [--model <m>] [--mode json] -p -- "<prompt>"
73
61
  */
74
62
  export const piBuilder = {
75
63
  platform: PI_PLATFORM,
@@ -78,7 +66,6 @@ export const piBuilder = {
78
66
  adapter: PI_PLATFORM,
79
67
  personaChannel: "native",
80
68
  tools: "none",
81
- outputSchema: true,
82
69
  }),
83
70
  build(profile, req) {
84
71
  const args = [...profile.args];
@@ -97,7 +84,7 @@ export const piBuilder = {
97
84
  // -p = non-interactive print mode; prompt is the trailing positional.
98
85
  args.push("-p");
99
86
  args.push("--");
100
- args.push(buildPromptPayload(req));
87
+ args.push(req.prompt);
101
88
  return { argv: [profile.bin, ...args] };
102
89
  },
103
90
  };
@@ -29,11 +29,6 @@ export class PiHarness extends BaseHarness {
29
29
  agentBuilder = piBuilder;
30
30
  resultExtractor = piResultExtractor;
31
31
  // ── Workflow-engine descriptor (plan §"Capability matrix", P2) ────────────
32
- // akm spawns the `pi` CLI locally per unit ⇒ local-runner.
33
- pattern = "local-runner";
34
- // `--mode json` emits a documented JSONL event stream akm parses, then
35
- // validates against the node schema ⇒ native-json tier.
36
- structuredOutput = "native-json";
37
32
  // Session-id env marker only — the matrix's bare PI_* presence vars must
38
33
  // not stamp identity onto manual runs (see `AkmHarness.identityEnv`).
39
34
  identityEnv = ["PI_SESSION_ID"];
@@ -412,6 +412,11 @@ async function chatCompletionAttemptOnce(config, messages, options, timeoutMs, i
412
412
  catch {
413
413
  throw new LlmCallError(`LLM response was not valid JSON ${config.endpoint}: ${redactSensitiveText(redactErrorBody(rawOkBody), resolvedKey ? [resolvedKey] : [])}`, "parse_error", response.status);
414
414
  }
415
+ // A 2xx can still carry a provider failure. OpenRouter answers a provider
416
+ // that fails after the headers with a body holding only `error`.
417
+ if (json.error !== undefined && json.error !== null && !json.choices?.length) {
418
+ throw new LlmCallError(`LLM provider error (${response.status}) ${config.endpoint}: ${redactSensitiveText(redactErrorBody(rawOkBody), resolvedKey ? [resolvedKey] : [])}`, "provider_error", response.status);
419
+ }
415
420
  const responseModel = typeof json.model === "string" && json.model.trim().length > 0 ? json.model : undefined;
416
421
  terminalFields = {
417
422
  model: responseModel ?? config.model,
@@ -44,7 +44,8 @@ const DEFAULT_TIMEOUT_MS = 600_000;
44
44
  /**
45
45
  * Run `fn()` only if `isLlmFeatureEnabled(config, feature)` is `true`. On
46
46
  * disablement, throw, or timeout, return `fallback` (or — if it is a
47
- * thunk — the value produced by calling it).
47
+ * thunk — the value produced by calling it). The timeout aborts the signal
48
+ * `fn` receives, so the work it started stops too.
48
49
  */
49
50
  export async function tryLlmFeature(feature, config, fn, fallback, opts) {
50
51
  const resolveFallback = async () => typeof fallback === "function" ? await fallback() : fallback;
@@ -55,7 +56,7 @@ export async function tryLlmFeature(feature, config, fn, fallback, opts) {
55
56
  const timeoutMs = opts && Object.hasOwn(opts, "timeoutMs") ? (opts.timeoutMs ?? null) : DEFAULT_TIMEOUT_MS;
56
57
  try {
57
58
  if (timeoutMs === null || timeoutMs <= 0) {
58
- return await fn();
59
+ return await fn(new AbortController().signal);
59
60
  }
60
61
  return await runWithTimeout(fn, timeoutMs, feature);
61
62
  }
@@ -97,11 +98,16 @@ export class LlmFeatureTimeoutError extends Error {
97
98
  }
98
99
  async function runWithTimeout(fn, timeoutMs, feature) {
99
100
  let timer;
101
+ const controller = new AbortController();
100
102
  try {
101
103
  return await new Promise((resolve, reject) => {
102
- timer = setTimeout(() => reject(new LlmFeatureTimeoutError(feature, timeoutMs)), timeoutMs);
104
+ timer = setTimeout(() => {
105
+ const timedOut = new LlmFeatureTimeoutError(feature, timeoutMs);
106
+ controller.abort(timedOut);
107
+ reject(timedOut);
108
+ }, timeoutMs);
103
109
  Promise.resolve()
104
- .then(() => fn())
110
+ .then(() => fn(controller.signal))
105
111
  .then(resolve, reject);
106
112
  });
107
113
  }