akm-cli 0.9.25-alpha.1 → 0.9.25-alpha.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/CHANGELOG.md +243 -280
  2. package/dist/assets/prompts/reflect-feedback-framing.md +1 -1
  3. package/dist/assets/prompts/reflect-llm-framed-contract.md +2 -9
  4. package/dist/assets/prompts/reflect-llm-schema-contract.md +1 -3
  5. package/dist/assets/prompts/reflect-output-repair.md +1 -1
  6. package/dist/cli.js +1 -1
  7. package/dist/commands/improve/consolidate/pair-pass.js +1 -0
  8. package/dist/commands/improve/consolidate.js +7 -2
  9. package/dist/commands/improve/execution.js +2 -3
  10. package/dist/commands/improve/extract-cli.js +3 -2
  11. package/dist/commands/improve/extract.js +2 -1
  12. package/dist/commands/improve/improve-cli.js +33 -1
  13. package/dist/commands/improve/loop-stages.js +3 -0
  14. package/dist/commands/improve/reflect-noise.js +125 -0
  15. package/dist/commands/improve/reflect.js +150 -333
  16. package/dist/commands/improve/retrieval-gate.js +7 -2
  17. package/dist/commands/improve/session-asset.js +6 -0
  18. package/dist/commands/improve/stage.js +31 -39
  19. package/dist/commands/proposal/drain.js +4 -7
  20. package/dist/commands/proposal/propose.js +2 -11
  21. package/dist/commands/proposal/validators/proposal-quality-validators.js +11 -5
  22. package/dist/commands/proposal/validators/proposal-validators.js +4 -5
  23. package/dist/commands/read/search-cli.js +0 -38
  24. package/dist/core/asset/asset-serialize.js +1 -1
  25. package/dist/core/config/schema/engines.js +15 -33
  26. package/dist/core/config/schema/improve-processes.js +16 -0
  27. package/dist/core/content-safety.js +0 -24
  28. package/dist/core/redaction.js +4 -0
  29. package/dist/core/spawn-env.js +25 -0
  30. package/dist/core/structured.js +1 -1
  31. package/dist/execution/source.js +8 -12
  32. package/dist/integrations/agent/config.js +1 -3
  33. package/dist/integrations/agent/engine-resolution.js +0 -3
  34. package/dist/integrations/agent/execution.js +14 -13
  35. package/dist/integrations/agent/index.js +1 -1
  36. package/dist/integrations/agent/model-map.js +15 -16
  37. package/dist/integrations/agent/profiles.js +2 -2
  38. package/dist/integrations/agent/prompts.js +51 -127
  39. package/dist/integrations/agent/request-lowering.js +9 -7
  40. package/dist/integrations/agent/runner-dispatch.js +25 -31
  41. package/dist/integrations/harnesses/aider/agent-builder.js +1 -2
  42. package/dist/integrations/harnesses/amazonq/agent-builder.js +1 -2
  43. package/dist/integrations/harnesses/claude/agent-builder.js +4 -16
  44. package/dist/integrations/harnesses/codex/agent-builder.js +1 -2
  45. package/dist/integrations/harnesses/codex/index.js +6 -11
  46. package/dist/integrations/harnesses/codex/session-log.js +211 -0
  47. package/dist/integrations/harnesses/ids.js +10 -16
  48. package/dist/integrations/harnesses/opencode/agent-builder.js +14 -24
  49. package/dist/integrations/harnesses/opencode/model-config.js +15 -62
  50. package/dist/integrations/harnesses/opencode/model-work-agent.js +71 -36
  51. package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -6
  52. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +32 -90
  53. package/dist/integrations/harnesses/openhands/agent-builder.js +1 -2
  54. package/dist/integrations/harnesses/pi/agent-builder.js +1 -2
  55. package/dist/integrations/harnesses/types.js +3 -3
  56. package/dist/llm/feature-gate.js +2 -5
  57. package/dist/llm/index-passes.js +2 -2
  58. package/dist/llm/structured-call.js +5 -5
  59. package/dist/output/shapes/passthrough.js +1 -0
  60. package/dist/scripts/akm-migrate-node.js +381 -239
  61. package/dist/scripts/akm-migrate.js +381 -239
  62. package/dist/workflows/exec/unit-dispatch.js +4 -13
  63. package/docs/reference/cli.md +29 -16
  64. package/docs/reference/configuration.md +79 -85
  65. package/docs/reference/data-and-telemetry.md +2 -3
  66. package/docs/reference/workflow-schema.md +6 -9
  67. package/package.json +1 -1
  68. package/schemas/akm-config.json +108 -36
@@ -3,18 +3,20 @@
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  /**
5
5
  * The opencode agent that runs unattended model work under the model-work
6
- * tool policy (`MODEL_WORK_TOOLS`). The CLI builder injects it through
6
+ * tool policy (`MODEL_WORK_POLICY_ID`). The CLI builder injects it through
7
7
  * `OPENCODE_CONFIG_CONTENT` and selects it with `--agent`; the SDK runner adds
8
8
  * it to its server config and names it in the prompt body.
9
9
  *
10
10
  * What opencode 1.18.25 confines, checked against a local stub:
11
- * - read and edit stay inside the session directory (`external_directory`
12
- * is denied). Write is part of opencode's edit permission, so it is
13
- * confined the same way and cannot be denied on its own.
14
- * - bash is denied, so `akm search` and `akm show` are not granted: opencode
15
- * matches a bash rule against the command's words only, so
16
- * `akm show x > ~/stash/asset.md` would pass an `akm show *` rule and
17
- * write anywhere.
11
+ * - read, grep and glob work in the session directory and akm's primary
12
+ * stash, and nowhere else. edit works in the session directory only: the
13
+ * stash is denied, by its path without the leading slash, because opencode
14
+ * matches edit patterns root-relative (to the git root, when the session
15
+ * directory is in a repository). Write is part of the edit permission.
16
+ * - `akm_search` and `akm_show` are the akm-opencode plugin's read tools; its
17
+ * other three are denied. bash is denied: opencode matches a bash rule
18
+ * against the command's words only, so `akm show x > ~/stash/asset.md`
19
+ * would pass an `akm show *` rule and write anywhere.
18
20
  * - every other tool is denied, `doom_loop` included (its default, `ask`,
19
21
  * would hang a headless server). Each permission opencode knows is named,
20
22
  * so a same-named agent in the user's config cannot re-allow one through
@@ -26,36 +28,62 @@
26
28
  * - its own short `prompt` replaces the provider's coding prompt, which
27
29
  * tells the model to search extensively; opencode appends a request's
28
30
  * system text after it and never substitutes it;
29
- * - `steps` bounds the agentic loop. opencode only asks the model to stop
30
- * at the limit, so the SDK runner also aborts the session a little past
31
- * it (`MODEL_WORK_STEP_GRACE`); on the CLI the dispatch timeout bounds it;
31
+ * - it sets no `steps`. At its step limit opencode sends a "maximum steps"
32
+ * text as a trailing assistant message, which a qwen chat template (LM
33
+ * Studio, llama-server) renders as the start of the model's reply: LM
34
+ * Studio then returns nothing and llama-server returns that text as the
35
+ * answer. The dispatch timeout bounds a run instead;
32
36
  * - automatic compaction is off, so a long run cannot summarize the task
33
37
  * away.
34
38
  */
39
+ import path from "node:path";
40
+ import { resolveStashDir } from "../../../core/common.js";
41
+ import { getStateDir } from "../../../core/paths.js";
35
42
  export const MODEL_WORK_OPENCODE_AGENT = "akm-model-work";
36
- /** The agentic iterations a model-work run may take: a judge answers in one, a generator in a few. */
37
- export const MODEL_WORK_STEPS = 8;
38
- /** Steps past {@link MODEL_WORK_STEPS} after which the SDK runner aborts the session. */
39
- export const MODEL_WORK_STEP_GRACE = 2;
40
43
  const MODEL_WORK_PROMPT = "You do one bounded task for akm. Use tools only to check what the task needs, never repeat a tool call, and reply with exactly what the task asks for.";
41
- const MODEL_WORK_PERMISSION = {
42
- "*": "deny",
43
- read: "allow",
44
- edit: "allow",
45
- external_directory: "deny",
46
- bash: "deny",
47
- doom_loop: "deny",
48
- glob: "deny",
49
- grep: "deny",
50
- list: "deny",
51
- lsp: "deny",
52
- question: "deny",
53
- skill: "deny",
54
- task: "deny",
55
- todowrite: "deny",
56
- webfetch: "deny",
57
- websearch: "deny",
58
- };
44
+ function modelWorkPermission(stash) {
45
+ return {
46
+ "*": "deny",
47
+ read: "allow",
48
+ grep: "allow",
49
+ glob: "allow",
50
+ edit: { "*": "allow", ...(stash ? { [`${stash.slice(1)}/*`]: "deny" } : {}) },
51
+ external_directory: {
52
+ ...(stash ? { [`${stash}/*`]: "allow" } : {}),
53
+ "~/.local/share/opencode/tool-output/*": "deny",
54
+ },
55
+ akm_search: "allow",
56
+ akm_show: "allow",
57
+ akm_feedback: "deny",
58
+ akm_remember: "deny",
59
+ akm_curate: "deny",
60
+ bash: "deny",
61
+ doom_loop: "deny",
62
+ list: "deny",
63
+ lsp: "deny",
64
+ question: "deny",
65
+ skill: "deny",
66
+ task: "deny",
67
+ todowrite: "deny",
68
+ webfetch: "deny",
69
+ websearch: "deny",
70
+ };
71
+ }
72
+ /**
73
+ * What an opencode model-work dispatch gives the akm-opencode plugin (it comes from the user's opencode config and gives
74
+ * the model `akm_search` and `akm_show`): its own switches off, and its state in akm's state directory, the same for every
75
+ * dispatch because an SDK server outlives its dispatch. win32 has no /bin/true, so the plugin keeps its CLI there.
76
+ */
77
+ export function modelWorkPluginEnv() {
78
+ return {
79
+ AKM_AUTO_CURATE: "0",
80
+ AKM_AUTO_LEARNING: "0",
81
+ AKM_AUTO_SKILL_PROPOSALS: "0",
82
+ AKM_WRITE_GATE: "off",
83
+ XDG_STATE_HOME: path.join(getStateDir(), "opencode-model-work"),
84
+ ...(process.platform === "win32" ? {} : { AKM_OPENCODE_CLI: "/bin/true" }),
85
+ };
86
+ }
59
87
  /**
60
88
  * The opencode config fragment that defines and confines the model-work agent.
61
89
  * The agent carries the request's inference options (`model-config.ts`), which
@@ -63,17 +91,24 @@ const MODEL_WORK_PERMISSION = {
63
91
  * the session, keep the model's defaults.
64
92
  */
65
93
  export function modelWorkOpencodeConfig(options) {
94
+ let stash;
95
+ try {
96
+ stash = resolveStashDir();
97
+ }
98
+ catch {
99
+ // akm has no stash here, so the agent is given no path into one.
100
+ }
101
+ const permission = modelWorkPermission(stash);
66
102
  return {
67
- permission: { ...MODEL_WORK_PERMISSION },
103
+ permission: { ...permission },
68
104
  compaction: { auto: false },
69
105
  agent: {
70
106
  [MODEL_WORK_OPENCODE_AGENT]: {
71
107
  mode: "primary",
72
108
  description: "akm unattended model work: read and edit inside its working directory only.",
73
109
  prompt: MODEL_WORK_PROMPT,
74
- steps: MODEL_WORK_STEPS,
75
110
  ...(options ? { options } : {}),
76
- permission: { ...MODEL_WORK_PERMISSION },
111
+ permission: { ...permission },
77
112
  },
78
113
  },
79
114
  };
@@ -21,9 +21,8 @@
21
21
  * Importing the descriptor from this leaf keeps the registry a config-leaf and
22
22
  * breaks the cycle.
23
23
  */
24
- import { isModelWorkTools } from "../../../execution/source.js";
25
24
  import { createAgentRequestLowerer } from "../../agent/request-lowering.js";
26
- import { opencodeCarriedKeys } from "../opencode/model-config.js";
25
+ import { MODEL_WORK_AGENT_INFERENCE } from "../opencode/model-config.js";
27
26
  import { caps } from "../shared.js";
28
27
  import { BaseHarness } from "../types.js";
29
28
  /**
@@ -43,10 +42,7 @@ export class OpencodeSdkHarness extends BaseHarness {
43
42
  personaChannel: "native",
44
43
  nativeAgentSelector: true,
45
44
  tools: "sdk",
46
- modelWorkTools: true,
47
- // The server config gives the routed model its inference entry, so the request must name a model, except
48
- // that model work's agent carries its options whichever model opencode picks.
49
- inference: (_profile, request) => opencodeCarriedKeys(Boolean(request.model?.resolved), request.inference, isModelWorkTools(request.tools)),
45
+ inference: MODEL_WORK_AGENT_INFERENCE,
50
46
  }),
51
47
  };
52
48
  // No flag-shaped resume: session reuse is programmatic — the SDK session id is
@@ -93,11 +93,10 @@
93
93
  import { spawn } from "node:child_process";
94
94
  import { createHash } from "node:crypto";
95
95
  import { isRecord } from "../../../core/common.js";
96
- import { COMMON_SPAWN_ENV_PASSTHROUGH, spawnEnvNamesFor } from "../../../core/spawn-env.js";
97
- import { isModelWorkTools } from "../../../execution/source.js";
96
+ import { COMMON_SPAWN_ENV_PASSTHROUGH, spawnEnvNamesFor, XDG_BASE_DIR_ENV_PASSTHROUGH } from "../../../core/spawn-env.js";
98
97
  import { DEFAULT_AGENT_TIMEOUT_MS } from "../../agent/config.js";
99
- import { opencodeInferenceConfig, opencodeModelConfig } from "../opencode/model-config.js";
100
- import { MODEL_WORK_OPENCODE_AGENT, MODEL_WORK_STEP_GRACE, MODEL_WORK_STEPS, modelWorkOpencodeConfig, } from "../opencode/model-work-agent.js";
98
+ import { opencodeInferenceConfig } from "../opencode/model-config.js";
99
+ import { MODEL_WORK_OPENCODE_AGENT, modelWorkOpencodeConfig, modelWorkPluginEnv } from "../opencode/model-work-agent.js";
101
100
  // Server registry — one server per complete server-material signature. Caller
102
101
  // deadlines race the shared promise independently; they never become startup
103
102
  // configuration inherited by later callers.
@@ -251,12 +250,12 @@ function fallbackInference(llmConfig) {
251
250
  * aliases resolve once before harness lowering. A server for model work also
252
251
  * defines the confined model-work agent (`../opencode/model-work-agent`).
253
252
  *
254
- * The routed model carries the dispatch's inference (`../opencode/model-config`):
255
- * the fallback LLM engine's, under `requestInference`, the request's own. A
256
- * model the config routes through `akm-custom` is declared in full; any other
257
- * model's entry merges over the user's own opencode config for it. For model
258
- * work the options go on the confined agent instead of the model, which needs
259
- * no model named: it runs whichever model opencode picks.
253
+ * The model the config routes through `akm-custom` is declared in full, with
254
+ * the dispatch's inference (`../opencode/model-config`): the fallback LLM
255
+ * engine's, under `requestInference`, the request's own. For model work the
256
+ * options go on the confined agent instead of the model, which needs no model
257
+ * named: it runs whichever model opencode picks. A model the user's own opencode
258
+ * config provides carries none: set inference there.
260
259
  */
261
260
  export function buildSdkConfig(profile, llmConfig, modelWork = false, requestInference) {
262
261
  const endpoint = llmConfig?.endpoint;
@@ -288,11 +287,6 @@ export function buildSdkConfig(profile, llmConfig, modelWork = false, requestInf
288
287
  if (modelId)
289
288
  sdkConfig.model = `akm-custom/${modelId}`;
290
289
  }
291
- else {
292
- const modelConfig = opencodeModelConfig(model, entry);
293
- if (modelConfig)
294
- Object.assign(sdkConfig, modelConfig);
295
- }
296
290
  return modelWork ? { ...sdkConfig, ...modelWorkOpencodeConfig(agentOptions) } : sdkConfig;
297
291
  }
298
292
  /** Digest the executable and exact environment received by the child. */
@@ -302,11 +296,23 @@ function serverRegistryKey(profile, env) {
302
296
  .update(JSON.stringify(canonicalize(material)))
303
297
  .digest("hex");
304
298
  }
305
- /** @internal Exact environment allowlist used to start the OpenCode SDK server. */
299
+ /**
300
+ * @internal Exact environment allowlist used to start the OpenCode SDK server:
301
+ * the common baseline, the XDG base-directory variables opencode resolves its
302
+ * config, data, cache and state from, and the profile's own names. The XDG names
303
+ * are the server's, not the profile's: profile `envPassthrough` is frozen into
304
+ * workflow plans, and an SDK profile's list has always been empty.
305
+ */
306
306
  export function opencodeSdkServerEnvironmentNames(profile) {
307
- return [...new Set([...spawnEnvNamesFor(COMMON_SPAWN_ENV_PASSTHROUGH), ...(profile.envPassthrough ?? [])])];
307
+ return [
308
+ ...new Set([
309
+ ...spawnEnvNamesFor(COMMON_SPAWN_ENV_PASSTHROUGH),
310
+ ...XDG_BASE_DIR_ENV_PASSTHROUGH,
311
+ ...(profile.envPassthrough ?? []),
312
+ ]),
313
+ ];
308
314
  }
309
- function buildServerEnv(profile, config, bindings, envSource) {
315
+ function buildServerEnv(profile, config, bindings, envSource, modelWork) {
310
316
  const env = {};
311
317
  for (const key of opencodeSdkServerEnvironmentNames(profile)) {
312
318
  const value = envSource[key];
@@ -315,6 +321,8 @@ function buildServerEnv(profile, config, bindings, envSource) {
315
321
  }
316
322
  for (const [key, value] of Object.entries(bindings ?? {}))
317
323
  env[key] = value;
324
+ if (modelWork)
325
+ Object.assign(env, modelWorkPluginEnv());
318
326
  env.OPENCODE_CONFIG_CONTENT = JSON.stringify(config);
319
327
  return env;
320
328
  }
@@ -586,7 +594,7 @@ function getOrStartServer(profile, llmConfig, env, envSource = process.env, mode
586
594
  if (_testServer)
587
595
  return { promise: Promise.resolve(_testServer), release() { } };
588
596
  const sdkConfig = buildSdkConfig(profile, llmConfig, modelWork, inference);
589
- const serverEnv = buildServerEnv(profile, sdkConfig, env, envSource);
597
+ const serverEnv = buildServerEnv(profile, sdkConfig, env, envSource, modelWork);
590
598
  const key = serverRegistryKey(profile, serverEnv);
591
599
  let entry = _servers.get(key);
592
600
  if (!entry) {
@@ -736,51 +744,6 @@ async function deleteSessionBestEffort(client, sessionId, query, setTimeoutFn, c
736
744
  function abortSessionBestEffort(client, sessionId, query) {
737
745
  void client.session.abort?.({ path: { id: sessionId }, ...(query ? { query } : {}) }).catch(() => { });
738
746
  }
739
- /** How often a model-work session's messages are polled for its step count. */
740
- const MODEL_WORK_STEP_POLL_MS = 1_000;
741
- /**
742
- * Count a model-work session's steps, its `step-start` parts, once a second
743
- * and abort the session once they pass `MODEL_WORK_STEPS` +
744
- * `MODEL_WORK_STEP_GRACE`: opencode 1.18.25 only asks the model to stop at the
745
- * agent's `steps`. Polled rather than read from the server's event stream,
746
- * because the SDK's stream cannot be closed mid-read without an unhandled
747
- * AbortError (its abort handler drops the promise `reader.cancel()` returns),
748
- * which the CLI turns into a crash. Best effort: the dispatch timeout still
749
- * bounds the session.
750
- */
751
- function watchModelWorkSteps(client, sessionId, query, timers) {
752
- let steps = 0;
753
- let timer;
754
- let stopped = false;
755
- const poll = async () => {
756
- const listed = await client.session.messages?.({ path: { id: sessionId }, ...(query ? { query } : {}) });
757
- if (stopped)
758
- return;
759
- steps = (listed?.data ?? [])
760
- .flatMap((message) => message.parts ?? [])
761
- .filter((p) => p.type === "step-start").length;
762
- if (steps > MODEL_WORK_STEPS + MODEL_WORK_STEP_GRACE)
763
- abortSessionBestEffort(client, sessionId, query);
764
- else
765
- schedule();
766
- };
767
- const schedule = () => {
768
- if (stopped || !client.session.messages)
769
- return;
770
- timer = timers.setTimeoutFn(() => void poll().catch(() => schedule()), MODEL_WORK_STEP_POLL_MS);
771
- if (typeof timer !== "number")
772
- timer.unref?.();
773
- };
774
- schedule();
775
- return {
776
- steps: () => steps,
777
- stop: () => {
778
- stopped = true;
779
- if (timer !== undefined)
780
- timers.clearTimeoutFn(timer);
781
- },
782
- };
783
- }
784
747
  function abortedBeforeSdkStart(profile) {
785
748
  return {
786
749
  ok: false,
@@ -801,7 +764,7 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
801
764
  const clearTimeoutImpl = opts.clearTimeoutFn ?? clearTimeout;
802
765
  if (opts.signal?.aborted)
803
766
  return abortedBeforeSdkStart(profile);
804
- const modelWork = isModelWorkTools(opts.dispatch?.tools);
767
+ const modelWork = opts.dispatch?.modelWork === true;
805
768
  let client;
806
769
  if (_testServer) {
807
770
  client = _testServer.client;
@@ -935,7 +898,7 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
935
898
  const dispatch = opts.dispatch;
936
899
  const agent = modelWork ? MODEL_WORK_OPENCODE_AGENT : dispatch?.agent;
937
900
  const system = dispatch?.systemPrompt;
938
- const tools = modelWork ? undefined : toolsToSdkAllowlist(dispatch?.tools);
901
+ const tools = toolsToSdkAllowlist(dispatch?.tools);
939
902
  const body = { parts: [{ type: "text", text: prompt }] };
940
903
  if (agent)
941
904
  body.agent = agent;
@@ -944,9 +907,6 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
944
907
  if (tools)
945
908
  body.tools = tools;
946
909
  let result;
947
- const steps = modelWork
948
- ? watchModelWorkSteps(client, sessionId, query, { setTimeoutFn: setTimeoutImpl, clearTimeoutFn: clearTimeoutImpl })
949
- : undefined;
950
910
  try {
951
911
  const prompted = await raceSdkOperation(client.session.prompt({ path: { id: sessionId }, body, ...(query ? { query } : {}) }), {
952
912
  timeoutMs: remainingTimeoutMs(),
@@ -957,8 +917,8 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
957
917
  void deleteSessionBestEffort(client, sessionId, query, setTimeoutImpl, clearTimeoutImpl);
958
918
  },
959
919
  });
960
- // A model-work session the dispatch stops early is aborted on the server too.
961
- if (modelWork && (prompted === SDK_OPERATION_ABORTED || prompted === SDK_OPERATION_TIMED_OUT)) {
920
+ // A session the dispatch stops early is aborted on the server too.
921
+ if (prompted === SDK_OPERATION_ABORTED || prompted === SDK_OPERATION_TIMED_OUT) {
962
922
  abortSessionBestEffort(client, sessionId, query);
963
923
  }
964
924
  if (prompted === SDK_OPERATION_ABORTED) {
@@ -994,22 +954,7 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
994
954
  // default sdk runner.
995
955
  const usage = extractUsage(prompted.data?.info);
996
956
  const sdkError = prompted.error ?? prompted.data?.info?.error;
997
- const overSteps = steps !== undefined && steps.steps() > MODEL_WORK_STEPS + MODEL_WORK_STEP_GRACE;
998
- if (overSteps) {
999
- const message = `opencode-sdk agent "${profile.name}" ran past the ${MODEL_WORK_STEPS}-step limit for model work; akm aborted the session.`;
1000
- result = {
1001
- ok: false,
1002
- stdout,
1003
- stderr: message,
1004
- durationMs: Date.now() - start,
1005
- exitCode: 1,
1006
- reason: "parse_error",
1007
- error: message,
1008
- sessionId,
1009
- ...(usage ? { usage } : {}),
1010
- };
1011
- }
1012
- else if (sdkError) {
957
+ if (sdkError) {
1013
958
  const failure = sdkErrorFailure(sdkError);
1014
959
  result = {
1015
960
  ok: false,
@@ -1048,9 +993,6 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
1048
993
  sessionId,
1049
994
  };
1050
995
  }
1051
- finally {
1052
- steps?.stop();
1053
- }
1054
996
  // Clean up session to prevent disk accumulation in ~/.local/share/opencode/.
1055
997
  // Failures are non-fatal to the agent result but must not be invisible.
1056
998
  const cleanupWarning = await deleteSessionBestEffort(client, sessionId, query, setTimeoutImpl, clearTimeoutImpl);
@@ -59,8 +59,7 @@
59
59
  * akm's `workflow_run_units` remains the durable source of truth either way
60
60
  * (plan §"Session, MCP, and identity across harnesses").
61
61
  * - **inference** — not translated: the shared lowering reports each field of
62
- * the request's inference as untranslated (`inference` in `harnesses/ids.ts`
63
- * lists none for this harness).
62
+ * the request's inference as untranslated.
64
63
  *
65
64
  * Registered: `openhandsBuilder` is `OpenhandsHarness.agentBuilder`
66
65
  * (`./index.ts`), one of the ten harnesses `HARNESS_REGISTRY` constructs
@@ -40,8 +40,7 @@
40
40
  * produce a silently broken command. A restrictive policy is therefore
41
41
  * dropped rather than approximated — never silently widened.
42
42
  * - **inference** — not translated: the shared lowering reports each field of
43
- * the request's inference as untranslated (`inference` in `harnesses/ids.ts`
44
- * lists none for this harness).
43
+ * the request's inference as untranslated.
45
44
  *
46
45
  * Registered: `piBuilder` is `PiHarness.agentBuilder` (`./index.ts`), one of
47
46
  * the ten harnesses `HARNESS_REGISTRY` constructs (`harnesses/index.ts`);
@@ -34,14 +34,14 @@ export function isSessionLogHarness(h) {
34
34
  * `HarnessCapabilities` union type here, and each subclass narrows it to a
35
35
  * literal `sessionLogs: true`/`false` variant via `caps({...})`.
36
36
  * `sessionLogProvider` is intentionally NOT declared here at all (not even
37
- * optional): the 8 non-session-log subclasses simply never declare it, which
37
+ * optional): the 7 non-session-log subclasses simply never declare it, which
38
38
  * satisfies `NonSessionLogHarness`'s `sessionLogProvider?: undefined` (an
39
39
  * absent optional property satisfies an `undefined`-typed optional); had this
40
40
  * class declared it as `sessionLogProvider?: () => SessionLogHarness`
41
41
  * instead, every non-session-log subclass would inherit that (function |
42
42
  * undefined) type and fail to satisfy `NonSessionLogHarness` at the
43
- * `HARNESS_REGISTRY` `satisfies` check. The two session-log subclasses
44
- * (Claude, OpenCode) declare their own required `sessionLogProvider`.
43
+ * `HARNESS_REGISTRY` `satisfies` check. The three session-log subclasses
44
+ * (Claude, Codex, OpenCode) declare their own required `sessionLogProvider`.
45
45
  */
46
46
  export class BaseHarness {
47
47
  setupDetectionDir;
@@ -1,6 +1,7 @@
1
1
  // This Source Code Form is subject to the terms of the Mozilla Public
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ import { DEFAULT_LLM_TIMEOUT_MS } from "../integrations/agent/config.js";
4
5
  /**
5
6
  * For each feature key, return the effective enabled state by reading the
6
7
  * 0.9.0 config shape.
@@ -37,10 +38,6 @@ export function isLlmFeatureEnabled(config, feature, improveEnabled) {
37
38
  return false;
38
39
  return resolver(config);
39
40
  }
40
- /**
41
- * Default hard timeout for every bounded in-tree LLM call.
42
- */
43
- const DEFAULT_TIMEOUT_MS = 600_000;
44
41
  /**
45
42
  * Run `fn()` only if `isLlmFeatureEnabled(config, feature)` is `true`. On
46
43
  * disablement, throw, or timeout, return `fallback` (or — if it is a
@@ -53,7 +50,7 @@ export async function tryLlmFeature(feature, config, fn, fallback, opts) {
53
50
  opts?.onFallback?.({ feature, reason: "disabled" });
54
51
  return resolveFallback();
55
52
  }
56
- const timeoutMs = opts && Object.hasOwn(opts, "timeoutMs") ? (opts.timeoutMs ?? null) : DEFAULT_TIMEOUT_MS;
53
+ const timeoutMs = opts && Object.hasOwn(opts, "timeoutMs") ? (opts.timeoutMs ?? null) : DEFAULT_LLM_TIMEOUT_MS;
57
54
  try {
58
55
  if (timeoutMs === null || timeoutMs <= 0) {
59
56
  return await fn(new AbortController().signal);
@@ -2,7 +2,6 @@
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  import { cloneExecutionJsonObject } from "../execution/json.js";
5
- import { MODEL_WORK_TOOLS } from "../execution/source.js";
6
5
  import { buildExecution, resolveExecution } from "../integrations/agent/execution.js";
7
6
  const NO_LOWERING_NOTICES = Object.freeze([]);
8
7
  function own(value, key) {
@@ -43,7 +42,8 @@ export function resolveIndexPassExecution(passName, config) {
43
42
  content: "",
44
43
  config,
45
44
  invocationDefaults,
46
- current: { ...indexExecutionDefaults(pass), tools: MODEL_WORK_TOOLS },
45
+ current: indexExecutionDefaults(pass),
46
+ modelWork: true,
47
47
  });
48
48
  const lowered = buildExecution(prepared.request, prepared.runner);
49
49
  return Object.freeze({ runner: lowered.runner, notices: lowered.notices });
@@ -2,8 +2,7 @@
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  import { ConfigError } from "../core/errors.js";
5
- import { MODEL_WORK_TOOLS } from "../execution/source.js";
6
- import { DEFAULT_MODEL_WORK_TIMEOUT_MS } from "../integrations/agent/config.js";
5
+ import { DEFAULT_LLM_TIMEOUT_MS } from "../integrations/agent/config.js";
7
6
  import { buildExecution, resolveExecution } from "../integrations/agent/execution.js";
8
7
  import { runExecution } from "../integrations/agent/runner-dispatch.js";
9
8
  import { isContextSizeError, LlmCallError } from "./client.js";
@@ -27,7 +26,7 @@ function own(value, key) {
27
26
  /**
28
27
  * @internal Exact request-to-cascade projection, exported for presence-semantics contracts.
29
28
  * Model work is bounded: with no timeout from the request or the runner, it
30
- * gets {@link DEFAULT_MODEL_WORK_TIMEOUT_MS} on every runner kind.
29
+ * gets {@link DEFAULT_LLM_TIMEOUT_MS} on every runner kind.
31
30
  */
32
31
  export function resolveStructuredCurrent(current, request, runner) {
33
32
  const out = current ? { ...current } : {};
@@ -53,7 +52,7 @@ export function resolveStructuredCurrent(current, request, runner) {
53
52
  if (own(request, "timeoutMs"))
54
53
  out.timeout = request?.timeoutMs ?? null;
55
54
  else if (runner && !Object.hasOwn(runner, "timeoutMs") && !own(current, "timeout")) {
56
- out.timeout = DEFAULT_MODEL_WORK_TIMEOUT_MS;
55
+ out.timeout = DEFAULT_LLM_TIMEOUT_MS;
57
56
  }
58
57
  return Object.keys(out).length > 0 ? out : undefined;
59
58
  }
@@ -104,7 +103,8 @@ export async function callStructured(opts) {
104
103
  content: terminal.content,
105
104
  conversation: terminal.conversation,
106
105
  runner,
107
- current: { ...resolveStructuredCurrent(opts.current, request, runner), tools: MODEL_WORK_TOOLS },
106
+ current: resolveStructuredCurrent(opts.current, request, runner),
107
+ modelWork: true,
108
108
  });
109
109
  const lowered = buildExecution(prepared.request, prepared.runner);
110
110
  opts.onNotices?.(lowered.notices);
@@ -43,6 +43,7 @@ const PASSTHROUGH_COMMANDS = [
43
43
  "extract",
44
44
  "health",
45
45
  "improve",
46
+ "improve-judge",
46
47
  "improve-report",
47
48
  "import",
48
49
  "index",