akm-cli 0.9.24 → 0.9.25-alpha.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +294 -0
- package/dist/commands/health/checks.js +10 -11
- package/dist/commands/improve/consolidate/pair-pass.js +3 -2
- package/dist/commands/improve/consolidate.js +3 -2
- package/dist/commands/improve/execution.js +4 -10
- package/dist/commands/improve/extract-prompt.js +4 -4
- package/dist/commands/improve/extract.js +10 -13
- package/dist/commands/improve/improve-cli.js +32 -33
- package/dist/commands/improve/improve-strategies.js +49 -43
- package/dist/commands/improve/improve-usage-report.js +8 -17
- package/dist/commands/improve/preparation.js +3 -1
- package/dist/commands/improve/reflect.js +108 -172
- package/dist/commands/improve/stage.js +69 -18
- package/dist/commands/proposal/drain.js +14 -2
- package/dist/commands/proposal/proposal-cli.js +1 -5
- package/dist/commands/proposal/propose-cli.js +2 -2
- package/dist/commands/proposal/propose.js +80 -83
- package/dist/commands/remember.js +3 -3
- package/dist/commands/sources/schema-repair.js +1 -1
- package/dist/core/config/config-schema.js +37 -60
- package/dist/core/config/engine-semantics.js +15 -11
- package/dist/core/config/schema/engines.js +33 -15
- package/dist/core/config/schema/improve-processes.js +2 -2
- package/dist/core/improve-result.js +3 -3
- package/dist/core/structured.js +10 -0
- package/dist/execution/source.js +14 -0
- package/dist/indexer/passes/memory-inference.js +2 -1
- package/dist/integrations/agent/builder-shared.js +15 -0
- package/dist/integrations/agent/config.js +2 -0
- package/dist/integrations/agent/engine-resolution.js +16 -31
- package/dist/integrations/agent/execution.js +46 -21
- package/dist/integrations/agent/index.js +1 -1
- package/dist/integrations/agent/model-map.js +16 -15
- package/dist/integrations/agent/prompts.js +53 -76
- package/dist/integrations/agent/request-lowering.js +20 -9
- package/dist/integrations/agent/runner-dispatch.js +103 -4
- package/dist/integrations/agent/runner.js +8 -2
- package/dist/integrations/harnesses/aider/agent-builder.js +11 -23
- package/dist/integrations/harnesses/aider/index.js +0 -5
- package/dist/integrations/harnesses/amazonq/agent-builder.js +10 -21
- package/dist/integrations/harnesses/amazonq/index.js +0 -5
- package/dist/integrations/harnesses/claude/agent-builder.js +61 -38
- package/dist/integrations/harnesses/claude/index.js +0 -14
- package/dist/integrations/harnesses/codex/agent-builder.js +4 -4
- package/dist/integrations/harnesses/codex/index.js +0 -4
- package/dist/integrations/harnesses/copilot/agent-builder.js +4 -13
- package/dist/integrations/harnesses/copilot/index.js +2 -7
- package/dist/integrations/harnesses/gemini/agent-builder.js +4 -13
- package/dist/integrations/harnesses/gemini/index.js +0 -5
- package/dist/integrations/harnesses/ids.js +18 -10
- package/dist/integrations/harnesses/opencode/agent-builder.js +52 -10
- package/dist/integrations/harnesses/opencode/index.js +0 -8
- package/dist/integrations/harnesses/opencode/model-config.js +80 -0
- package/dist/integrations/harnesses/opencode/model-work-agent.js +80 -0
- package/dist/integrations/harnesses/opencode-sdk/harness.js +6 -7
- package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +168 -24
- package/dist/integrations/harnesses/openhands/agent-builder.js +8 -21
- package/dist/integrations/harnesses/openhands/index.js +0 -5
- package/dist/integrations/harnesses/pi/agent-builder.js +8 -21
- package/dist/integrations/harnesses/pi/index.js +0 -5
- package/dist/llm/client.js +5 -0
- package/dist/llm/feature-gate.js +10 -4
- package/dist/llm/index-passes.js +3 -6
- package/dist/llm/memory-infer.js +6 -5
- package/dist/llm/structured-call.js +33 -11
- package/dist/scripts/akm-migrate-node.js +298 -204
- package/dist/scripts/akm-migrate.js +298 -204
- package/dist/workflows/exec/step-work.js +6 -5
- package/dist/workflows/freeze/step-values.js +1 -1
- package/docs/reference/cli.md +29 -11
- package/docs/reference/configuration.md +168 -11
- package/docs/reference/workflow-schema.md +13 -9
- package/package.json +1 -1
- package/schemas/akm-config.json +36 -0
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
/**
|
|
5
|
+
* The opencode agent that runs unattended model work under the model-work
|
|
6
|
+
* tool policy (`MODEL_WORK_TOOLS`). The CLI builder injects it through
|
|
7
|
+
* `OPENCODE_CONFIG_CONTENT` and selects it with `--agent`; the SDK runner adds
|
|
8
|
+
* it to its server config and names it in the prompt body.
|
|
9
|
+
*
|
|
10
|
+
* What opencode 1.18.25 confines, checked against a local stub:
|
|
11
|
+
* - read and edit stay inside the session directory (`external_directory`
|
|
12
|
+
* is denied). Write is part of opencode's edit permission, so it is
|
|
13
|
+
* confined the same way and cannot be denied on its own.
|
|
14
|
+
* - bash is denied, so `akm search` and `akm show` are not granted: opencode
|
|
15
|
+
* matches a bash rule against the command's words only, so
|
|
16
|
+
* `akm show x > ~/stash/asset.md` would pass an `akm show *` rule and
|
|
17
|
+
* write anywhere.
|
|
18
|
+
* - every other tool is denied, `doom_loop` included (its default, `ask`,
|
|
19
|
+
* would hang a headless server). Each permission opencode knows is named,
|
|
20
|
+
* so a same-named agent in the user's config cannot re-allow one through
|
|
21
|
+
* opencode's config merge, and `*` covers the rest.
|
|
22
|
+
* The same rules also go in the top-level `permission`, so an opencode that
|
|
23
|
+
* falls back to its default agent is confined too.
|
|
24
|
+
*
|
|
25
|
+
* The agent also keeps opencode's coding-assistant defaults out of model work:
|
|
26
|
+
* - its own short `prompt` replaces the provider's coding prompt, which
|
|
27
|
+
* tells the model to search extensively; opencode appends a request's
|
|
28
|
+
* system text after it and never substitutes it;
|
|
29
|
+
* - `steps` bounds the agentic loop. opencode only asks the model to stop
|
|
30
|
+
* at the limit, so the SDK runner also aborts the session a little past
|
|
31
|
+
* it (`MODEL_WORK_STEP_GRACE`); on the CLI the dispatch timeout bounds it;
|
|
32
|
+
* - automatic compaction is off, so a long run cannot summarize the task
|
|
33
|
+
* away.
|
|
34
|
+
*/
|
|
35
|
+
export const MODEL_WORK_OPENCODE_AGENT = "akm-model-work";
|
|
36
|
+
/** The agentic iterations a model-work run may take: a judge answers in one, a generator in a few. */
|
|
37
|
+
export const MODEL_WORK_STEPS = 8;
|
|
38
|
+
/** Steps past {@link MODEL_WORK_STEPS} after which the SDK runner aborts the session. */
|
|
39
|
+
export const MODEL_WORK_STEP_GRACE = 2;
|
|
40
|
+
const MODEL_WORK_PROMPT = "You do one bounded task for akm. Use tools only to check what the task needs, never repeat a tool call, and reply with exactly what the task asks for.";
|
|
41
|
+
const MODEL_WORK_PERMISSION = {
|
|
42
|
+
"*": "deny",
|
|
43
|
+
read: "allow",
|
|
44
|
+
edit: "allow",
|
|
45
|
+
external_directory: "deny",
|
|
46
|
+
bash: "deny",
|
|
47
|
+
doom_loop: "deny",
|
|
48
|
+
glob: "deny",
|
|
49
|
+
grep: "deny",
|
|
50
|
+
list: "deny",
|
|
51
|
+
lsp: "deny",
|
|
52
|
+
question: "deny",
|
|
53
|
+
skill: "deny",
|
|
54
|
+
task: "deny",
|
|
55
|
+
todowrite: "deny",
|
|
56
|
+
webfetch: "deny",
|
|
57
|
+
websearch: "deny",
|
|
58
|
+
};
|
|
59
|
+
/**
|
|
60
|
+
* The opencode config fragment that defines and confines the model-work agent.
|
|
61
|
+
* The agent carries the request's inference options (`model-config.ts`), which
|
|
62
|
+
* apply to its calls only: opencode's own calls on the same model, a title for
|
|
63
|
+
* the session, keep the model's defaults.
|
|
64
|
+
*/
|
|
65
|
+
export function modelWorkOpencodeConfig(options) {
|
|
66
|
+
return {
|
|
67
|
+
permission: { ...MODEL_WORK_PERMISSION },
|
|
68
|
+
compaction: { auto: false },
|
|
69
|
+
agent: {
|
|
70
|
+
[MODEL_WORK_OPENCODE_AGENT]: {
|
|
71
|
+
mode: "primary",
|
|
72
|
+
description: "akm unattended model work: read and edit inside its working directory only.",
|
|
73
|
+
prompt: MODEL_WORK_PROMPT,
|
|
74
|
+
steps: MODEL_WORK_STEPS,
|
|
75
|
+
...(options ? { options } : {}),
|
|
76
|
+
permission: { ...MODEL_WORK_PERMISSION },
|
|
77
|
+
},
|
|
78
|
+
},
|
|
79
|
+
};
|
|
80
|
+
}
|
|
@@ -21,7 +21,9 @@
|
|
|
21
21
|
* Importing the descriptor from this leaf keeps the registry a config-leaf and
|
|
22
22
|
* breaks the cycle.
|
|
23
23
|
*/
|
|
24
|
+
import { isModelWorkTools } from "../../../execution/source.js";
|
|
24
25
|
import { createAgentRequestLowerer } from "../../agent/request-lowering.js";
|
|
26
|
+
import { opencodeCarriedKeys } from "../opencode/model-config.js";
|
|
25
27
|
import { caps } from "../shared.js";
|
|
26
28
|
import { BaseHarness } from "../types.js";
|
|
27
29
|
/**
|
|
@@ -33,12 +35,6 @@ export class OpencodeSdkHarness extends BaseHarness {
|
|
|
33
35
|
id = "opencode-sdk";
|
|
34
36
|
displayName = "OpenCode SDK";
|
|
35
37
|
// ── Workflow-engine descriptor (plan §"Capability matrix", P2) ────────────
|
|
36
|
-
// Embedded-SDK dispatch on this machine ⇒ local-runner (the matrix's
|
|
37
|
-
// "local (sdk/cli)" row, SDK half).
|
|
38
|
-
pattern = "local-runner";
|
|
39
|
-
// `session.prompt` returns structured SDK events/messages; akm extracts the
|
|
40
|
-
// final message then validates against the node schema ⇒ native-json tier.
|
|
41
|
-
structuredOutput = "native-json";
|
|
42
38
|
executionLowerer = {
|
|
43
39
|
platform: "opencode-sdk",
|
|
44
40
|
personaChannel: "native",
|
|
@@ -47,7 +43,10 @@ export class OpencodeSdkHarness extends BaseHarness {
|
|
|
47
43
|
personaChannel: "native",
|
|
48
44
|
nativeAgentSelector: true,
|
|
49
45
|
tools: "sdk",
|
|
50
|
-
|
|
46
|
+
modelWorkTools: true,
|
|
47
|
+
// The server config gives the routed model its inference entry, so the request must name a model, except
|
|
48
|
+
// that model work's agent carries its options whichever model opencode picks.
|
|
49
|
+
inference: (_profile, request) => opencodeCarriedKeys(Boolean(request.model?.resolved), request.inference, isModelWorkTools(request.tools)),
|
|
51
50
|
}),
|
|
52
51
|
};
|
|
53
52
|
// No flag-shaped resume: session reuse is programmatic — the SDK session id is
|
|
@@ -92,8 +92,12 @@
|
|
|
92
92
|
*/
|
|
93
93
|
import { spawn } from "node:child_process";
|
|
94
94
|
import { createHash } from "node:crypto";
|
|
95
|
+
import { isRecord } from "../../../core/common.js";
|
|
95
96
|
import { COMMON_SPAWN_ENV_PASSTHROUGH, spawnEnvNamesFor } from "../../../core/spawn-env.js";
|
|
97
|
+
import { isModelWorkTools } from "../../../execution/source.js";
|
|
96
98
|
import { DEFAULT_AGENT_TIMEOUT_MS } from "../../agent/config.js";
|
|
99
|
+
import { opencodeInferenceConfig, opencodeModelConfig } from "../opencode/model-config.js";
|
|
100
|
+
import { MODEL_WORK_OPENCODE_AGENT, MODEL_WORK_STEP_GRACE, MODEL_WORK_STEPS, modelWorkOpencodeConfig, } from "../opencode/model-work-agent.js";
|
|
97
101
|
// Server registry — one server per complete server-material signature. Caller
|
|
98
102
|
// deadlines race the shared promise independently; they never become startup
|
|
99
103
|
// configuration inherited by later callers.
|
|
@@ -232,21 +236,45 @@ function toolsToSdkAllowlist(tools) {
|
|
|
232
236
|
out[n] = true;
|
|
233
237
|
return out;
|
|
234
238
|
}
|
|
239
|
+
/** The inference a fallback LLM connection carries, which opencode can take. */
|
|
240
|
+
function fallbackInference(llmConfig) {
|
|
241
|
+
const out = {};
|
|
242
|
+
for (const key of ["temperature", "maxTokens", "contextLength", "enableThinking", "reasoningEffort"]) {
|
|
243
|
+
if (llmConfig?.[key] !== undefined)
|
|
244
|
+
out[key] = llmConfig[key];
|
|
245
|
+
}
|
|
246
|
+
return out;
|
|
247
|
+
}
|
|
235
248
|
/**
|
|
236
249
|
* Assemble the OpenCode SDK server config from the profile + LLM fallback.
|
|
237
250
|
* Pure and exported for tests. `profile.model` is already exact because model
|
|
238
|
-
* aliases resolve once before harness lowering.
|
|
251
|
+
* aliases resolve once before harness lowering. A server for model work also
|
|
252
|
+
* defines the confined model-work agent (`../opencode/model-work-agent`).
|
|
253
|
+
*
|
|
254
|
+
* The routed model carries the dispatch's inference (`../opencode/model-config`):
|
|
255
|
+
* the fallback LLM engine's, under `requestInference`, the request's own. A
|
|
256
|
+
* model the config routes through `akm-custom` is declared in full; any other
|
|
257
|
+
* model's entry merges over the user's own opencode config for it. For model
|
|
258
|
+
* work the options go on the confined agent instead of the model, which needs
|
|
259
|
+
* no model named: it runs whichever model opencode picks.
|
|
239
260
|
*/
|
|
240
|
-
export function buildSdkConfig(profile, llmConfig) {
|
|
261
|
+
export function buildSdkConfig(profile, llmConfig, modelWork = false, requestInference) {
|
|
241
262
|
const endpoint = llmConfig?.endpoint;
|
|
242
263
|
const apiKey = llmConfig?.apiKey;
|
|
243
264
|
const profileModel = profile.model;
|
|
244
265
|
const model = profileModel ?? llmConfig?.model;
|
|
266
|
+
const inference = requestInference === null ? undefined : { ...fallbackInference(llmConfig), ...requestInference };
|
|
267
|
+
const { entry, agentOptions } = opencodeInferenceConfig(inference, modelWork);
|
|
245
268
|
const sdkConfig = {};
|
|
246
269
|
if (model)
|
|
247
270
|
sdkConfig.model = model;
|
|
248
271
|
if (endpoint || apiKey) {
|
|
249
|
-
//
|
|
272
|
+
// The first path segment selects the OpenCode provider. Model IDs may
|
|
273
|
+
// themselves contain slashes, but still belong to this custom endpoint.
|
|
274
|
+
const modelId = model?.startsWith("akm-custom/") ? model.slice("akm-custom/".length) : model;
|
|
275
|
+
// Configure a custom OpenAI-compatible provider. OpenCode registers only
|
|
276
|
+
// the models a custom provider lists, so the routed model is declared
|
|
277
|
+
// (#1015: without it every dispatch failed with ProviderModelNotFoundError).
|
|
250
278
|
sdkConfig.provider = {
|
|
251
279
|
"akm-custom": {
|
|
252
280
|
npm: "@ai-sdk/openai-compatible",
|
|
@@ -254,14 +282,18 @@ export function buildSdkConfig(profile, llmConfig) {
|
|
|
254
282
|
baseURL: canonicalProviderBase(endpoint) ?? undefined,
|
|
255
283
|
...(apiKey ? { apiKey } : {}),
|
|
256
284
|
},
|
|
285
|
+
...(modelId ? { models: { [modelId]: entry } } : {}),
|
|
257
286
|
},
|
|
258
287
|
};
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
288
|
+
if (modelId)
|
|
289
|
+
sdkConfig.model = `akm-custom/${modelId}`;
|
|
290
|
+
}
|
|
291
|
+
else {
|
|
292
|
+
const modelConfig = opencodeModelConfig(model, entry);
|
|
293
|
+
if (modelConfig)
|
|
294
|
+
Object.assign(sdkConfig, modelConfig);
|
|
263
295
|
}
|
|
264
|
-
return sdkConfig;
|
|
296
|
+
return modelWork ? { ...sdkConfig, ...modelWorkOpencodeConfig(agentOptions) } : sdkConfig;
|
|
265
297
|
}
|
|
266
298
|
/** Digest the executable and exact environment received by the child. */
|
|
267
299
|
function serverRegistryKey(profile, env) {
|
|
@@ -550,10 +582,10 @@ async function startServer(profile, sdkConfig, env, registryKey, startupSignal)
|
|
|
550
582
|
* start (the registry stores the in-flight promise). A failed start is
|
|
551
583
|
* evicted so the next call can retry instead of caching the error forever.
|
|
552
584
|
*/
|
|
553
|
-
function getOrStartServer(profile, llmConfig, env, envSource = process.env) {
|
|
585
|
+
function getOrStartServer(profile, llmConfig, env, envSource = process.env, modelWork = false, inference) {
|
|
554
586
|
if (_testServer)
|
|
555
587
|
return { promise: Promise.resolve(_testServer), release() { } };
|
|
556
|
-
const sdkConfig = buildSdkConfig(profile, llmConfig);
|
|
588
|
+
const sdkConfig = buildSdkConfig(profile, llmConfig, modelWork, inference);
|
|
557
589
|
const serverEnv = buildServerEnv(profile, sdkConfig, env, envSource);
|
|
558
590
|
const key = serverRegistryKey(profile, serverEnv);
|
|
559
591
|
let entry = _servers.get(key);
|
|
@@ -665,6 +697,25 @@ function errorText(err) {
|
|
|
665
697
|
function appendStderr(stderr, message) {
|
|
666
698
|
return stderr ? `${stderr}\n${message}` : message;
|
|
667
699
|
}
|
|
700
|
+
/**
|
|
701
|
+
* Map an SDK `{ error }` result or a reply's `info.error` onto a failure
|
|
702
|
+
* (#1015). Both are opencode NamedErrors, `{ name, data: { message } }`. An
|
|
703
|
+
* aborted message is `aborted`, a reply cut off at the output limit is
|
|
704
|
+
* `parse_error`, and every other error (auth, API, unknown, an HTTP error
|
|
705
|
+
* body) is `non_zero_exit`.
|
|
706
|
+
*/
|
|
707
|
+
function sdkErrorFailure(error) {
|
|
708
|
+
if (!isRecord(error) || typeof error.name !== "string") {
|
|
709
|
+
return { reason: "non_zero_exit", message: typeof error === "string" ? error : JSON.stringify(error) };
|
|
710
|
+
}
|
|
711
|
+
const detail = isRecord(error.data) ? error.data.message : undefined;
|
|
712
|
+
const message = typeof detail === "string" ? `${error.name}: ${detail}` : error.name;
|
|
713
|
+
if (error.name === "MessageAbortedError")
|
|
714
|
+
return { reason: "aborted", message };
|
|
715
|
+
if (error.name === "MessageOutputLengthError")
|
|
716
|
+
return { reason: "parse_error", message };
|
|
717
|
+
return { reason: "non_zero_exit", message };
|
|
718
|
+
}
|
|
668
719
|
async function deleteSessionBestEffort(client, sessionId, query, setTimeoutFn, clearTimeoutFn) {
|
|
669
720
|
try {
|
|
670
721
|
const deleted = await raceSdkOperation(client.session.delete({ path: { id: sessionId }, ...(query ? { query } : {}) }), {
|
|
@@ -681,6 +732,55 @@ async function deleteSessionBestEffort(client, sessionId, query, setTimeoutFn, c
|
|
|
681
732
|
return `OpenCode session cleanup failed: ${errorText(err)}`;
|
|
682
733
|
}
|
|
683
734
|
}
|
|
735
|
+
/** Stop a server-side session that the dispatch has given up on, so it stops calling the model. */
|
|
736
|
+
function abortSessionBestEffort(client, sessionId, query) {
|
|
737
|
+
void client.session.abort?.({ path: { id: sessionId }, ...(query ? { query } : {}) }).catch(() => { });
|
|
738
|
+
}
|
|
739
|
+
/** How often a model-work session's messages are polled for its step count. */
|
|
740
|
+
const MODEL_WORK_STEP_POLL_MS = 1_000;
|
|
741
|
+
/**
|
|
742
|
+
* Count a model-work session's steps, its `step-start` parts, once a second
|
|
743
|
+
* and abort the session once they pass `MODEL_WORK_STEPS` +
|
|
744
|
+
* `MODEL_WORK_STEP_GRACE`: opencode 1.18.25 only asks the model to stop at the
|
|
745
|
+
* agent's `steps`. Polled rather than read from the server's event stream,
|
|
746
|
+
* because the SDK's stream cannot be closed mid-read without an unhandled
|
|
747
|
+
* AbortError (its abort handler drops the promise `reader.cancel()` returns),
|
|
748
|
+
* which the CLI turns into a crash. Best effort: the dispatch timeout still
|
|
749
|
+
* bounds the session.
|
|
750
|
+
*/
|
|
751
|
+
function watchModelWorkSteps(client, sessionId, query, timers) {
|
|
752
|
+
let steps = 0;
|
|
753
|
+
let timer;
|
|
754
|
+
let stopped = false;
|
|
755
|
+
const poll = async () => {
|
|
756
|
+
const listed = await client.session.messages?.({ path: { id: sessionId }, ...(query ? { query } : {}) });
|
|
757
|
+
if (stopped)
|
|
758
|
+
return;
|
|
759
|
+
steps = (listed?.data ?? [])
|
|
760
|
+
.flatMap((message) => message.parts ?? [])
|
|
761
|
+
.filter((p) => p.type === "step-start").length;
|
|
762
|
+
if (steps > MODEL_WORK_STEPS + MODEL_WORK_STEP_GRACE)
|
|
763
|
+
abortSessionBestEffort(client, sessionId, query);
|
|
764
|
+
else
|
|
765
|
+
schedule();
|
|
766
|
+
};
|
|
767
|
+
const schedule = () => {
|
|
768
|
+
if (stopped || !client.session.messages)
|
|
769
|
+
return;
|
|
770
|
+
timer = timers.setTimeoutFn(() => void poll().catch(() => schedule()), MODEL_WORK_STEP_POLL_MS);
|
|
771
|
+
if (typeof timer !== "number")
|
|
772
|
+
timer.unref?.();
|
|
773
|
+
};
|
|
774
|
+
schedule();
|
|
775
|
+
return {
|
|
776
|
+
steps: () => steps,
|
|
777
|
+
stop: () => {
|
|
778
|
+
stopped = true;
|
|
779
|
+
if (timer !== undefined)
|
|
780
|
+
timers.clearTimeoutFn(timer);
|
|
781
|
+
},
|
|
782
|
+
};
|
|
783
|
+
}
|
|
684
784
|
function abortedBeforeSdkStart(profile) {
|
|
685
785
|
return {
|
|
686
786
|
ok: false,
|
|
@@ -701,12 +801,13 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
|
|
|
701
801
|
const clearTimeoutImpl = opts.clearTimeoutFn ?? clearTimeout;
|
|
702
802
|
if (opts.signal?.aborted)
|
|
703
803
|
return abortedBeforeSdkStart(profile);
|
|
804
|
+
const modelWork = isModelWorkTools(opts.dispatch?.tools);
|
|
704
805
|
let client;
|
|
705
806
|
if (_testServer) {
|
|
706
807
|
client = _testServer.client;
|
|
707
808
|
}
|
|
708
809
|
else {
|
|
709
|
-
const startupHandle = getOrStartServer(profile, llmConfig, opts.env, opts.envSource);
|
|
810
|
+
const startupHandle = getOrStartServer(profile, llmConfig, opts.env, opts.envSource, modelWork, opts.dispatch?.inference);
|
|
710
811
|
try {
|
|
711
812
|
const startup = await raceSdkOperation(startupHandle.promise, {
|
|
712
813
|
timeoutMs: remainingTimeoutMs(),
|
|
@@ -830,10 +931,11 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
|
|
|
830
931
|
// dispatch request. Both were previously accepted on AgentDispatchRequest but
|
|
831
932
|
// silently dropped on the SDK path, so SDK-mode dispatch ignored agent-asset
|
|
832
933
|
// system prompts and tool policies entirely (the CLI path honours both).
|
|
934
|
+
// Model work runs the confined agent its server config defines; its tools are that agent's.
|
|
833
935
|
const dispatch = opts.dispatch;
|
|
834
|
-
const agent = dispatch?.agent;
|
|
936
|
+
const agent = modelWork ? MODEL_WORK_OPENCODE_AGENT : dispatch?.agent;
|
|
835
937
|
const system = dispatch?.systemPrompt;
|
|
836
|
-
const tools = toolsToSdkAllowlist(dispatch?.tools);
|
|
938
|
+
const tools = modelWork ? undefined : toolsToSdkAllowlist(dispatch?.tools);
|
|
837
939
|
const body = { parts: [{ type: "text", text: prompt }] };
|
|
838
940
|
if (agent)
|
|
839
941
|
body.agent = agent;
|
|
@@ -842,6 +944,9 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
|
|
|
842
944
|
if (tools)
|
|
843
945
|
body.tools = tools;
|
|
844
946
|
let result;
|
|
947
|
+
const steps = modelWork
|
|
948
|
+
? watchModelWorkSteps(client, sessionId, query, { setTimeoutFn: setTimeoutImpl, clearTimeoutFn: clearTimeoutImpl })
|
|
949
|
+
: undefined;
|
|
845
950
|
try {
|
|
846
951
|
const prompted = await raceSdkOperation(client.session.prompt({ path: { id: sessionId }, body, ...(query ? { query } : {}) }), {
|
|
847
952
|
timeoutMs: remainingTimeoutMs(),
|
|
@@ -852,6 +957,10 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
|
|
|
852
957
|
void deleteSessionBestEffort(client, sessionId, query, setTimeoutImpl, clearTimeoutImpl);
|
|
853
958
|
},
|
|
854
959
|
});
|
|
960
|
+
// A model-work session the dispatch stops early is aborted on the server too.
|
|
961
|
+
if (modelWork && (prompted === SDK_OPERATION_ABORTED || prompted === SDK_OPERATION_TIMED_OUT)) {
|
|
962
|
+
abortSessionBestEffort(client, sessionId, query);
|
|
963
|
+
}
|
|
855
964
|
if (prompted === SDK_OPERATION_ABORTED) {
|
|
856
965
|
result = {
|
|
857
966
|
ok: false,
|
|
@@ -878,21 +987,53 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
|
|
|
878
987
|
}
|
|
879
988
|
else {
|
|
880
989
|
const parts = prompted.data?.parts ?? [];
|
|
881
|
-
|
|
882
|
-
const stdout =
|
|
990
|
+
// The last text part is the answer; earlier ones narrate the steps before it.
|
|
991
|
+
const stdout = parts.filter((p) => p.type === "text").at(-1)?.text ?? "";
|
|
883
992
|
// Token accounting from the AssistantMessage (previously discarded) —
|
|
884
993
|
// the seam that makes workflow budget.maxTokens meterable on the
|
|
885
994
|
// default sdk runner.
|
|
886
995
|
const usage = extractUsage(prompted.data?.info);
|
|
887
|
-
|
|
888
|
-
|
|
889
|
-
|
|
890
|
-
|
|
891
|
-
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
|
|
895
|
-
|
|
996
|
+
const sdkError = prompted.error ?? prompted.data?.info?.error;
|
|
997
|
+
const overSteps = steps !== undefined && steps.steps() > MODEL_WORK_STEPS + MODEL_WORK_STEP_GRACE;
|
|
998
|
+
if (overSteps) {
|
|
999
|
+
const message = `opencode-sdk agent "${profile.name}" ran past the ${MODEL_WORK_STEPS}-step limit for model work; akm aborted the session.`;
|
|
1000
|
+
result = {
|
|
1001
|
+
ok: false,
|
|
1002
|
+
stdout,
|
|
1003
|
+
stderr: message,
|
|
1004
|
+
durationMs: Date.now() - start,
|
|
1005
|
+
exitCode: 1,
|
|
1006
|
+
reason: "parse_error",
|
|
1007
|
+
error: message,
|
|
1008
|
+
sessionId,
|
|
1009
|
+
...(usage ? { usage } : {}),
|
|
1010
|
+
};
|
|
1011
|
+
}
|
|
1012
|
+
else if (sdkError) {
|
|
1013
|
+
const failure = sdkErrorFailure(sdkError);
|
|
1014
|
+
result = {
|
|
1015
|
+
ok: false,
|
|
1016
|
+
stdout,
|
|
1017
|
+
stderr: failure.message,
|
|
1018
|
+
durationMs: Date.now() - start,
|
|
1019
|
+
exitCode: failure.reason === "aborted" ? null : 1,
|
|
1020
|
+
reason: failure.reason,
|
|
1021
|
+
error: failure.message,
|
|
1022
|
+
sessionId,
|
|
1023
|
+
...(usage ? { usage } : {}),
|
|
1024
|
+
};
|
|
1025
|
+
}
|
|
1026
|
+
else {
|
|
1027
|
+
result = {
|
|
1028
|
+
ok: true,
|
|
1029
|
+
stdout,
|
|
1030
|
+
stderr: "",
|
|
1031
|
+
durationMs: Date.now() - start,
|
|
1032
|
+
exitCode: 0,
|
|
1033
|
+
sessionId,
|
|
1034
|
+
...(usage ? { usage } : {}),
|
|
1035
|
+
};
|
|
1036
|
+
}
|
|
896
1037
|
}
|
|
897
1038
|
}
|
|
898
1039
|
catch (err) {
|
|
@@ -907,6 +1048,9 @@ export async function runOpencodeSdk(profile, prompt, opts = {}, llmConfig) {
|
|
|
907
1048
|
sessionId,
|
|
908
1049
|
};
|
|
909
1050
|
}
|
|
1051
|
+
finally {
|
|
1052
|
+
steps?.stop();
|
|
1053
|
+
}
|
|
910
1054
|
// Clean up session to prevent disk accumulation in ~/.local/share/opencode/.
|
|
911
1055
|
// Failures are non-fatal to the agent result but must not be invisible.
|
|
912
1056
|
const cleanupWarning = await deleteSessionBestEffort(client, sessionId, query, setTimeoutImpl, clearTimeoutImpl);
|
|
@@ -44,10 +44,8 @@
|
|
|
44
44
|
* - **schema** — the matrix places OpenHands in the "via prompt+validate"
|
|
45
45
|
* tier (plan §"Structured-output normalization", tier "native-json"): no
|
|
46
46
|
* Codex-style `--output-schema` flag exists, so NO temp schema file is
|
|
47
|
-
* written; the JSON Schema
|
|
48
|
-
*
|
|
49
|
-
* (`step-work.ts` `buildUnitPrompt`) and the pi/aider builders, so
|
|
50
|
-
* all dispatch paths speak one dialect. Downstream, the extractor pulls the
|
|
47
|
+
* written; the JSON Schema reaches it as the instruction the shared request
|
|
48
|
+
* lowering appends to the prompt. Downstream, the extractor pulls the
|
|
51
49
|
* final message out of the JSONL stream and the engine's shared
|
|
52
50
|
* retry-until-valid loop performs the actual validation.
|
|
53
51
|
* - **tools** — deliberately unconsumed. OpenHands has no per-tool allowlist
|
|
@@ -60,16 +58,15 @@
|
|
|
60
58
|
* conversation/session id opportunistically when the stream reveals one;
|
|
61
59
|
* akm's `workflow_run_units` remains the durable source of truth either way
|
|
62
60
|
* (plan §"Session, MCP, and identity across harnesses").
|
|
63
|
-
* - **
|
|
64
|
-
*
|
|
61
|
+
* - **inference** — not translated: the shared lowering reports each field of
|
|
62
|
+
* the request's inference as untranslated (`inference` in `harnesses/ids.ts`
|
|
63
|
+
* lists none for this harness).
|
|
65
64
|
*
|
|
66
65
|
* Registered: `openhandsBuilder` is `OpenhandsHarness.agentBuilder`
|
|
67
66
|
* (`./index.ts`), one of the ten harnesses `HARNESS_REGISTRY` constructs
|
|
68
67
|
* (`harnesses/index.ts`); `agent/builders.ts` derives `BUILTIN_BUILDERS` from
|
|
69
68
|
* that registry, so this builder is reachable under the `"openhands"`
|
|
70
|
-
* platform name without any further wiring.
|
|
71
|
-
* `pattern: "local-runner"`, `structuredOutput: "native-json"` alongside it
|
|
72
|
-
* (`./index.ts`).
|
|
69
|
+
* platform name without any further wiring.
|
|
73
70
|
*/
|
|
74
71
|
import { resolveDispatchModel } from "../../agent/builder-shared.js";
|
|
75
72
|
import { createAgentRequestLowerer } from "../../agent/request-lowering.js";
|
|
@@ -82,27 +79,18 @@ export const OPENHANDS_PLATFORM = "openhands";
|
|
|
82
79
|
* share the one constant.
|
|
83
80
|
*/
|
|
84
81
|
export const OPENHANDS_MODEL_ENV = "LLM_MODEL";
|
|
85
|
-
/**
|
|
86
|
-
* Assemble the `--task` payload: optional system text, the task prompt, and —
|
|
87
|
-
* when a schema is requested — the same schema directive the workflow
|
|
88
|
-
* engine's prompt assembly uses (OpenHands has no native schema flag, so the
|
|
89
|
-
* prompt is the schema's only channel; plan §"Structured-output
|
|
90
|
-
* normalization").
|
|
91
|
-
*/
|
|
82
|
+
/** Assemble the `--task` payload: optional system text, then the task prompt. */
|
|
92
83
|
function buildTaskPayload(req) {
|
|
93
84
|
const sections = [];
|
|
94
85
|
if (req.systemPrompt)
|
|
95
86
|
sections.push(req.systemPrompt);
|
|
96
87
|
sections.push(req.prompt);
|
|
97
|
-
if (req.schema) {
|
|
98
|
-
sections.push(`Respond with ONLY a JSON value matching this JSON Schema (no prose, no code fences):\n${JSON.stringify(req.schema)}`);
|
|
99
|
-
}
|
|
100
88
|
return sections.join("\n\n");
|
|
101
89
|
}
|
|
102
90
|
/**
|
|
103
91
|
* OpenHands builder.
|
|
104
92
|
* Command shape:
|
|
105
|
-
* openhands --headless --json --task=<[system\n\n]prompt
|
|
93
|
+
* openhands --headless --json --task=<[system\n\n]prompt>
|
|
106
94
|
* with the resolved model (if any) carried on env as LLM_MODEL.
|
|
107
95
|
*/
|
|
108
96
|
export const openhandsBuilder = {
|
|
@@ -112,7 +100,6 @@ export const openhandsBuilder = {
|
|
|
112
100
|
adapter: OPENHANDS_PLATFORM,
|
|
113
101
|
personaChannel: "prompt",
|
|
114
102
|
tools: "none",
|
|
115
|
-
outputSchema: true,
|
|
116
103
|
}),
|
|
117
104
|
build(profile, req) {
|
|
118
105
|
const args = [...profile.args];
|
|
@@ -29,11 +29,6 @@ export class OpenhandsHarness extends BaseHarness {
|
|
|
29
29
|
agentBuilder = openhandsBuilder;
|
|
30
30
|
resultExtractor = openhandsResultExtractor;
|
|
31
31
|
// ── Workflow-engine descriptor (plan §"Capability matrix", P2) ────────────
|
|
32
|
-
// akm spawns `openhands --headless` locally per unit ⇒ local-runner.
|
|
33
|
-
pattern = "local-runner";
|
|
34
|
-
// `--json` emits a documented JSONL event stream akm parses, then validates
|
|
35
|
-
// against the node schema ⇒ native-json tier.
|
|
36
|
-
structuredOutput = "native-json";
|
|
37
32
|
// No flag-shaped resume: per the matrix OpenHands resumes from workspace state, not a
|
|
38
33
|
// session-id flag. The extractor still captures a conversation id
|
|
39
34
|
// opportunistically; akm's `workflow_run_units` remains the durable source
|
|
@@ -27,10 +27,9 @@
|
|
|
27
27
|
* - **systemPrompt** — passed via `--system-prompt` (Pi follows the Claude
|
|
28
28
|
* Code flag conventions).
|
|
29
29
|
* - **schema** — the matrix places Pi in the "via prompt+validate" tier (no
|
|
30
|
-
* native `--output-schema` equivalent, unlike Codex), so the JSON Schema
|
|
31
|
-
*
|
|
32
|
-
*
|
|
33
|
-
* payload, and `--mode json` is emitted so stdout is the documented JSONL
|
|
30
|
+
* native `--output-schema` equivalent, unlike Codex), so the JSON Schema
|
|
31
|
+
* reaches it as the instruction the shared request lowering appends to the
|
|
32
|
+
* prompt, and `--mode json` is emitted so stdout is the documented JSONL
|
|
34
33
|
* event stream that `./result-extractor.ts` normalizes. The engine's shared
|
|
35
34
|
* retry-until-valid loop performs the actual validation. Without a schema
|
|
36
35
|
* the argv matches the matrix's bare headless shape (`pi -p "<p>"`) and the
|
|
@@ -40,8 +39,9 @@
|
|
|
40
39
|
* there is no documented per-tool allowlist flag, and inventing one would
|
|
41
40
|
* produce a silently broken command. A restrictive policy is therefore
|
|
42
41
|
* dropped rather than approximated — never silently widened.
|
|
43
|
-
* - **
|
|
44
|
-
*
|
|
42
|
+
* - **inference** — not translated: the shared lowering reports each field of
|
|
43
|
+
* the request's inference as untranslated (`inference` in `harnesses/ids.ts`
|
|
44
|
+
* lists none for this harness).
|
|
45
45
|
*
|
|
46
46
|
* Registered: `piBuilder` is `PiHarness.agentBuilder` (`./index.ts`), one of
|
|
47
47
|
* the ten harnesses `HARNESS_REGISTRY` constructs (`harnesses/index.ts`);
|
|
@@ -54,22 +54,10 @@ import { resolveDispatchModel } from "../../agent/builder-shared.js";
|
|
|
54
54
|
import { createAgentRequestLowerer } from "../../agent/request-lowering.js";
|
|
55
55
|
/** Canonical harness/platform id used for model-alias resolution. */
|
|
56
56
|
export const PI_PLATFORM = "pi";
|
|
57
|
-
/**
|
|
58
|
-
* Assemble the positional prompt payload: the task prompt and — when a schema
|
|
59
|
-
* is requested — the same schema directive the workflow engine's prompt
|
|
60
|
-
* assembly uses, so both dispatch paths speak one dialect.
|
|
61
|
-
*/
|
|
62
|
-
function buildPromptPayload(req) {
|
|
63
|
-
const sections = [req.prompt];
|
|
64
|
-
if (req.schema) {
|
|
65
|
-
sections.push(`Respond with ONLY a JSON value matching this JSON Schema (no prose, no code fences):\n${JSON.stringify(req.schema)}`);
|
|
66
|
-
}
|
|
67
|
-
return sections.join("\n\n");
|
|
68
|
-
}
|
|
69
57
|
/**
|
|
70
58
|
* Pi builder.
|
|
71
59
|
* Command shape:
|
|
72
|
-
* pi [--system-prompt "..."] [--model <m>] [--mode json] -p -- "<prompt
|
|
60
|
+
* pi [--system-prompt "..."] [--model <m>] [--mode json] -p -- "<prompt>"
|
|
73
61
|
*/
|
|
74
62
|
export const piBuilder = {
|
|
75
63
|
platform: PI_PLATFORM,
|
|
@@ -78,7 +66,6 @@ export const piBuilder = {
|
|
|
78
66
|
adapter: PI_PLATFORM,
|
|
79
67
|
personaChannel: "native",
|
|
80
68
|
tools: "none",
|
|
81
|
-
outputSchema: true,
|
|
82
69
|
}),
|
|
83
70
|
build(profile, req) {
|
|
84
71
|
const args = [...profile.args];
|
|
@@ -97,7 +84,7 @@ export const piBuilder = {
|
|
|
97
84
|
// -p = non-interactive print mode; prompt is the trailing positional.
|
|
98
85
|
args.push("-p");
|
|
99
86
|
args.push("--");
|
|
100
|
-
args.push(
|
|
87
|
+
args.push(req.prompt);
|
|
101
88
|
return { argv: [profile.bin, ...args] };
|
|
102
89
|
},
|
|
103
90
|
};
|
|
@@ -29,11 +29,6 @@ export class PiHarness extends BaseHarness {
|
|
|
29
29
|
agentBuilder = piBuilder;
|
|
30
30
|
resultExtractor = piResultExtractor;
|
|
31
31
|
// ── Workflow-engine descriptor (plan §"Capability matrix", P2) ────────────
|
|
32
|
-
// akm spawns the `pi` CLI locally per unit ⇒ local-runner.
|
|
33
|
-
pattern = "local-runner";
|
|
34
|
-
// `--mode json` emits a documented JSONL event stream akm parses, then
|
|
35
|
-
// validates against the node schema ⇒ native-json tier.
|
|
36
|
-
structuredOutput = "native-json";
|
|
37
32
|
// Session-id env marker only — the matrix's bare PI_* presence vars must
|
|
38
33
|
// not stamp identity onto manual runs (see `AkmHarness.identityEnv`).
|
|
39
34
|
identityEnv = ["PI_SESSION_ID"];
|
package/dist/llm/client.js
CHANGED
|
@@ -412,6 +412,11 @@ async function chatCompletionAttemptOnce(config, messages, options, timeoutMs, i
|
|
|
412
412
|
catch {
|
|
413
413
|
throw new LlmCallError(`LLM response was not valid JSON ${config.endpoint}: ${redactSensitiveText(redactErrorBody(rawOkBody), resolvedKey ? [resolvedKey] : [])}`, "parse_error", response.status);
|
|
414
414
|
}
|
|
415
|
+
// A 2xx can still carry a provider failure. OpenRouter answers a provider
|
|
416
|
+
// that fails after the headers with a body holding only `error`.
|
|
417
|
+
if (json.error !== undefined && json.error !== null && !json.choices?.length) {
|
|
418
|
+
throw new LlmCallError(`LLM provider error (${response.status}) ${config.endpoint}: ${redactSensitiveText(redactErrorBody(rawOkBody), resolvedKey ? [resolvedKey] : [])}`, "provider_error", response.status);
|
|
419
|
+
}
|
|
415
420
|
const responseModel = typeof json.model === "string" && json.model.trim().length > 0 ? json.model : undefined;
|
|
416
421
|
terminalFields = {
|
|
417
422
|
model: responseModel ?? config.model,
|
package/dist/llm/feature-gate.js
CHANGED
|
@@ -44,7 +44,8 @@ const DEFAULT_TIMEOUT_MS = 600_000;
|
|
|
44
44
|
/**
|
|
45
45
|
* Run `fn()` only if `isLlmFeatureEnabled(config, feature)` is `true`. On
|
|
46
46
|
* disablement, throw, or timeout, return `fallback` (or — if it is a
|
|
47
|
-
* thunk — the value produced by calling it).
|
|
47
|
+
* thunk — the value produced by calling it). The timeout aborts the signal
|
|
48
|
+
* `fn` receives, so the work it started stops too.
|
|
48
49
|
*/
|
|
49
50
|
export async function tryLlmFeature(feature, config, fn, fallback, opts) {
|
|
50
51
|
const resolveFallback = async () => typeof fallback === "function" ? await fallback() : fallback;
|
|
@@ -55,7 +56,7 @@ export async function tryLlmFeature(feature, config, fn, fallback, opts) {
|
|
|
55
56
|
const timeoutMs = opts && Object.hasOwn(opts, "timeoutMs") ? (opts.timeoutMs ?? null) : DEFAULT_TIMEOUT_MS;
|
|
56
57
|
try {
|
|
57
58
|
if (timeoutMs === null || timeoutMs <= 0) {
|
|
58
|
-
return await fn();
|
|
59
|
+
return await fn(new AbortController().signal);
|
|
59
60
|
}
|
|
60
61
|
return await runWithTimeout(fn, timeoutMs, feature);
|
|
61
62
|
}
|
|
@@ -97,11 +98,16 @@ export class LlmFeatureTimeoutError extends Error {
|
|
|
97
98
|
}
|
|
98
99
|
async function runWithTimeout(fn, timeoutMs, feature) {
|
|
99
100
|
let timer;
|
|
101
|
+
const controller = new AbortController();
|
|
100
102
|
try {
|
|
101
103
|
return await new Promise((resolve, reject) => {
|
|
102
|
-
timer = setTimeout(() =>
|
|
104
|
+
timer = setTimeout(() => {
|
|
105
|
+
const timedOut = new LlmFeatureTimeoutError(feature, timeoutMs);
|
|
106
|
+
controller.abort(timedOut);
|
|
107
|
+
reject(timedOut);
|
|
108
|
+
}, timeoutMs);
|
|
103
109
|
Promise.resolve()
|
|
104
|
-
.then(() => fn())
|
|
110
|
+
.then(() => fn(controller.signal))
|
|
105
111
|
.then(resolve, reject);
|
|
106
112
|
});
|
|
107
113
|
}
|