akm-cli 0.9.25-alpha.1 → 0.9.25-alpha.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +159 -280
- package/dist/cli.js +1 -1
- package/dist/commands/improve/consolidate/pair-pass.js +1 -0
- package/dist/commands/improve/consolidate.js +7 -2
- package/dist/commands/improve/execution.js +2 -3
- package/dist/commands/improve/extract.js +1 -0
- package/dist/commands/improve/improve-cli.js +33 -1
- package/dist/commands/improve/loop-stages.js +3 -0
- package/dist/commands/improve/reflect-noise.js +125 -0
- package/dist/commands/improve/reflect.js +13 -16
- package/dist/commands/improve/retrieval-gate.js +7 -2
- package/dist/commands/improve/stage.js +31 -39
- package/dist/commands/proposal/drain.js +4 -7
- package/dist/commands/proposal/propose.js +2 -11
- package/dist/commands/proposal/validators/proposal-quality-validators.js +4 -2
- package/dist/commands/read/search-cli.js +0 -38
- package/dist/core/config/schema/engines.js +15 -33
- package/dist/core/config/schema/improve-processes.js +16 -0
- package/dist/core/redaction.js +4 -0
- package/dist/core/spawn-env.js +25 -0
- package/dist/core/structured.js +1 -1
- package/dist/execution/source.js +8 -12
- package/dist/integrations/agent/config.js +1 -3
- package/dist/integrations/agent/engine-resolution.js +0 -3
- package/dist/integrations/agent/execution.js +14 -13
- package/dist/integrations/agent/index.js +1 -1
- package/dist/integrations/agent/model-map.js +15 -16
- package/dist/integrations/agent/profiles.js +2 -2
- package/dist/integrations/agent/prompts.js +3 -39
- package/dist/integrations/agent/request-lowering.js +9 -7
- package/dist/integrations/agent/runner-dispatch.js +25 -31
- package/dist/integrations/harnesses/aider/agent-builder.js +1 -2
- package/dist/integrations/harnesses/amazonq/agent-builder.js +1 -2
- package/dist/integrations/harnesses/claude/agent-builder.js +4 -16
- package/dist/integrations/harnesses/codex/agent-builder.js +1 -2
- package/dist/integrations/harnesses/ids.js +10 -16
- package/dist/integrations/harnesses/opencode/agent-builder.js +14 -24
- package/dist/integrations/harnesses/opencode/model-config.js +15 -62
- package/dist/integrations/harnesses/opencode/model-work-agent.js +71 -36
- package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -6
- package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +32 -90
- package/dist/integrations/harnesses/openhands/agent-builder.js +1 -2
- package/dist/integrations/harnesses/pi/agent-builder.js +1 -2
- package/dist/llm/feature-gate.js +2 -5
- package/dist/llm/index-passes.js +2 -2
- package/dist/llm/structured-call.js +5 -5
- package/dist/output/shapes/passthrough.js +1 -0
- package/dist/scripts/akm-migrate-node.js +170 -194
- package/dist/scripts/akm-migrate.js +170 -194
- package/dist/workflows/exec/unit-dispatch.js +4 -13
- package/docs/reference/cli.md +12 -7
- package/docs/reference/configuration.md +78 -82
- package/docs/reference/data-and-telemetry.md +2 -3
- package/docs/reference/workflow-schema.md +6 -9
- package/package.json +1 -1
- package/schemas/akm-config.json +108 -36
|
@@ -103,6 +103,21 @@ const fidelityCheckField = z.object({ enabled: z.boolean().optional() }).passthr
|
|
|
103
103
|
* byte-identical behaviour). Reflect process only.
|
|
104
104
|
*/
|
|
105
105
|
const lowValueFilterField = z.object({ enabled: z.boolean().optional() }).passthrough().optional();
|
|
106
|
+
/**
|
|
107
|
+
* The wording lists of reflect's pre-judge defect filter (`findReflectDefect`).
|
|
108
|
+
* Each list is optional: one that is set replaces that rule's default list, and
|
|
109
|
+
* an empty one turns the rule off. Phrases (`placeholders`, `metaCommentary`)
|
|
110
|
+
* match as whole words in any case; `frontmatterKeys` are exact key names.
|
|
111
|
+
* Reflect process only.
|
|
112
|
+
*/
|
|
113
|
+
const defectFilterField = z
|
|
114
|
+
.object({
|
|
115
|
+
placeholders: z.array(nonEmptyString).optional(),
|
|
116
|
+
metaCommentary: z.array(nonEmptyString).optional(),
|
|
117
|
+
frontmatterKeys: z.array(nonEmptyString).optional(),
|
|
118
|
+
})
|
|
119
|
+
.passthrough()
|
|
120
|
+
.optional();
|
|
106
121
|
/**
|
|
107
122
|
* #626 — extract process: pre-LLM heuristic triage gate. When enabled, a
|
|
108
123
|
* deterministic scorer decides BEFORE the extraction LLM call whether a
|
|
@@ -159,6 +174,7 @@ const REFLECT_PROCESS_FIELDS = {
|
|
|
159
174
|
limit: processLimitField,
|
|
160
175
|
qualityGate: qualityGateField,
|
|
161
176
|
lowValueFilter: lowValueFilterField,
|
|
177
|
+
defectFilter: defectFilterField,
|
|
162
178
|
};
|
|
163
179
|
const DISTILL_PROCESS_FIELDS = {
|
|
164
180
|
allowedTypes: allowedTypesField,
|
package/dist/core/redaction.js
CHANGED
|
@@ -19,6 +19,10 @@ const ENV_PASSTHROUGH_REDACTION_POLICY = {
|
|
|
19
19
|
OPENCODE_CONFIG: "path",
|
|
20
20
|
CLAUDE_CONFIG: "path",
|
|
21
21
|
CODEX_CONFIG: "path",
|
|
22
|
+
XDG_CONFIG_HOME: "path",
|
|
23
|
+
XDG_DATA_HOME: "path",
|
|
24
|
+
XDG_CACHE_HOME: "path",
|
|
25
|
+
XDG_STATE_HOME: "path",
|
|
22
26
|
AWS_PROFILE: "identifier",
|
|
23
27
|
AWS_REGION: "identifier",
|
|
24
28
|
LLM_MODEL: "identifier",
|
package/dist/core/spawn-env.js
CHANGED
|
@@ -45,6 +45,31 @@ export const COMMON_SPAWN_ENV_PASSTHROUGH = [
|
|
|
45
45
|
"TMPDIR",
|
|
46
46
|
"AKM_EVENT_SOURCE",
|
|
47
47
|
];
|
|
48
|
+
/**
|
|
49
|
+
* The XDG base-directory variables. opencode resolves its config, data, cache
|
|
50
|
+
* and state directories from them, so it must receive them: an akm that runs
|
|
51
|
+
* under a custom `XDG_CONFIG_HOME` otherwise spawns an opencode that reads
|
|
52
|
+
* `$HOME/.config/opencode` and misses the provider config its caller named.
|
|
53
|
+
*
|
|
54
|
+
* Deliberately NOT part of {@link COMMON_SPAWN_ENV_PASSTHROUGH}, which is every
|
|
55
|
+
* harness's baseline, the workflow exec unit's default allowlist (a documented
|
|
56
|
+
* list) and, through profile `envPassthrough`, frozen into workflow plans.
|
|
57
|
+
* codex, gemini and pi keep their own dotdirs under `$HOME`, and handing the
|
|
58
|
+
* names to a shell command would redirect the `git` and `gh` config it reads. A
|
|
59
|
+
* harness that reads them asks for them by name: the opencode profile's list,
|
|
60
|
+
* and the opencode-sdk server's allowlist (`opencodeSdkServerEnvironmentNames`).
|
|
61
|
+
*
|
|
62
|
+
* A name added to a profile's list changes the plans frozen after it (their
|
|
63
|
+
* bytes, so their `plan_hash`); a stored plan keeps the list it was frozen with
|
|
64
|
+
* and still resumes, because nothing gates on that hash. The SDK server's
|
|
65
|
+
* allowlist is not part of a plan, so it takes the names by code.
|
|
66
|
+
*/
|
|
67
|
+
export const XDG_BASE_DIR_ENV_PASSTHROUGH = [
|
|
68
|
+
"XDG_CONFIG_HOME",
|
|
69
|
+
"XDG_DATA_HOME",
|
|
70
|
+
"XDG_CACHE_HOME",
|
|
71
|
+
"XDG_STATE_HOME",
|
|
72
|
+
];
|
|
48
73
|
/**
|
|
49
74
|
* The names Windows itself requires of ANY child, whatever the caller's
|
|
50
75
|
* allowlist says. Applied at build time rather than added to
|
package/dist/core/structured.js
CHANGED
|
@@ -34,7 +34,7 @@ import { parseEmbeddedJsonResponse } from "./parse.js";
|
|
|
34
34
|
export function withSchemaInstruction(prompt, schema) {
|
|
35
35
|
return `${prompt}\n\nRespond with ONLY a JSON value matching this JSON Schema (no prose, no code fences):\n${JSON.stringify(schema)}`;
|
|
36
36
|
}
|
|
37
|
-
function defaultFeedback(failure) {
|
|
37
|
+
export function defaultFeedback(failure) {
|
|
38
38
|
if (failure.reason === "parse_error") {
|
|
39
39
|
return "Your previous response contained no parseable JSON. Respond with ONLY a JSON value that matches the requested schema — no prose, no code fences.";
|
|
40
40
|
}
|
package/dist/execution/source.js
CHANGED
|
@@ -6,19 +6,15 @@ export const EXECUTION_SOURCE_SCHEMA_VERSION = 1;
|
|
|
6
6
|
/** Current internal adapter identifiers are lowercase kebab-case registry keys. */
|
|
7
7
|
export const EXECUTION_ADAPTER_ID_PATTERN = /^[a-z][a-z0-9]*(?:-[a-z0-9]+)*$/;
|
|
8
8
|
/**
|
|
9
|
-
* The model-work tool policy: unattended model work (improve, the judges,
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
9
|
+
* The model-work tool policy: unattended model work (improve, the judges, index
|
|
10
|
+
* passes, remember) may read, edit only inside the dispatch's own scratch
|
|
11
|
+
* working directory, and run `akm search` and `akm show`; the stash stays
|
|
12
|
+
* read-only to it. A transport grants what it can confine and refuses the policy
|
|
13
|
+
* at build when it can confine nothing; an LLM has no tools. A request carries it
|
|
14
|
+
* as `authorization.policy.id`, set only when its caller asks (`modelWork`): no
|
|
15
|
+
* `tools` value names it, so an asset's own `tools:` cannot.
|
|
14
16
|
*/
|
|
15
|
-
export const
|
|
16
|
-
/** Whether a selection names the model-work tool policy. */
|
|
17
|
-
export function isModelWorkTools(tools) {
|
|
18
|
-
return (Array.isArray(tools) &&
|
|
19
|
-
tools.length === MODEL_WORK_TOOLS.length &&
|
|
20
|
-
tools.every((tool, index) => tool === MODEL_WORK_TOOLS[index]));
|
|
21
|
-
}
|
|
17
|
+
export const MODEL_WORK_POLICY_ID = "model-work";
|
|
22
18
|
function requireRecord(value, path) {
|
|
23
19
|
if (value === null || typeof value !== "object" || Array.isArray(value)) {
|
|
24
20
|
throw new TypeError(`${path} must be an object`);
|
|
@@ -3,7 +3,5 @@
|
|
|
3
3
|
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
4
|
/** Default agent CLI timeout; null means agents run until they finish. */
|
|
5
5
|
export const DEFAULT_AGENT_TIMEOUT_MS = null;
|
|
6
|
-
/** Default hard timeout for direct LLM
|
|
6
|
+
/** Default hard timeout for a direct LLM call, and the bound on model work on every runner kind, when no engine/use override exists. */
|
|
7
7
|
export const DEFAULT_LLM_TIMEOUT_MS = 600_000;
|
|
8
|
-
/** Default bound on model work (structured calls, judgments) on every runner kind. */
|
|
9
|
-
export const DEFAULT_MODEL_WORK_TIMEOUT_MS = 600_000;
|
|
@@ -13,7 +13,6 @@ import { collectSensitiveValues } from "../../core/redaction.js";
|
|
|
13
13
|
import { resolveSecretFromStore } from "../../sources/snapshot-fetchers/secret-seam.js";
|
|
14
14
|
import { getHarness } from "../harnesses/index.js";
|
|
15
15
|
import { DEFAULT_LLM_TIMEOUT_MS } from "./config.js";
|
|
16
|
-
import { engineModelAndInference } from "./model-map.js";
|
|
17
16
|
import { getBuiltinAgentProfile, OPENCODE_SDK_SERVER_BIN } from "./profiles.js";
|
|
18
17
|
const LLM_CONNECTION_FIELDS = [
|
|
19
18
|
"provider",
|
|
@@ -274,7 +273,6 @@ function lowerAgentEngine(name, engine, config) {
|
|
|
274
273
|
const platform = harness.id;
|
|
275
274
|
const sdk = platform === "opencode-sdk";
|
|
276
275
|
const builtin = getBuiltinAgentProfile(platform);
|
|
277
|
-
const { inference } = engineModelAndInference(engine);
|
|
278
276
|
const profile = {
|
|
279
277
|
name,
|
|
280
278
|
platform,
|
|
@@ -287,7 +285,6 @@ function lowerAgentEngine(name, engine, config) {
|
|
|
287
285
|
parseOutput: "text",
|
|
288
286
|
...(engine.workspace ? { workspace: path.resolve(engine.workspace) } : {}),
|
|
289
287
|
...(engine.model ? { model: engine.model } : {}),
|
|
290
|
-
...(inference ? { inference } : {}),
|
|
291
288
|
};
|
|
292
289
|
// An engine that sets no timeoutMs leaves it unset, so a caller's own default
|
|
293
290
|
// (model work's 600 s) can apply; a dispatch with none runs unbounded.
|
|
@@ -6,7 +6,7 @@ import { ConfigError } from "../../core/errors.js";
|
|
|
6
6
|
import { DURATION_UNITS, parseDuration } from "../../core/time.js";
|
|
7
7
|
import { EXECUTION_MAX_TIMEOUT_MS } from "../../execution/limits.js";
|
|
8
8
|
import { createInlineResolvedCommand, createResolvedExecutionRequest, decodeResolvedExecutionRequest, } from "../../execution/resolved-request.js";
|
|
9
|
-
import { cloneToolSelection,
|
|
9
|
+
import { cloneToolSelection, isPortableExecutionAgentSelector, MODEL_WORK_POLICY_ID, } from "../../execution/source.js";
|
|
10
10
|
import { getHarness } from "../harnesses/index.js";
|
|
11
11
|
import { DEFAULT_LLM_TIMEOUT_MS } from "./config.js";
|
|
12
12
|
import { FALLBACK_ENGINE_NAME, fallbackEngineConfig, NO_ENGINE_MESSAGE_SUFFIX, NO_ENGINE_REMEDY, } from "./engine-fallback.js";
|
|
@@ -78,11 +78,8 @@ function engineDefaults(name, engine, config) {
|
|
|
78
78
|
modelMapKey: own.model === undefined ? fallbackName : "opencode-sdk",
|
|
79
79
|
values: {
|
|
80
80
|
...(inherited.model !== undefined ? { model: inherited.model } : {}),
|
|
81
|
+
...(inherited.inference !== undefined ? { inference: inherited.inference } : {}),
|
|
81
82
|
...values,
|
|
82
|
-
// The engine's own inference is added over the fallback's, field by field.
|
|
83
|
-
...(inherited.inference !== undefined || own.inference !== undefined
|
|
84
|
-
? { inference: { ...inherited.inference, ...own.inference } }
|
|
85
|
-
: {}),
|
|
86
83
|
timeout: Object.hasOwn(engine, "timeoutMs")
|
|
87
84
|
? (engine.timeoutMs ?? null)
|
|
88
85
|
: Object.hasOwn(fallback, "timeoutMs")
|
|
@@ -191,12 +188,16 @@ function requestedToolNames(tools) {
|
|
|
191
188
|
* nothing is allowed. The model-work policy is akm's own and always allowed:
|
|
192
189
|
* it confines an engine more tightly than leaving tools unset does.
|
|
193
190
|
*/
|
|
194
|
-
function authorizeTools(tools, config) {
|
|
191
|
+
function authorizeTools(tools, config, modelWork) {
|
|
192
|
+
if (modelWork) {
|
|
193
|
+
return {
|
|
194
|
+
status: "allowed",
|
|
195
|
+
reason: "The model-work tool policy is akm's own.",
|
|
196
|
+
policy: { id: MODEL_WORK_POLICY_ID },
|
|
197
|
+
};
|
|
198
|
+
}
|
|
195
199
|
if (!hasToolSelection(tools))
|
|
196
200
|
return { status: "not-required" };
|
|
197
|
-
if (isModelWorkTools(tools)) {
|
|
198
|
-
return { status: "allowed", reason: "The model-work tool policy is akm's own.", policy: { id: "model-work" } };
|
|
199
|
-
}
|
|
200
201
|
if (!config) {
|
|
201
202
|
return {
|
|
202
203
|
status: "denied",
|
|
@@ -392,13 +393,14 @@ export function resolveExecution(input) {
|
|
|
392
393
|
return layer;
|
|
393
394
|
};
|
|
394
395
|
const schemaLayer = select("outputSchema", "outputSchema");
|
|
395
|
-
const
|
|
396
|
+
const modelWork = input.modelWork === true;
|
|
397
|
+
const toolsLayer = modelWork ? undefined : select("tools", "tools");
|
|
396
398
|
const timeoutLayer = select("timeout", "runtime.timeoutMs");
|
|
397
399
|
const workspaceLayer = select("workspace", "runtime.workspace");
|
|
398
400
|
const environmentLayer = select("environment", "runtime.environment");
|
|
399
401
|
const settingsLayer = select("runtime", "runtime.settings");
|
|
400
402
|
const tools = toolsLayer ? cloneToolSelection(toolsLayer.values.tools ?? null, "tools") : undefined;
|
|
401
|
-
const authorization = authorizeTools(tools, input.runner ? undefined : input.config);
|
|
403
|
+
const authorization = authorizeTools(tools, input.runner ? undefined : input.config, modelWork);
|
|
402
404
|
provenance.authorization = {
|
|
403
405
|
layer: typeof authorization.policy?.id === "string" ? authorization.policy.id : "not-required",
|
|
404
406
|
kind: "authorization",
|
|
@@ -438,8 +440,7 @@ function buildLlm(request, runner, options) {
|
|
|
438
440
|
if (typeof request.agent === "string" && request.persona === null) {
|
|
439
441
|
throw new ConfigError(`The direct LLM transport cannot consume native agent selector ${JSON.stringify(request.agent)}.`, "INVALID_CONFIG_FILE");
|
|
440
442
|
}
|
|
441
|
-
|
|
442
|
-
if (hasToolSelection(request.tools) && !isModelWorkTools(request.tools)) {
|
|
443
|
+
if (hasToolSelection(request.tools)) {
|
|
443
444
|
throw new ConfigError("The direct LLM transport cannot enforce the resolved tool policy.", "INVALID_CONFIG_FILE");
|
|
444
445
|
}
|
|
445
446
|
const notices = [...request.notices];
|
|
@@ -4,4 +4,4 @@
|
|
|
4
4
|
export { DEFAULT_AGENT_TIMEOUT_MS } from "./config.js";
|
|
5
5
|
export { _setAgentDetectForTests, defaultWhich, detectAgentCliProfiles, pickDefaultAgentProfile } from "./detect.js";
|
|
6
6
|
export { BUILTIN_AGENT_PROFILE_NAMES, getBuiltinAgentProfile, listBuiltinAgentProfiles, } from "./profiles.js";
|
|
7
|
-
export { buildProposePrompt, buildReflectPrompt,
|
|
7
|
+
export { buildProposePrompt, buildReflectPrompt, parseAgentProposalPayload } from "./prompts.js";
|
|
@@ -220,33 +220,32 @@ function mergeProfiles(base, overlay) {
|
|
|
220
220
|
* copied verbatim from the engine's own config value; it must already be
|
|
221
221
|
* meaningful for the model-map column's platform (akm does not translate
|
|
222
222
|
* between an engine's connection and an agent platform's own provider
|
|
223
|
-
* registry).
|
|
224
|
-
*
|
|
225
|
-
* also contributes `supportsJsonSchema` and `extraParams`.
|
|
223
|
+
* registry). Only `kind: "llm"` engines contribute inference defaults — an
|
|
224
|
+
* agent-kind engine's schema carries no temperature/thinking fields.
|
|
226
225
|
*/
|
|
227
226
|
export function engineModelAndInference(engine) {
|
|
228
227
|
const out = {};
|
|
229
228
|
if (Object.hasOwn(engine, "model") && engine.model !== undefined)
|
|
230
229
|
out.model = engine.model;
|
|
231
|
-
const inference = {};
|
|
232
|
-
if (Object.hasOwn(engine, "temperature"))
|
|
233
|
-
inference.temperature = engine.temperature;
|
|
234
|
-
if (Object.hasOwn(engine, "maxTokens"))
|
|
235
|
-
inference.maxTokens = engine.maxTokens;
|
|
236
230
|
if (engine.kind === "llm") {
|
|
231
|
+
const inference = {};
|
|
232
|
+
if (Object.hasOwn(engine, "temperature"))
|
|
233
|
+
inference.temperature = engine.temperature;
|
|
234
|
+
if (Object.hasOwn(engine, "maxTokens"))
|
|
235
|
+
inference.maxTokens = engine.maxTokens;
|
|
237
236
|
if (Object.hasOwn(engine, "supportsJsonSchema"))
|
|
238
237
|
inference.supportsJsonSchema = engine.supportsJsonSchema;
|
|
239
238
|
if (Object.hasOwn(engine, "extraParams"))
|
|
240
239
|
inference.extraParams = engine.extraParams;
|
|
240
|
+
if (Object.hasOwn(engine, "contextLength"))
|
|
241
|
+
inference.contextLength = engine.contextLength;
|
|
242
|
+
if (Object.hasOwn(engine, "enableThinking"))
|
|
243
|
+
inference.enableThinking = engine.enableThinking;
|
|
244
|
+
if (Object.hasOwn(engine, "reasoningEffort"))
|
|
245
|
+
inference.reasoningEffort = engine.reasoningEffort;
|
|
246
|
+
if (Object.keys(inference).length > 0)
|
|
247
|
+
out.inference = inference;
|
|
241
248
|
}
|
|
242
|
-
if (Object.hasOwn(engine, "contextLength"))
|
|
243
|
-
inference.contextLength = engine.contextLength;
|
|
244
|
-
if (Object.hasOwn(engine, "enableThinking"))
|
|
245
|
-
inference.enableThinking = engine.enableThinking;
|
|
246
|
-
if (Object.hasOwn(engine, "reasoningEffort"))
|
|
247
|
-
inference.reasoningEffort = engine.reasoningEffort;
|
|
248
|
-
if (Object.keys(inference).length > 0)
|
|
249
|
-
out.inference = inference;
|
|
250
249
|
return Object.freeze(out);
|
|
251
250
|
}
|
|
252
251
|
/** Overlay user fields over installed fields, per (alias, column), without resolving `engine` indirection. */
|
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
* coding-agent CLI. Named engines lower canonical harness metadata into this
|
|
9
9
|
* intentionally small internal shape. The wrapper is in `./spawn.ts`.
|
|
10
10
|
*/
|
|
11
|
-
import { COMMON_SPAWN_ENV_PASSTHROUGH } from "../../core/spawn-env.js";
|
|
11
|
+
import { COMMON_SPAWN_ENV_PASSTHROUGH, XDG_BASE_DIR_ENV_PASSTHROUGH } from "../../core/spawn-env.js";
|
|
12
12
|
// AKM_EVENT_SOURCE carries usage-event provenance (improve/task) so that akm
|
|
13
13
|
// invocations a spawned agent makes are recorded as machine traffic, not user
|
|
14
14
|
// demand (DRIFT-6). Without it in the passthrough whitelist, buildChildEnv drops
|
|
@@ -30,7 +30,7 @@ const BUILTINS = {
|
|
|
30
30
|
bin: "opencode",
|
|
31
31
|
args: ["run"],
|
|
32
32
|
stdio: "interactive",
|
|
33
|
-
envPassthrough: [...COMMON_PASSTHROUGH, "OPENCODE_API_KEY", "OPENCODE_CONFIG"],
|
|
33
|
+
envPassthrough: [...COMMON_PASSTHROUGH, "OPENCODE_API_KEY", "OPENCODE_CONFIG", ...XDG_BASE_DIR_ENV_PASSTHROUGH],
|
|
34
34
|
parseOutput: "text",
|
|
35
35
|
},
|
|
36
36
|
claude: {
|
|
@@ -341,7 +341,8 @@ export function buildReflectPrompt(input) {
|
|
|
341
341
|
// Embed concrete counts only when the gate will actually fire (source >= 200 chars).
|
|
342
342
|
const showCharBounds = sourceBodyLen >= 200;
|
|
343
343
|
const minChars = Math.max(Math.round(0.5 * sourceBodyLen), 150);
|
|
344
|
-
|
|
344
|
+
// A source already past the 25000 cap may not grow, and need not shrink to the cap.
|
|
345
|
+
const maxChars = Math.max(Math.min(Math.max(Math.round(2.5 * sourceBodyLen), 2500), 25000), sourceBodyLen);
|
|
345
346
|
sections.push([
|
|
346
347
|
"## Content preservation rules (MUST follow)",
|
|
347
348
|
"1. PRESERVE ALL concrete content: code blocks, fenced snippets, CLI commands, numbered/bulleted checklists, tables, YAML/JSON examples, file paths, configuration keys, environment variable names, and CSS/HTML selectors. These are load-bearing — do NOT replace them with prose summaries.",
|
|
@@ -350,7 +351,7 @@ export function buildReflectPrompt(input) {
|
|
|
350
351
|
? `3. DO NOT shrink the asset. Your body must be at least ${minChars} characters (source body is ${sourceBodyLen} chars; floor is 50%). If you genuinely need to remove a major section, explain why in a comment line at the top of the body (e.g. \`<!-- removed obsolete section X because ... -->\`).`
|
|
351
352
|
: "3. DO NOT shrink the asset dramatically. The improved body must be at least 50% of the source body length. If you genuinely need to remove a major section, explain why in a comment line at the top of the body (e.g. `<!-- removed obsolete section X because ... -->`).",
|
|
352
353
|
showCharBounds
|
|
353
|
-
? `4. DO NOT pad the asset with speculative material. Your body must be at most ${maxChars} characters (source body is ${sourceBodyLen} chars; ceiling is 250%). Do not add invented sections, hypothetical examples, or padding prose.`
|
|
354
|
+
? `4. DO NOT pad the asset with speculative material. Your body must be at most ${maxChars} characters (source body is ${sourceBodyLen} chars; ceiling is ${maxChars === sourceBodyLen ? "100%" : "250%"}). Do not add invented sections, hypothetical examples, or padding prose.`
|
|
354
355
|
: "4. DO NOT pad the asset with speculative material. The improved body must be at most 250% of the source body length unless the feedback explicitly requests added sections.",
|
|
355
356
|
"5. Improve clarity of surrounding prose, fix structural issues, add missing required frontmatter fields. Do NOT rewrite a runbook into an essay.",
|
|
356
357
|
].join("\n"));
|
|
@@ -388,43 +389,6 @@ export function buildProposePrompt(input) {
|
|
|
388
389
|
sections.push(RESPONSE_CONTRACT_JSON);
|
|
389
390
|
return sections.join("\n\n");
|
|
390
391
|
}
|
|
391
|
-
/**
|
|
392
|
-
* Build the prompt for the schema repair pass in `akm improve`. Asks the
|
|
393
|
-
* agent to add the minimal required frontmatter to an asset that failed
|
|
394
|
-
* validation — without rewriting the body.
|
|
395
|
-
*/
|
|
396
|
-
export function buildSchemaRepairPrompt(input) {
|
|
397
|
-
const sections = [];
|
|
398
|
-
sections.push(`This ${input.type} asset failed schema validation with the error: "${input.reason}". ` +
|
|
399
|
-
`Your task is to fix the schema issue by adding or correcting the missing/invalid field(s) ` +
|
|
400
|
-
`while preserving all existing content.`);
|
|
401
|
-
sections.push(`Target ref: ${input.ref}`);
|
|
402
|
-
sections.push(`Schema requirements for ${input.type} assets: ${hintForType(input.type)}`);
|
|
403
|
-
if (input.standardsContext?.trim()) {
|
|
404
|
-
sections.push("Standards to follow (the rulebook for this target):");
|
|
405
|
-
sections.push(input.standardsContext.trim());
|
|
406
|
-
}
|
|
407
|
-
{
|
|
408
|
-
const authoringRules = authoringRulesForType(input.type);
|
|
409
|
-
if (authoringRules) {
|
|
410
|
-
sections.push(authoringRules);
|
|
411
|
-
}
|
|
412
|
-
}
|
|
413
|
-
const CONTENT_CAP = 3000;
|
|
414
|
-
const body = input.assetContent.trimEnd();
|
|
415
|
-
const truncated = body.length > CONTENT_CAP;
|
|
416
|
-
sections.push("Current asset content (first 3000 chars — sufficient to generate missing frontmatter):");
|
|
417
|
-
sections.push("```");
|
|
418
|
-
sections.push(truncated ? `${body.slice(0, CONTENT_CAP)}\n... [truncated]` : body);
|
|
419
|
-
sections.push("```");
|
|
420
|
-
sections.push("Produce the minimal fix: add ONLY the missing required frontmatter field(s). " +
|
|
421
|
-
"Do not rewrite the body unless it is empty. " +
|
|
422
|
-
"If `description` is missing, generate a concise one-sentence description from the content. " +
|
|
423
|
-
"If `when_to_use` is missing, generate a one-line trigger sentence. " +
|
|
424
|
-
"Preserve all existing frontmatter keys and the full body verbatim.");
|
|
425
|
-
sections.push(RESPONSE_CONTRACT_JSON);
|
|
426
|
-
return sections.join("\n\n");
|
|
427
|
-
}
|
|
428
392
|
/**
|
|
429
393
|
* Parse agent stdout into a proposal payload. The agent contract requires a
|
|
430
394
|
* single JSON object; anything else is reported as a parse error so callers
|
|
@@ -3,7 +3,8 @@
|
|
|
3
3
|
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
4
|
import { ConfigError } from "../../core/errors.js";
|
|
5
5
|
import { withSchemaInstruction } from "../../core/structured.js";
|
|
6
|
-
import {
|
|
6
|
+
import { MODEL_WORK_POLICY_ID } from "../../execution/source.js";
|
|
7
|
+
import { HARNESS_MODEL_WORK_IDS } from "../harnesses/ids.js";
|
|
7
8
|
import { composeConversationFallbackPrompt } from "./conversation-fallback.js";
|
|
8
9
|
import { composePersonaFallbackPrompt } from "./persona-fallback.js";
|
|
9
10
|
/** A tool selection that actually names tools (not omitted, null, or empty). */
|
|
@@ -47,8 +48,9 @@ export function extensionFields(request) {
|
|
|
47
48
|
* composing into the prompt what the harness has no channel for.
|
|
48
49
|
*/
|
|
49
50
|
export function createAgentRequestLowerer(options) {
|
|
50
|
-
|
|
51
|
-
|
|
51
|
+
const supportedInference = new Set(options.inference ?? []);
|
|
52
|
+
return (_profile, request) => {
|
|
53
|
+
const modelWork = request.authorization.policy?.id === MODEL_WORK_POLICY_ID;
|
|
52
54
|
const notices = [];
|
|
53
55
|
const skip = (field) => {
|
|
54
56
|
notices.push(untranslated(options.adapter, field));
|
|
@@ -92,19 +94,19 @@ export function createAgentRequestLowerer(options) {
|
|
|
92
94
|
if (Object.hasOwn(request, "inference")) {
|
|
93
95
|
dispatch.inference = request.inference ?? null;
|
|
94
96
|
for (const key of Object.keys(request.inference ?? {}).sort()) {
|
|
95
|
-
if (!supportedInference.has(key))
|
|
97
|
+
if (!(modelWork && supportedInference.has(key)))
|
|
96
98
|
skip(`inference.${key}`);
|
|
97
99
|
}
|
|
98
100
|
}
|
|
99
|
-
if (
|
|
100
|
-
if (!options.
|
|
101
|
+
if (modelWork) {
|
|
102
|
+
if (!HARNESS_MODEL_WORK_IDS.has(options.adapter)) {
|
|
101
103
|
throw new ConfigError(`The ${options.adapter} transport cannot enforce the model-work tool policy.`, "INVALID_CONFIG_FILE");
|
|
102
104
|
}
|
|
103
105
|
// The builder selects its own confined agent or flags; another agent would replace them.
|
|
104
106
|
if (dispatch.agent) {
|
|
105
107
|
throw new ConfigError(`The ${options.adapter} transport cannot run native agent ${JSON.stringify(dispatch.agent)} under the model-work tool policy.`, "INVALID_CONFIG_FILE");
|
|
106
108
|
}
|
|
107
|
-
dispatch.
|
|
109
|
+
dispatch.modelWork = true;
|
|
108
110
|
}
|
|
109
111
|
else if (request.tools !== undefined) {
|
|
110
112
|
// An explicit empty selection still reaches the builder (e.g. an empty allowlist).
|
|
@@ -12,13 +12,14 @@ import fs from "node:fs";
|
|
|
12
12
|
import os from "node:os";
|
|
13
13
|
import path from "node:path";
|
|
14
14
|
import { assertNever } from "../../core/assert.js";
|
|
15
|
-
import {
|
|
15
|
+
import { UsageError } from "../../core/errors.js";
|
|
16
16
|
import { collectSensitiveValues, isEnvPassthroughValueSafeToExpose, redactSensitiveText, redactSensitiveValue, } from "../../core/redaction.js";
|
|
17
|
-
import {
|
|
17
|
+
import { MODEL_WORK_POLICY_ID } from "../../execution/source.js";
|
|
18
18
|
import { chatCompletion, LlmCallError } from "../../llm/client.js";
|
|
19
19
|
import { emitLlmUsage } from "../../llm/usage-telemetry.js";
|
|
20
20
|
import { getHarness } from "../harnesses/index.js";
|
|
21
21
|
import { closeServer as disposeOpencodeSdkServers, runOpencodeSdk } from "../harnesses/opencode-sdk/sdk-runner.js";
|
|
22
|
+
import { modelFromArgs } from "./builder-shared.js";
|
|
22
23
|
import { lookupApiKeyFileValue, lookupApiKeySecretRefValue, lookupCredentialFromEnv, resolveEngine, } from "./engine-resolution.js";
|
|
23
24
|
import { materializeLlmRunnerConnection, materializeSdkFallbackConnection } from "./runner.js";
|
|
24
25
|
import { runAgent } from "./spawn.js";
|
|
@@ -86,18 +87,28 @@ const USAGE_ERROR_CODES = {
|
|
|
86
87
|
parse_error: "parse_error",
|
|
87
88
|
llm_rate_limit: "rate_limited",
|
|
88
89
|
};
|
|
90
|
+
/**
|
|
91
|
+
* The model an agent or SDK dispatch ran: the request's, else the one an agent
|
|
92
|
+
* CLI's own `args` select, which its command carries when the request names
|
|
93
|
+
* none. An SDK server never sees `args`, so an SDK engine with no model named
|
|
94
|
+
* has none to report: opencode picks it.
|
|
95
|
+
*/
|
|
96
|
+
function dispatchedModel({ request, runner }) {
|
|
97
|
+
return request.model?.resolved ?? (runner.kind === "agent" ? modelFromArgs(runner.profile.args) : undefined);
|
|
98
|
+
}
|
|
89
99
|
/**
|
|
90
100
|
* One usage record for an agent or SDK dispatch, through the same sink and
|
|
91
101
|
* ambient `withLlmStage` attribution as the LLM transport's per-HTTP-attempt
|
|
92
|
-
* records: the
|
|
102
|
+
* records: the model it ran, and tokens when the runner reported them.
|
|
93
103
|
*/
|
|
94
104
|
function recordDispatchUsage(execution, result) {
|
|
95
105
|
const { inputTokens, outputTokens, reasoningTokens } = result.usage ?? {};
|
|
96
106
|
const reported = [inputTokens, outputTokens, reasoningTokens].filter((count) => count !== undefined);
|
|
107
|
+
const model = dispatchedModel(execution);
|
|
97
108
|
emitLlmUsage({
|
|
98
109
|
outcome: result.ok ? "success" : "error",
|
|
99
110
|
modelSource: "configured",
|
|
100
|
-
...(
|
|
111
|
+
...(model ? { model } : {}),
|
|
101
112
|
durationMs: result.durationMs,
|
|
102
113
|
...(inputTokens !== undefined ? { promptTokens: inputTokens } : {}),
|
|
103
114
|
...(outputTokens !== undefined ? { completionTokens: outputTokens } : {}),
|
|
@@ -135,30 +146,13 @@ async function dispatchRunner(runner, prompt, opts, seams, llm) {
|
|
|
135
146
|
}
|
|
136
147
|
return redactResult(result, collectSensitiveValues(secrets));
|
|
137
148
|
}
|
|
138
|
-
/** The git repository a directory is inside, if any: a `.git` file, or a `.git` directory with a HEAD. */
|
|
139
|
-
function enclosingGitRepository(dir) {
|
|
140
|
-
for (let current = fs.realpathSync(dir);; current = path.dirname(current)) {
|
|
141
|
-
const marker = path.join(current, ".git");
|
|
142
|
-
if (fs.existsSync(path.join(marker, "HEAD")) || (fs.existsSync(marker) && fs.statSync(marker).isFile())) {
|
|
143
|
-
return current;
|
|
144
|
-
}
|
|
145
|
-
if (path.dirname(current) === current)
|
|
146
|
-
return undefined;
|
|
147
|
-
}
|
|
148
|
-
}
|
|
149
149
|
/**
|
|
150
150
|
* The scratch working directory for one model-work dispatch on an agent or SDK
|
|
151
151
|
* engine, so the edit the model-work tool policy grants never reaches the
|
|
152
|
-
* stash.
|
|
153
|
-
* directory, so a scratch directory inside one is refused.
|
|
152
|
+
* stash.
|
|
154
153
|
*/
|
|
155
154
|
function createModelWorkDirectory() {
|
|
156
|
-
|
|
157
|
-
const repository = enclosingGitRepository(dir);
|
|
158
|
-
if (repository === undefined)
|
|
159
|
-
return dir;
|
|
160
|
-
fs.rmSync(dir, { recursive: true, force: true });
|
|
161
|
-
throw new ConfigError(`Model work runs an agent in a scratch directory outside any git repository, but ${dir} is inside the repository at ${repository}.`, "INVALID_CONFIG_FILE", "Point TMPDIR at a directory outside any git repository.");
|
|
155
|
+
return fs.mkdtempSync(path.join(os.tmpdir(), "akm-model-work-"));
|
|
162
156
|
}
|
|
163
157
|
/**
|
|
164
158
|
* Run a built execution. Credentials are read here, once per call, and never
|
|
@@ -168,7 +162,7 @@ function createModelWorkDirectory() {
|
|
|
168
162
|
* limit, for one) has failed with `parse_error`.
|
|
169
163
|
*/
|
|
170
164
|
export async function runExecution(execution, options = {}) {
|
|
171
|
-
const scratch = execution.runner.kind !== "llm" &&
|
|
165
|
+
const scratch = execution.runner.kind !== "llm" && execution.request.authorization.policy?.id === MODEL_WORK_POLICY_ID
|
|
172
166
|
? createModelWorkDirectory()
|
|
173
167
|
: undefined;
|
|
174
168
|
try {
|
|
@@ -179,14 +173,14 @@ export async function runExecution(execution, options = {}) {
|
|
|
179
173
|
fs.rmSync(scratch, { recursive: true, force: true });
|
|
180
174
|
}
|
|
181
175
|
}
|
|
182
|
-
/**
|
|
183
|
-
|
|
184
|
-
* framing (claude's `--output-format json` result envelope, for one), as a
|
|
185
|
-
* workflow unit's does, and no answer is a `parse_error`.
|
|
186
|
-
*/
|
|
187
|
-
function modelWorkAnswer(runner, result) {
|
|
176
|
+
/** A successful reply's text, unwrapped from its harness's framing (claude's `--output-format json` result envelope, for one). */
|
|
177
|
+
export function unwrapHarnessReply(runner, result) {
|
|
188
178
|
const extractor = runner.kind === "agent" ? getHarness(runner.profile.platform ?? runner.profile.name)?.resultExtractor : undefined;
|
|
189
|
-
|
|
179
|
+
return extractor ? extractor(result) : { text: result.stdout };
|
|
180
|
+
}
|
|
181
|
+
/** A model-work reply's answer: its harness's framing stripped, and no answer is a `parse_error`. */
|
|
182
|
+
function modelWorkAnswer(runner, result) {
|
|
183
|
+
const extracted = unwrapHarnessReply(runner, result);
|
|
190
184
|
const answer = {
|
|
191
185
|
...result,
|
|
192
186
|
stdout: extracted.text,
|
|
@@ -56,8 +56,7 @@
|
|
|
56
56
|
* session model (plan §"Session, MCP, and identity across harnesses" —
|
|
57
57
|
* Aider is the plan's named example).
|
|
58
58
|
* - **inference** — not translated: the shared lowering reports each field of
|
|
59
|
-
* the request's inference as untranslated
|
|
60
|
-
* lists none for this harness).
|
|
59
|
+
* the request's inference as untranslated.
|
|
61
60
|
*
|
|
62
61
|
* Registered: `aiderBuilder` is `AiderHarness.agentBuilder` (`./index.ts`),
|
|
63
62
|
* one of the ten harnesses `HARNESS_REGISTRY` constructs
|
|
@@ -51,8 +51,7 @@
|
|
|
51
51
|
* — Q then refuses untrusted tool actions in non-interactive mode, which is
|
|
52
52
|
* the conservative failure mode.
|
|
53
53
|
* - **inference** — not translated: the shared lowering reports each field of
|
|
54
|
-
* the request's inference as untranslated
|
|
55
|
-
* lists none for this harness).
|
|
54
|
+
* the request's inference as untranslated.
|
|
56
55
|
*
|
|
57
56
|
* Registered: `amazonqBuilder` is `AmazonqHarness.agentBuilder`
|
|
58
57
|
* (`./index.ts`), one of the ten harnesses `HARNESS_REGISTRY` constructs
|
|
@@ -28,7 +28,6 @@
|
|
|
28
28
|
*
|
|
29
29
|
* The builder's `platform` stays `'claude'` (the canonical harness id).
|
|
30
30
|
*/
|
|
31
|
-
import { isModelWorkTools } from "../../../execution/source.js";
|
|
32
31
|
import { modelFromArgs, normalizeTools, resolveDispatchModel, } from "../../agent/builder-shared.js";
|
|
33
32
|
import { createAgentRequestLowerer } from "../../agent/request-lowering.js";
|
|
34
33
|
/** The model-work tool policy on Claude Code: read, edit in the working directory, `akm search`, `akm show`. */
|
|
@@ -45,17 +44,11 @@ export const MODEL_WORK_CLAUDE_FLAGS = Object.freeze([
|
|
|
45
44
|
/**
|
|
46
45
|
* Claude Code builder.
|
|
47
46
|
* Command shape:
|
|
48
|
-
* claude [--agent <name>] [--system-prompt "..."] [--model <m>] [--
|
|
49
|
-
* [--
|
|
47
|
+
* claude [--agent <name>] [--system-prompt "..."] [--model <m>] [--allowedTools <t>]
|
|
48
|
+
* [--output-format json] --print -- "<prompt>"
|
|
50
49
|
*
|
|
51
50
|
* --print switches Claude Code to non-interactive captured output mode.
|
|
52
51
|
*
|
|
53
|
-
* `--effort` carries the request's `reasoningEffort` as the harness's own
|
|
54
|
-
* level (`low`, `medium`, `high`, `xhigh` or `max` in 2.1.283), passed through
|
|
55
|
-
* as given: a value Claude Code rejects fails the dispatch. It is the only
|
|
56
|
-
* inference field Claude Code takes on its command line, so the others are
|
|
57
|
-
* reported as untranslated.
|
|
58
|
-
*
|
|
59
52
|
* The model-work tool policy lowers to {@link MODEL_WORK_CLAUDE_FLAGS} in place
|
|
60
53
|
* of the engine's `args` and `--allowedTools`, verified against Claude Code
|
|
61
54
|
* 2.1.283 and a local stub:
|
|
@@ -78,11 +71,9 @@ export const claudeBuilder = {
|
|
|
78
71
|
personaChannel: "native",
|
|
79
72
|
nativeAgentSelector: true,
|
|
80
73
|
tools: "all",
|
|
81
|
-
modelWorkTools: true,
|
|
82
|
-
inference: ["reasoningEffort"],
|
|
83
74
|
}),
|
|
84
75
|
build(profile, req) {
|
|
85
|
-
const modelWork =
|
|
76
|
+
const modelWork = req.modelWork === true;
|
|
86
77
|
const args = modelWork ? [...MODEL_WORK_CLAUDE_FLAGS] : [...profile.args];
|
|
87
78
|
if (req.agent) {
|
|
88
79
|
args.push("--agent", req.agent);
|
|
@@ -99,10 +90,7 @@ export const claudeBuilder = {
|
|
|
99
90
|
if (model)
|
|
100
91
|
args.push("--model", model);
|
|
101
92
|
}
|
|
102
|
-
|
|
103
|
-
if (typeof effort === "string" && effort.length > 0)
|
|
104
|
-
args.push("--effort", effort);
|
|
105
|
-
if (req.tools && !modelWork) {
|
|
93
|
+
if (req.tools) {
|
|
106
94
|
args.push("--allowedTools", normalizeTools(req.tools));
|
|
107
95
|
}
|
|
108
96
|
if (req.schema) {
|
|
@@ -43,8 +43,7 @@
|
|
|
43
43
|
* field yet); {@link codexResumeArgs} exposes the argv prefix for the
|
|
44
44
|
* integration task that wires session-id reuse from `workflow_run_units`.
|
|
45
45
|
* - The request's inference is not translated: the shared lowering reports
|
|
46
|
-
* each field as untranslated
|
|
47
|
-
* for codex). codex would take `reasoningEffort` as
|
|
46
|
+
* each field as untranslated. codex would take `reasoningEffort` as
|
|
48
47
|
* `-c model_reasoning_effort=<v>`, which is left to the integration task.
|
|
49
48
|
*
|
|
50
49
|
* Registered: `codexBuilder` is `CodexHarness.agentBuilder` (`./index.ts`),
|