akm-cli 0.9.25-alpha.1 → 0.9.25-alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/CHANGELOG.md +159 -280
  2. package/dist/cli.js +1 -1
  3. package/dist/commands/improve/consolidate/pair-pass.js +1 -0
  4. package/dist/commands/improve/consolidate.js +7 -2
  5. package/dist/commands/improve/execution.js +2 -3
  6. package/dist/commands/improve/extract.js +1 -0
  7. package/dist/commands/improve/improve-cli.js +33 -1
  8. package/dist/commands/improve/loop-stages.js +3 -0
  9. package/dist/commands/improve/reflect-noise.js +125 -0
  10. package/dist/commands/improve/reflect.js +13 -16
  11. package/dist/commands/improve/retrieval-gate.js +7 -2
  12. package/dist/commands/improve/stage.js +31 -39
  13. package/dist/commands/proposal/drain.js +4 -7
  14. package/dist/commands/proposal/propose.js +2 -11
  15. package/dist/commands/proposal/validators/proposal-quality-validators.js +4 -2
  16. package/dist/commands/read/search-cli.js +0 -38
  17. package/dist/core/config/schema/engines.js +15 -33
  18. package/dist/core/config/schema/improve-processes.js +16 -0
  19. package/dist/core/redaction.js +4 -0
  20. package/dist/core/spawn-env.js +25 -0
  21. package/dist/core/structured.js +1 -1
  22. package/dist/execution/source.js +8 -12
  23. package/dist/integrations/agent/config.js +1 -3
  24. package/dist/integrations/agent/engine-resolution.js +0 -3
  25. package/dist/integrations/agent/execution.js +14 -13
  26. package/dist/integrations/agent/index.js +1 -1
  27. package/dist/integrations/agent/model-map.js +15 -16
  28. package/dist/integrations/agent/profiles.js +2 -2
  29. package/dist/integrations/agent/prompts.js +3 -39
  30. package/dist/integrations/agent/request-lowering.js +9 -7
  31. package/dist/integrations/agent/runner-dispatch.js +25 -31
  32. package/dist/integrations/harnesses/aider/agent-builder.js +1 -2
  33. package/dist/integrations/harnesses/amazonq/agent-builder.js +1 -2
  34. package/dist/integrations/harnesses/claude/agent-builder.js +4 -16
  35. package/dist/integrations/harnesses/codex/agent-builder.js +1 -2
  36. package/dist/integrations/harnesses/ids.js +10 -16
  37. package/dist/integrations/harnesses/opencode/agent-builder.js +14 -24
  38. package/dist/integrations/harnesses/opencode/model-config.js +15 -62
  39. package/dist/integrations/harnesses/opencode/model-work-agent.js +71 -36
  40. package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -6
  41. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +32 -90
  42. package/dist/integrations/harnesses/openhands/agent-builder.js +1 -2
  43. package/dist/integrations/harnesses/pi/agent-builder.js +1 -2
  44. package/dist/llm/feature-gate.js +2 -5
  45. package/dist/llm/index-passes.js +2 -2
  46. package/dist/llm/structured-call.js +5 -5
  47. package/dist/output/shapes/passthrough.js +1 -0
  48. package/dist/scripts/akm-migrate-node.js +170 -194
  49. package/dist/scripts/akm-migrate.js +170 -194
  50. package/dist/workflows/exec/unit-dispatch.js +4 -13
  51. package/docs/reference/cli.md +12 -7
  52. package/docs/reference/configuration.md +78 -82
  53. package/docs/reference/data-and-telemetry.md +2 -3
  54. package/docs/reference/workflow-schema.md +6 -9
  55. package/package.json +1 -1
  56. package/schemas/akm-config.json +108 -36
@@ -9,8 +9,13 @@
9
9
  * cosmetic edit through costs a review, suppressing a real fix loses work — so
10
10
  * anything uncertain is `substantive`: code (fenced or indented) compares
11
11
  * verbatim, and headings, tables and breaks never absorb the next line.
12
+ *
13
+ * `findReflectDefect` is its counterpart for an edit that is not noise but a
14
+ * defect no judge needs to weigh: placeholder text, talk about the edit itself,
15
+ * or frontmatter copied into the body.
12
16
  */
13
17
  import { parse as yamlParse } from "yaml";
18
+ import { parseFrontmatter } from "../../core/asset/frontmatter.js";
14
19
  /**
15
20
  * `noop` (identical up to trailing whitespace) and `cosmetic` (identical after
16
21
  * normalizing frontmatter as parsed YAML and prose as unwrapped text) never
@@ -36,6 +41,126 @@ export function classifyReflectChange(sourceContent, candidateContent) {
36
41
  }
37
42
  return "substantive";
38
43
  }
44
+ /** Not TODO, TBD or FIXME: an asset may carry those on purpose, and the owner does not want them refused. */
45
+ const DEFAULT_PLACEHOLDERS = [
46
+ "please confirm",
47
+ "please verify",
48
+ "to be confirmed",
49
+ "to be determined",
50
+ "to be verified",
51
+ ];
52
+ const DEFAULT_META_COMMENTARY = [
53
+ "feedback signal",
54
+ "feedback signals",
55
+ "feedback indicate",
56
+ "feedback indicates",
57
+ "feedback suggest",
58
+ "feedback suggests",
59
+ "feedback ask",
60
+ "feedback asks",
61
+ "feedback says",
62
+ "feedback report",
63
+ "feedback reports",
64
+ "feedback request",
65
+ "feedback requests",
66
+ "this revision",
67
+ "the source asset",
68
+ "the source note",
69
+ "the source memory",
70
+ "the original asset",
71
+ "the original note",
72
+ "the original memory",
73
+ "the original version of this",
74
+ "quality gate rejected",
75
+ "proposal rejected",
76
+ ];
77
+ const DEFAULT_FRONTMATTER_KEYS = [
78
+ "sources",
79
+ "updated",
80
+ "inferenceProcessed",
81
+ "captureMode",
82
+ "beliefState",
83
+ "xrefs",
84
+ "contradictedBy",
85
+ "outcomeData",
86
+ "orderedActions",
87
+ "generated",
88
+ "verified",
89
+ "description",
90
+ "when_to_use",
91
+ "tags",
92
+ "searchHints",
93
+ "quality",
94
+ "salience",
95
+ "salienceInputs",
96
+ "lint_skip",
97
+ "type",
98
+ ];
99
+ /**
100
+ * The first defect the candidate has and its source lacks, or `undefined`. Each
101
+ * rule counts only what the revision adds, so text the asset already carried is
102
+ * not held against it. On 396 labelled reflect edits (83 good, 313 bad) the
103
+ * default lists hit 22 bad edits and no good one, so a hit is refused unjudged.
104
+ */
105
+ export function findReflectDefect(sourceContent, candidateContent, filter = {}) {
106
+ if (gainsPhrases(filter.placeholders ?? DEFAULT_PLACEHOLDERS, sourceContent, candidateContent)) {
107
+ return "placeholder_added";
108
+ }
109
+ if (gainsPhrases(filter.metaCommentary ?? DEFAULT_META_COMMENTARY, sourceContent, candidateContent)) {
110
+ return "meta_commentary_added";
111
+ }
112
+ if (frontmatterCopiedIntoBody(sourceContent, candidateContent, filter.frontmatterKeys ?? DEFAULT_FRONTMATTER_KEYS)) {
113
+ return "frontmatter_copied_into_body";
114
+ }
115
+ return undefined;
116
+ }
117
+ /** Frontmatter fields that name other assets; one of their values newly in the body is provenance copied over. */
118
+ const PROVENANCE_KEYS = ["sources", "xrefs", "contradictedBy"];
119
+ /** Shorter values (a bare name or id) say too little to find in a body. */
120
+ const PROVENANCE_VALUE_MIN_CHARS = 12;
121
+ function escapeRegExp(text) {
122
+ return text.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
123
+ }
124
+ /** Whether the candidate holds more of the phrases than the source: whole words, any case, any whitespace between words. */
125
+ function gainsPhrases(phrases, source, candidate) {
126
+ const alternatives = phrases
127
+ .map((phrase) => phrase.trim().split(/\s+/).map(escapeRegExp).join("\\s+"))
128
+ .filter((alternative) => alternative !== "");
129
+ if (alternatives.length === 0)
130
+ return false;
131
+ const pattern = new RegExp(`(?<!\\w)(?:${alternatives.join("|")})(?!\\w)`, "gi");
132
+ return (candidate.match(pattern)?.length ?? 0) > (source.match(pattern)?.length ?? 0);
133
+ }
134
+ /** The body gains a line that starts with one of the frontmatter keys outside code, or a provenance value from either asset's frontmatter. */
135
+ function frontmatterCopiedIntoBody(source, candidate, keys) {
136
+ if (keys.length === 0)
137
+ return false;
138
+ const keyLine = new RegExp(`^(?:${keys.map(escapeRegExp).join("|")}):(?:[ \\t].*)?$`);
139
+ const keyLines = (text) => parseLowValueSections(text).proseLines.filter((line) => keyLine.test(line)).length;
140
+ if (keyLines(candidate) > keyLines(source))
141
+ return true;
142
+ const sourceBody = lettersAndDigits(splitFrontmatter(source).body);
143
+ const candidateBody = lettersAndDigits(splitFrontmatter(candidate).body);
144
+ return [...provenanceValues(source), ...provenanceValues(candidate)].some((value) => {
145
+ const v = lettersAndDigits(value);
146
+ return v.length >= PROVENANCE_VALUE_MIN_CHARS && candidateBody.includes(v) && !sourceBody.includes(v);
147
+ });
148
+ }
149
+ /** Every string under the provenance keys of a frontmatter block: a scalar, or the items of a list. */
150
+ function provenanceValues(content) {
151
+ const { data } = parseFrontmatter(content);
152
+ return PROVENANCE_KEYS.flatMap((key) => {
153
+ const value = data[key];
154
+ return (Array.isArray(value) ? value : [value]).filter((item) => typeof item === "string");
155
+ });
156
+ }
157
+ /** Lower-case letters and digits, each run of anything else a single space. */
158
+ function lettersAndDigits(text) {
159
+ return text
160
+ .toLowerCase()
161
+ .replace(/[^a-z0-9]+/g, " ")
162
+ .trim();
163
+ }
39
164
  /** A `---` frontmatter block and the rest (`fmText: null` when there is none). */
40
165
  export function splitFrontmatter(raw) {
41
166
  const m = raw.match(/^---\r?\n([\s\S]*?)\r?\n---\r?\n?([\s\S]*)$/);
@@ -26,9 +26,8 @@ import { parseEmbeddedJsonResponse } from "../../core/parse.js";
26
26
  import { redactSensitiveText } from "../../core/redaction.js";
27
27
  import { resolveStandardsContext } from "../../core/standards/resolve-standards-context.js";
28
28
  import { warn, warnOnce } from "../../core/warn.js";
29
- import { MODEL_WORK_TOOLS } from "../../execution/source.js";
30
29
  import { lookup } from "../../indexer/indexer.js";
31
- import { DEFAULT_MODEL_WORK_TIMEOUT_MS } from "../../integrations/agent/config.js";
30
+ import { DEFAULT_LLM_TIMEOUT_MS } from "../../integrations/agent/config.js";
32
31
  import { fallbackAnnouncement, NO_ENGINE_MESSAGE_SUFFIX, NO_ENGINE_REMEDY, withEngineFallback, } from "../../integrations/agent/engine-fallback.js";
33
32
  import { buildExecution, resolveExecution } from "../../integrations/agent/execution.js";
34
33
  import { buildReflectOutputRepairPrompt, buildReflectPrompt, parseAgentProposalPayload, REFLECT_CONTENT_CAP, REFLECT_TRUNCATION_MARKER, } from "../../integrations/agent/prompts.js";
@@ -42,7 +41,7 @@ import { deriveLessonRef } from "./distill.js";
42
41
  import { findAssetFilePath } from "./eligibility.js";
43
42
  import { resolveImproveExecution } from "./execution.js";
44
43
  import { recordLedgerAttempt } from "./ledger.js";
45
- import { classifyReflectChange, splitFrontmatter } from "./reflect-noise.js";
44
+ import { classifyReflectChange, findReflectDefect, splitFrontmatter } from "./reflect-noise.js";
46
45
  import { loadRetrievalQueries, runRetrievalRegressionGate } from "./retrieval-gate.js";
47
46
  import { callStageOnce, errMessage, mintProposal, noticeSet, rejectedProposalContext, resolveQualityGateJudge, runReflectQualityJudge, } from "./stage.js";
48
47
  const MAX_FEEDBACK_LINES = 10;
@@ -494,7 +493,7 @@ export async function runReflectIteration(opts) {
494
493
  ? (opts.timeoutMs ?? null)
495
494
  : Object.hasOwn(opts.runner, "timeoutMs")
496
495
  ? (opts.runner.timeoutMs ?? null)
497
- : DEFAULT_MODEL_WORK_TIMEOUT_MS;
496
+ : DEFAULT_LLM_TIMEOUT_MS;
498
497
  const deadline = typeof configuredTimeout === "number" ? start + configuredTimeout : undefined;
499
498
  const messages = [{ role: "user", content: opts.prompt ?? "" }];
500
499
  if (opts.priorDraft !== undefined && opts.iteration > 0) {
@@ -516,16 +515,8 @@ export async function runReflectIteration(opts) {
516
515
  parsed: { outputMode: opts.outputMode, repairAttempts },
517
516
  };
518
517
  };
519
- // A reply that broke the contract. An agent or SDK engine's failure names the engine; an LLM
520
- // reply keeps its message, because improve feeds a failed reflect's `error` into the next
521
- // prompts as a pattern to avoid, and rewording it would change those requests.
522
- const invalidReply = (err, reply) => {
523
- if (runnerIsLlm(opts.runner))
524
- return failure(err, "parse_error", reply, 0);
525
- const attempts = repairAttempts + 1;
526
- const message = `Engine "${opts.runner.engine}" reply was not a valid reflect proposal after ${attempts} attempt${attempts === 1 ? "" : "s"}: ${errMessage(err)}`;
527
- return failure(new Error(message), "parse_error", reply, 0);
528
- };
518
+ // A reply that broke the contract; its message is the parser's, on every engine kind.
519
+ const invalidReply = (err, reply) => failure(err, "parse_error", reply, 0);
529
520
  // The result of the dispatch that failed, kept for an agent or SDK engine.
530
521
  let dispatched;
531
522
  // Reflect parses and repairs its own reply (the repair turn below), so one dispatch, unvalidated.
@@ -756,7 +747,7 @@ function preflightReflectDispatch(runnerSpec, onNotices) {
756
747
  const prepared = resolveExecution({
757
748
  content: "Validate reflect operation transport before dispatch.",
758
749
  runner: runnerSpec,
759
- current: { tools: MODEL_WORK_TOOLS },
750
+ modelWork: true,
760
751
  });
761
752
  const lowered = buildExecution(prepared.request, prepared.runner);
762
753
  onNotices(lowered.notices);
@@ -907,7 +898,8 @@ const NOISE_SUBREASONS = {
907
898
  * exact content that would be persisted, then mint. A judge pass is staged only
908
899
  * when the body is unchanged: a body edit the judge passes, or one made with the
909
900
  * gate off, waits for review. Size-flagged or truncation-leaking content skips
910
- * the judge and waits for review.
901
+ * the judge and waits for review. A revision with a deterministic defect is
902
+ * refused before the judge runs, whether or not the gate is on.
911
903
  */
912
904
  async function finalizeReflectProposal(args) {
913
905
  const { run, assetContent, result, judge, feedback } = args;
@@ -953,11 +945,16 @@ async function finalizeReflectProposal(args) {
953
945
  }, options.eventsCtx);
954
946
  return reflectFailure(run, result, "quality_rejected", message, false);
955
947
  };
948
+ // A defect no judge needs to weigh is refused before any judge call, whether or not the gate is on.
949
+ const defect = assetContent === undefined ? undefined : findReflectDefect(assetContent, payload.content, options.defectFilter);
950
+ if (defect)
951
+ return refuse(defect, { reflectDefect: defect }, `Reflect proposal refused before the judge: ${defect}`);
956
952
  let verdict;
957
953
  let judgeFailed = false;
958
954
  if (judged) {
959
955
  verdict = await runReflectQualityJudge(run.config, payload.content, assetContent ?? "", feedback, options.chat, {
960
956
  runnerSelectionFrozen: true,
957
+ ref: payload.ref,
961
958
  ...(judge.runner ? { llmRunner: judge.runner } : {}),
962
959
  ...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
963
960
  ...(options.signal ? { signal: options.signal } : {}),
@@ -32,6 +32,10 @@ const GRADE_SCHEMA = {
32
32
  additionalProperties: false,
33
33
  properties: { grade: { type: "integer", minimum: 0, maximum: 3 }, reason: { type: "string" } },
34
34
  };
35
+ function parseGrade(raw) {
36
+ const grade = parseEmbeddedJsonResponse(raw)?.grade;
37
+ return typeof grade === "number" && Number.isInteger(grade) && grade >= 0 && grade <= 3 ? grade : undefined;
38
+ }
35
39
  /** Up to five distinct task queries, in the given order, whitespace collapsed. */
36
40
  export function usableRetrievalQueries(raw) {
37
41
  const out = [];
@@ -100,10 +104,11 @@ export async function runRetrievalRegressionGate(args) {
100
104
  ...(args.signal ? { signal: args.signal } : {}),
101
105
  ...(args.chat ? { chat: args.chat } : {}),
102
106
  },
107
+ parse: parseGrade,
103
108
  ...(args.onNotices ? { onNotices: args.onNotices } : {}),
104
109
  });
105
- const grade = outcome.ok ? parseEmbeddedJsonResponse(outcome.raw)?.grade : undefined;
106
- if (typeof grade !== "number" || !Number.isInteger(grade) || grade < 0 || grade > 3) {
110
+ const grade = outcome.ok ? parseGrade(outcome.raw) : undefined;
111
+ if (grade === undefined) {
107
112
  return {
108
113
  pass: false,
109
114
  queries: args.queries.length,
@@ -3,9 +3,8 @@
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  import { getImproveProcessConfig } from "../../core/config/config.js";
5
5
  import { ConfigError } from "../../core/errors.js";
6
- import { validateJsonSchemaSubset } from "../../core/json-schema.js";
7
6
  import { parseEmbeddedJsonResponse } from "../../core/parse.js";
8
- import { runStructured } from "../../core/structured.js";
7
+ import { defaultFeedback } from "../../core/structured.js";
9
8
  import { warn } from "../../core/warn.js";
10
9
  import { runnerLlmConnection } from "../../integrations/agent/runner.js";
11
10
  import { callStructured, dispatchFailureReason, dispatchFailureResult, } from "../../llm/structured-call.js";
@@ -45,43 +44,23 @@ export function stageRunner(frozen, config, profile, processName, onNotices) {
45
44
  onNotices?.(resolved.notices);
46
45
  return resolved?.runner;
47
46
  }
48
- const TRANSPORT_FAILED = Symbol("stage-transport-failed");
49
47
  /**
50
48
  * One model call. Provider trouble (transport error, timeout, abort, a
51
49
  * disabled feature) comes back as `{ ok: false }`; only a configuration
52
- * failure throws. A reply to a call with `request.responseSchema` that fails
53
- * the schema gets one corrective retry. The caller's own parse still decides
54
- * what it accepts, so the last reply comes back even when it fails the schema.
50
+ * failure throws. A reply to a call with `request.responseSchema` that the
51
+ * stage's own `parse` rejects gets one corrective retry; the last reply comes
52
+ * back either way, and the caller parses it again.
55
53
  */
56
54
  export async function callStage(call) {
57
- const schema = call.request?.responseSchema;
58
- if (!schema)
59
- return callStageOnce(call);
60
- let reply = undefined;
61
- let failure = undefined;
62
- try {
63
- await runStructured({
64
- dispatch: async (feedback) => {
65
- const outcome = await callStageOnce(feedback ? { ...call, prompt: `${call.prompt}\n\n${feedback}` } : call);
66
- if (!outcome.ok) {
67
- failure = outcome;
68
- throw TRANSPORT_FAILED;
69
- }
70
- reply = outcome;
71
- return outcome.raw;
72
- },
73
- validate: (candidate) => {
74
- const errors = validateJsonSchemaSubset(candidate, schema);
75
- return errors.length === 0 ? { ok: true, value: candidate } : { ok: false, errors };
76
- },
77
- });
78
- }
79
- catch (err) {
80
- if (err !== TRANSPORT_FAILED)
81
- throw err;
82
- }
55
+ const reply = await callStageOnce(call);
56
+ if (!reply.ok || !call.request?.responseSchema)
57
+ return reply;
58
+ if ((call.parse ?? parseEmbeddedJsonResponse)(reply.raw) !== undefined)
59
+ return reply;
60
+ const feedback = defaultFeedback({ reason: "parse_error", errors: [] });
61
+ const retry = await callStageOnce({ ...call, prompt: `${call.prompt}\n\n${feedback}` });
83
62
  // A retry that fails in transport keeps the first reply, which the caller may still accept.
84
- return reply ?? failure ?? { ok: false, reason: "error" };
63
+ return retry.ok ? retry : reply;
85
64
  }
86
65
  /** Timeout and abort come from the dispatch's own reason, whatever the runner's kind. */
87
66
  function failureReason(err) {
@@ -283,15 +262,24 @@ function buildChangedRegion(sourceContent, candidateContent) {
283
262
  const added = candidate.slice(prefix, candidate.length - suffix).join("\n");
284
263
  return boundedDocument(`Removed or replaced:\n${removed || "(none)"}\n\nAdded or replacement:\n${added || "(none)"}`);
285
264
  }
286
- /** Judge prompt for an in-place revision. */
287
- export function buildReflectJudgePrompt(candidateContent, sourceContent, feedback) {
265
+ /**
266
+ * What the judge may do with tools when it runs on an agent engine: verify a
267
+ * fact the revision adds or alters, and nothing else. The plain judge's prompt
268
+ * is unchanged (its rubric is tuned and measured without this paragraph).
269
+ */
270
+ function reflectJudgeToolRules(ref) {
271
+ const asset = ref ? `The asset is \`${ref}\`: read it with akm_show, ` : "Read an asset with akm_show ";
272
+ return `Tools: ${asset}or an asset the changed region names, only to verify a fact the revision adds or alters; the text above already shows every change. Do not search, do not read anything else, and do not use a tool to judge structure or wording. One or two reads at most. Before scoring, check three lists: (1) every statement the revision adds: find each in the asset, or as a fact the feedback states about the subject, and score QUALITY 1-2 if any is in neither; a statement is found only when the asset or the feedback says it, in any words: a new step, cause, consequence or detail that merely seems to follow is not found; feedback says what to fix and is not content, so an added statement about how the asset was used, found or verified is unsupported; (2) every fact, caveat and field of the source: find each in the revision, and score PRESERVATION 1-3 if any is missing; (3) every point the feedback makes: find the text it is about changed in the revision, and score NEED 2-3 if any is not; a note that restates the feedback does not address it. A read that finds nothing wrong raises no score above what these lists support. Then reply with the JSON.`;
273
+ }
274
+ /** Judge prompt for an in-place revision. `tools` is set when the judge runs on an agent engine. */
275
+ export function buildReflectJudgePrompt(candidateContent, sourceContent, feedback, tools) {
288
276
  return [
289
277
  "You are evaluating a proposed revision to an existing akm asset.",
290
278
  "",
291
279
  "Score this revision on each criterion from 1 (poor) to 5 (excellent):",
292
- "1. NEED: Does the revision fix a concrete problem in the source? Concrete problems are: something the feedback reports as wrong or missing; a factual error; or broken, garbled, truncated or missing text, including frontmatter fields such as description or when_to_use. Score 4-5 when it fixes one, even a small one. Score 1-2 when the source was already correct and the revision only rewords, restates, reformats, or adds headings, an introduction or a table of contents.",
293
- "2. PRESERVATION: Does it keep every concrete fact, identifier, command, path, number and example from the source, without truncation?",
294
- "3. QUALITY: Is it coherent and accurate, with no claims, steps or details that the source or the feedback does not support?",
280
+ "1. NEED: Does the revision fix a concrete problem in the source? Concrete problems are: something the feedback reports as wrong or missing; a factual error; broken, garbled, truncated or missing text; and a missing or broken title, description or when_to_use field. Compare the source's description with the revision's: a description with a sentence split in its middle by a stray period or line-wrap artifact (as in 'calls. asset writes'), an unbalanced or escaped quote, or a truncated ending is broken, and repairing it is a concrete problem fixed even when the rest of the revision only adds stamps or reformats; adding or removing a trailing period, or rewording a readable description, repairs nothing. A missing title, description or when_to_use is a concrete problem whether or not the feedback mentions it: empty or positive feedback does not mean the source was complete, and the frontmatter is complete only when it has all three. A when_to_use is missing when the frontmatter has none, even if the body has a 'when to use' section; moving or copying that text into the field is the fix. A missing type field or a provenance stamp such as generated or verified is not a concrete problem. Score 4-5 when the revision fixes one, even a small one, whatever else it also reformats. Score 1-2 when the source was already complete and correct and the revision only rewords, restates, reformats, adds a type field or a stamp, or adds headings, an introduction or a table of contents.",
281
+ "2. PRESERVATION: Does it keep every concrete fact, identifier, command, path, number, example, caveat and frontmatter field from the source, without truncation? Check the changed region line by line. Score 1-3 when any of them is dropped or weakened, even when the revision also fixes something. Trimming narrative that states no fact, or removing what the feedback asks to remove or rescope, is not a drop.",
282
+ "3. QUALITY: Is it coherent and accurate, with no claims, steps or details that the source or the feedback does not support? Check each added or changed statement against the source and the feedback: a statement that follows from either counts as supported, and frontmatter stamps such as type, generated, verified or quality, whatever their values, and a restatement of existing content are not claims. Score 1-2 when the revision adds a claim neither supports, turns a draft or proposal into a decision, strengthens a statement beyond the source (a preference into a requirement, a possibility into a fact, a pending fix into a done one), adds a hedge such as 'may be outdated', or a placeholder such as 'TODO' or 'verify'; a fix elsewhere in the revision does not raise this score.",
295
283
  "",
296
284
  "Feedback:",
297
285
  "```",
@@ -313,6 +301,7 @@ export function buildReflectJudgePrompt(candidateContent, sourceContent, feedbac
313
301
  buildChangedRegion(sourceContent, candidateContent),
314
302
  "```",
315
303
  "",
304
+ ...(tools ? [reflectJudgeToolRules(tools.ref), ""] : []),
316
305
  'Return ONLY valid JSON, no prose: {"scores": {"need": <1-5 integer>, "preservation": <1-5 integer>, "quality": <1-5 integer>}, "reason": "<one sentence>"}',
317
306
  ].join("\n");
318
307
  }
@@ -421,6 +410,7 @@ async function runQualityJudge(feature, config, prompt, keys, chat, options) {
421
410
  ...(options.signal ? { signal: options.signal } : {}),
422
411
  ...(chat ? { chat } : {}),
423
412
  },
413
+ parse: (raw) => parseJudgeResponse(raw, keys),
424
414
  ...(options.onNotices ? { onNotices: options.onNotices } : {}),
425
415
  });
426
416
  if (!outcome.ok) {
@@ -462,6 +452,8 @@ export function runLessonQualityJudge(config, lessonContent, sourceContent, chat
462
452
  }
463
453
  /** Judge an in-place reflect revision without new-lesson novelty criteria. */
464
454
  export function runReflectQualityJudge(config, candidateContent, sourceContent, feedback, chat, options = {}) {
465
- const prompt = buildReflectJudgePrompt(candidateContent, sourceContent, feedback);
455
+ // A judge on an agent engine gets the tool rules; the runner is the frozen one or none.
456
+ const tools = options.llmRunner && options.llmRunner.kind !== "llm" ? { ref: options.ref } : undefined;
457
+ const prompt = buildReflectJudgePrompt(candidateContent, sourceContent, feedback, tools);
466
458
  return runQualityJudge("proposal_quality_gate", config, prompt, REFLECT_JUDGE_CRITERIA, chat, options);
467
459
  }
@@ -25,8 +25,7 @@ import { ConfigError } from "../../core/errors.js";
25
25
  import { appendEvent } from "../../core/events.js";
26
26
  import { escapeJsonStringControls, stripCodeFences, stripThinkBlocks } from "../../core/parse.js";
27
27
  import { info, warn } from "../../core/warn.js";
28
- import { MODEL_WORK_TOOLS } from "../../execution/source.js";
29
- import { DEFAULT_MODEL_WORK_TIMEOUT_MS } from "../../integrations/agent/config.js";
28
+ import { DEFAULT_LLM_TIMEOUT_MS } from "../../integrations/agent/config.js";
30
29
  import { buildExecution, resolveExecution } from "../../integrations/agent/execution.js";
31
30
  import { assertRunnerCredentials, runExecution, } from "../../integrations/agent/runner-dispatch.js";
32
31
  import { errMessage, noticeSet } from "../improve/stage.js";
@@ -177,10 +176,8 @@ async function dispatchJudgment(runner, prompt, seams) {
177
176
  const prepared = resolveExecution({
178
177
  content: prompt,
179
178
  runner,
180
- current: {
181
- ...(Object.hasOwn(runner, "timeoutMs") ? {} : { timeout: DEFAULT_MODEL_WORK_TIMEOUT_MS }),
182
- tools: MODEL_WORK_TOOLS,
183
- },
179
+ current: Object.hasOwn(runner, "timeoutMs") ? {} : { timeout: DEFAULT_LLM_TIMEOUT_MS },
180
+ modelWork: true,
184
181
  });
185
182
  const lowered = buildExecution(prepared.request, prepared.runner);
186
183
  notices = lowered.notices;
@@ -345,7 +342,7 @@ export async function drainProposals(opts, promoteFn = akmProposalAccept, reject
345
342
  const prepared = resolveExecution({
346
343
  content: "Validate the selected proposal judgment runner before mutation.",
347
344
  runner: opts.judgment,
348
- current: { tools: MODEL_WORK_TOOLS },
345
+ modelWork: true,
349
346
  });
350
347
  assertRunnerCredentials(buildExecution(prepared.request, prepared.runner).runner);
351
348
  }
@@ -27,8 +27,7 @@ import { deriveEntryProvenance } from "../../indexer/installations.js";
27
27
  import { fallbackAnnouncement } from "../../integrations/agent/engine-fallback.js";
28
28
  import { buildExecution, resolveExecution } from "../../integrations/agent/execution.js";
29
29
  import { buildProposePrompt, PROPOSAL_JSON_SCHEMA, validateProposalPayload, } from "../../integrations/agent/prompts.js";
30
- import { assertRunnerCredentials, collectDispatchSensitiveValues, runExecution, } from "../../integrations/agent/runner-dispatch.js";
31
- import { getHarness } from "../../integrations/harnesses/index.js";
30
+ import { assertRunnerCredentials, collectDispatchSensitiveValues, runExecution, unwrapHarnessReply, } from "../../integrations/agent/runner-dispatch.js";
32
31
  import { baseFailureFields, enoentHintMessage, isEnoentFailure } from "../agent/agent-support.js";
33
32
  import { createProposal, resolveProposalQueueTarget, } from "./repository.js";
34
33
  function failureEnvelope(result, type, name, engine, notices, fallbackReason = "non_zero_exit") {
@@ -45,14 +44,6 @@ function noticeFields(notices) {
45
44
  return notices.length > 0 ? { notices } : {};
46
45
  }
47
46
  const DISPATCH_FAILED = Symbol("proposal-dispatch-failed");
48
- /** A reply's text, unwrapped from its harness's framing (claude's `--output-format json` envelope). */
49
- function replyText(execution, result) {
50
- const runner = execution.runner;
51
- if (runner.kind !== "agent")
52
- return result.stdout;
53
- const extractor = getHarness(runner.profile.platform ?? runner.profile.name)?.resultExtractor;
54
- return extractor ? extractor(result).text : result.stdout;
55
- }
56
47
  /**
57
48
  * Resolve, lower, and dispatch the already-rendered proposal prompt with the
58
49
  * proposal's JSON Schema as its output schema, and capture the reply. A reply
@@ -92,7 +83,7 @@ async function dispatchProposalPrompt(prompt, config, options, onDispatchReady)
92
83
  results.push(result);
93
84
  if (!result.ok)
94
85
  throw DISPATCH_FAILED;
95
- return replyText(execution, result);
86
+ return unwrapHarnessReply(execution.runner, result).text;
96
87
  },
97
88
  validate: validateProposalPayload,
98
89
  });
@@ -9,7 +9,8 @@
9
9
  * growing past min(max(250% of the source, 2500 bytes), 25000 bytes) suggests
10
10
  * speculation. The absolute bounds keep small assets (a p25 source is ~780
11
11
  * bytes) from tripping on one good paragraph, and 25000 (below p99) still
12
- * catches runaway expansion. Sources under 200 bytes are too noisy to judge.
12
+ * catches runaway expansion. Sources under 200 bytes are too noisy to judge. A
13
+ * body that does not grow is never expansion, however long its source.
13
14
  */
14
15
  import { parseFrontmatter } from "../../../core/asset/frontmatter.js";
15
16
  import { parseRefInput } from "../../../core/asset/resolve-ref.js";
@@ -159,7 +160,8 @@ export function checkReflectSize(sourceBody, proposedBody) {
159
160
  return { ok: false, code: "EXCESSIVE_SHRINKAGE", ratio };
160
161
  }
161
162
  const expandCeiling = Math.min(Math.max(REFLECT_EXPAND_RATIO_MAX * sourceLen, REFLECT_ABSOLUTE_CEILING_BYTES), REFLECT_ABSOLUTE_MAX_BYTES);
162
- if (proposedLen > expandCeiling)
163
+ // A body that does not grow is never expansion, even where the cap sits below the source's own length.
164
+ if (proposedLen > sourceLen && proposedLen > expandCeiling)
163
165
  return { ok: false, code: "EXCESSIVE_EXPANSION", ratio };
164
166
  return { ok: true };
165
167
  }
@@ -81,20 +81,6 @@ export const searchCommand = defineJsonCommand({
81
81
  description: "Include session assets (excluded from default search results via config.search.defaultExcludeTypes).",
82
82
  default: false,
83
83
  },
84
- // Declared as the POSITIVE name with `default: true` so citty's native
85
- // `--no-<name>` negation (it strips a leading `--no-` from ANY token and
86
- // negates the remainder BEFORE consulting the declared-args table — see
87
- // node_modules/citty/dist/index.mjs) does the work, the same pattern
88
- // `sync --push/--no-push` uses. A flag DECLARED as `no-track-usage` could
89
- // never be negated: `--no-track-usage` parses as "negate `track-usage`",
90
- // a name nothing declared, leaving the real key at its default forever
91
- // (F1/A1).
92
- "track-usage": {
93
- type: "boolean",
94
- default: true,
95
- description: "A successful search records usage-events telemetry. Default: on. Use --no-track-usage to run a " +
96
- "search that records nothing.",
97
- },
98
84
  },
99
85
  async run({ args }) {
100
86
  rejectRetiredSourceFlag();
@@ -108,7 +94,6 @@ export const searchCommand = defineJsonCommand({
108
94
  const filters = parseScopeFilterFlags(filterTokens, "--filter");
109
95
  const includeProposed = args["include-proposed"] === true;
110
96
  const belief = parseBeliefFilterMode(typeof args.belief === "string" ? args.belief : undefined);
111
- const skipLogging = args["track-usage"] === false;
112
97
  const includeSessions = args["include-sessions"];
113
98
  const assets = args.assets === true;
114
99
  const outputMode = getOutputMode();
@@ -121,7 +106,6 @@ export const searchCommand = defineJsonCommand({
121
106
  includeProposed,
122
107
  belief,
123
108
  includeSessions,
124
- skipLogging,
125
109
  assets,
126
110
  eventSource: resolveUsageEventSource(),
127
111
  attributionProjection: outputMode.shape === "agent" ? "agent" : outputMode.detail,
@@ -154,15 +138,6 @@ export const curateCommand = defineJsonCommand({
154
138
  "with a workflow asset's own `budget` field (a run-cost cap) — this is a context-size target for this " +
155
139
  "one curate call.",
156
140
  },
157
- // Declared as the POSITIVE name with `default: true` — see the
158
- // `track-usage` comment on `searchCommand` above for why a flag NAME
159
- // must never start with `no-`.
160
- "track-usage": {
161
- type: "boolean",
162
- default: true,
163
- description: "A successful curate records usage-events telemetry for the curated items. Default: on. Use " +
164
- "--no-track-usage to run a curate that records nothing.",
165
- },
166
141
  },
167
142
  async run({ args }) {
168
143
  rejectRetiredSourceFlag();
@@ -173,7 +148,6 @@ export const curateCommand = defineJsonCommand({
173
148
  const limitParsed = parsePositiveIntFlag(args.limit ?? undefined);
174
149
  const limit = limitParsed && limitParsed > 0 ? limitParsed : 4;
175
150
  const source = parseSearchSource(args.from ?? "local");
176
- const skipLogging = args["track-usage"] === false;
177
151
  const outputMode = getOutputMode();
178
152
  const packBudget = parsePositiveIntFlag(args.pack ?? undefined, "--pack");
179
153
  const curated = await akmCurate({
@@ -181,7 +155,6 @@ export const curateCommand = defineJsonCommand({
181
155
  type,
182
156
  limit,
183
157
  source,
184
- skipLogging,
185
158
  eventSource: resolveUsageEventSource(),
186
159
  attributionProjection: outputMode.shape === "agent" ? "agent" : outputMode.detail,
187
160
  });
@@ -285,15 +258,6 @@ export const showCommand = defineJsonCommand({
285
258
  type: "string",
286
259
  description: "Exact context budget in characters. Requires --context lead; mutually exclusive with --max-tokens.",
287
260
  },
288
- // Declared as the POSITIVE name with `default: true` — see the
289
- // `track-usage` comment on `searchCommand` above for why a flag NAME
290
- // must never start with `no-`.
291
- "track-usage": {
292
- type: "boolean",
293
- default: true,
294
- description: "A successful show records usage-events telemetry, including the search-selection linkage when this " +
295
- "show follows a recent search. Default: on. Use --no-track-usage to run a show that records nothing.",
296
- },
297
261
  },
298
262
  async run({ args }) {
299
263
  // `[origin//]meta[:name]` targets the stash `.meta/` convention, which is
@@ -342,14 +306,12 @@ export const showCommand = defineJsonCommand({
342
306
  if (maxContextChars !== undefined && !Number.isSafeInteger(maxContextChars)) {
343
307
  throw new UsageError("Fragment context budget is too large.", "INVALID_FLAG_VALUE");
344
308
  }
345
- const skipLogging = args["track-usage"] === false;
346
309
  const result = await akmShowUnified({
347
310
  ref: args.ref,
348
311
  detail: showDetail,
349
312
  contextMode,
350
313
  maxContextChars,
351
314
  scope,
352
- skipLogging,
353
315
  eventSource: resolveUsageEventSource(),
354
316
  });
355
317
  output("show", result);
@@ -13,7 +13,7 @@ import { z } from "zod";
13
13
  // `config-types`, which type-derives from this barrel via
14
14
  // `typeof import("./config-schema")` — routing through config-types would mint
15
15
  // a config-schema ↔ config-types type cycle that collapses inference.
16
- import { HARNESS_AGENT_DISPATCH_IDS, harnessInferenceKeys, VALID_HARNESS_IDS, } from "../../../integrations/harnesses/ids.js";
16
+ import { HARNESS_AGENT_DISPATCH_IDS, VALID_HARNESS_IDS } from "../../../integrations/harnesses/ids.js";
17
17
  import { WORKFLOW_MAX_TIMEOUT_MS } from "../../../workflows/resource-limits.js";
18
18
  import { chatCompletionsEndpoint, ExtraParamsSchema, engineName, nonEmptyString, positiveInt, symbolicOrWarnApiKey, } from "./primitives.js";
19
19
  /**
@@ -105,20 +105,6 @@ const LlmEngineSchema = z
105
105
  });
106
106
  }
107
107
  });
108
- /**
109
- * The inference fields an agent engine may set: the ones its platform
110
- * translates (`harnesses/ids.ts`). An asset's or a caller's inference reaches
111
- * every engine and reports what the engine does not translate as a lowering
112
- * notice; an engine the operator configures for a platform names only what
113
- * that platform carries, so the rest is an error here.
114
- */
115
- const AGENT_INFERENCE_KEYS = [
116
- "temperature",
117
- "maxTokens",
118
- "contextLength",
119
- "enableThinking",
120
- "reasoningEffort",
121
- ];
122
108
  const AgentEngineSchema = z
123
109
  .object({
124
110
  kind: z.literal("agent"),
@@ -131,30 +117,26 @@ const AgentEngineSchema = z
131
117
  model: nonEmptyString.optional(),
132
118
  timeoutMs: timeoutMsField,
133
119
  llmEngine: engineName.optional(),
134
- temperature: z.number().finite().optional(),
135
- maxTokens: positiveInt.optional(),
136
- contextLength: positiveInt.optional(),
137
- enableThinking: z.boolean().optional(),
138
- reasoningEffort: nonEmptyString.optional(),
139
120
  })
140
121
  .passthrough()
141
122
  .superRefine((value, ctx) => {
142
- for (const key of ["provider", "endpoint", "apiKey", "apiKeyFile", "concurrency", "extraParams", "modelAliases"]) {
123
+ for (const key of [
124
+ "provider",
125
+ "endpoint",
126
+ "apiKey",
127
+ "apiKeyFile",
128
+ "temperature",
129
+ "maxTokens",
130
+ "concurrency",
131
+ "extraParams",
132
+ "contextLength",
133
+ "enableThinking",
134
+ "reasoningEffort",
135
+ "modelAliases",
136
+ ]) {
143
137
  if (key in value)
144
138
  ctx.addIssue({ code: z.ZodIssueCode.custom, path: [key], message: `${key} is not valid on an agent engine` });
145
139
  }
146
- const translated = harnessInferenceKeys(value.platform);
147
- for (const key of AGENT_INFERENCE_KEYS) {
148
- if (key in value && !translated.includes(key)) {
149
- ctx.addIssue({
150
- code: z.ZodIssueCode.custom,
151
- path: [key],
152
- message: `${key} is not valid on a ${value.platform} engine: ${translated.length > 0
153
- ? `the platform translates only ${translated.join(", ")}`
154
- : "the platform translates no inference fields"}`,
155
- });
156
- }
157
- }
158
140
  if (value.platform !== "opencode-sdk" && value.llmEngine !== undefined) {
159
141
  ctx.addIssue({
160
142
  code: z.ZodIssueCode.custom,